{"id":"5d47b191-e8c5-40e4-9b32-10be142e7453","entityType":"agent","slug":"clawhub-chpomob-adversarial-code-review-2","name":"Adversarial Code Review","canonicalUrl":"https://www.xpersona.co/agent/clawhub-chpomob-adversarial-code-review-2","canonicalPath":"/agent/clawhub-chpomob-adversarial-code-review-2","generatedAt":"2026-10-09T23:09:42.044Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-09T11:52:36.428Z","emptyReason":null},"description":"Multi-perspective adversarial code review with git-isolated worktrees. Two reviewers (Architect + Inspector), cross-validation, and synthesis report. The synthesis is the final arbiter — its verdict takes priority over individual reviewer outputs. Skill: Adversarial Code Review Owner: chpomob Summary: Multi-perspective adversarial code review with git-isolated worktrees. Two reviewers (Architect + Inspector), cross-validation, and synthesis report. The synthesis is the final arbiter — its verdict takes priority over individual reviewer outputs. Tags: latest:0.1.0 Version history: v0.1.0 | 2026-08-03T18:11:17.731Z | auto Initial release of adversarial-code-revi","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 2.7K downloads reported by the source. Last updated 10/9/2026.","installCommand":"clawhub skill install s17435m3chty5jmw4jhpkyhnb58brn8g:adversarial-code-review-2","sourceUrl":"https://clawhub.ai/chpomob/adversarial-code-review-2","homepage":"https://clawhub.ai/chpomob/skills/adversarial-code-review-2","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/chpomob/adversarial-code-review-2","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/chpomob/skills/adversarial-code-review-2","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":69,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Multi-perspective adversarial code review with git-isolated worktrees. Two reviewers (Architect + Inspector), cross-validation, and synthesis report. The synthe"},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-09T11:52:36.428Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T11:52:36.428Z","emptyReason":null},"stars":null,"forks":null,"downloads":2736,"packageName":null,"latestVersion":"0.1.0","tractionLabel":"2.7K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T11:52:36.428Z","emptyReason":null},"lastUpdatedAt":"2026-10-09T11:52:36.428Z","lastCrawledAt":"2026-10-09T11:52:36.428Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-10T11:52:36.428Z","lastVerifiedAt":null,"highlights":[{"version":"0.1.0","createdAt":"2026-08-03T18:11:17.731Z","changelog":"Initial release of adversarial-code-review v0.1.0. - Introduces a multi-perspective adversarial review workflow using two independent personas (Architect + Inspector) with cross-review and synthesis steps. - Supports reviewing diffs, files, directories, or entire projects with flexible command overrides for each reviewer phase. - Provides comprehensive configuration via CLI options and environment variables, allowing custom model commands per role. - Outputs findings and consolidated reports in both JSON and Markdown/HTML formats for easy consumption by humans and automation. - Enforces model diversity between Architect and Inspector reviewers and explains preferred model pairing strategies.","fileCount":41,"zipByteSize":95819}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s17435m3chty5jmw4jhpkyhnb58brn8g:adversarial-code-review-2","setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-chpomob-adversarial-code-review-2/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-chpomob-adversarial-code-review-2/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-chpomob-adversarial-code-review-2/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-chpomob-adversarial-code-review-2/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-chpomob-adversarial-code-review-2/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-chpomob-adversarial-code-review-2/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-09T23:09:42.043Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-chpomob-adversarial-code-review-2/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-chpomob-adversarial-code-review-2/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-chpomob-adversarial-code-review-2/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-chpomob-adversarial-code-review-2/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-09T11:52:36.428Z","emptyReason":null},"readme":"Skill: Adversarial Code Review\n\nOwner: chpomob\n\nSummary: Multi-perspective adversarial code review with git-isolated worktrees. Two reviewers (Architect + Inspector), cross-validation, and synthesis report. The synthesis is the final arbiter — its verdict takes priority over individual reviewer outputs.\n\nTags: latest:0.1.0\n\nVersion history:\n\nv0.1.0 | 2026-08-03T18:11:17.731Z | auto\n\nInitial release of adversarial-code-review v0.1.0.\n\n- Introduces a multi-perspective adversarial review workflow using two independent personas (Architect + Inspector) with cross-review and synthesis steps.\n- Supports reviewing diffs, files, directories, or entire projects with flexible command overrides for each reviewer phase.\n- Provides comprehensive configuration via CLI options and environment variables, allowing custom model commands per role.\n- Outputs findings and consolidated reports in both JSON and Markdown/HTML formats for easy consumption by humans and automation.\n- Enforces model diversity between Architect and Inspector reviewers and explains preferred model pairing strategies.\n\nArchive index:\n\nArchive v0.1.0: 41 files, 95819 bytes\n\nFiles: .gitignore (179b), conftest.py (279b), LICENSE (665b), README.md (2122b), references (0b), references/ai-quota-apis.md (5954b), references/api-input-limits.md (2307b), references/codex-claude-hardware-review.md (5917b), references/cross-model-diversity.md (1924b), references/cross-review-flags.md (1835b), references/debugging-review-pipeline.md (4037b), references/gemini-quota-research.md (3363b), references/github-push-protection-bypass.md (6202b), references/glm5-adversarial-perf.md (1681b), references/manual-recovery-procedure.md (1983b), references/model-pairing-diversity.md (2226b), references/monitoring-background-reviews.md (2726b), references/multi-model-review-strategy.md (1436b), references/persona-hardware-bias.md (1649b), references/phase-challenge-prompt-reduction.md (2489b), references/post-review-fix-planning.md (4738b), references/pre-publication-cleanup.md (4844b), references/pre-publication-review-checklist.md (3059b), references/pre-publication-review.md (7304b), references/role-swap-complementarity.md (1320b), references/role-swapped-reviews.md (1824b), references/tokio-mutex-guard-lifetime.md (3080b), scripts (0b), scripts/adversarial_review.py (64799b), scripts/check-ai-quota.py (20507b), scripts/install.sh (1975b), skill-card.md (2738b), SKILL.md (14667b), tests (0b), tests/test_contract_gate.py (5133b), tests/test_diff_git.py (12099b), tests/test_optional_modes.py (10178b), tests/test_quota_cli.py (10369b), tests/test_review_flags.py (33303b), tests/test_review.py (2916b), _meta.json (144b)\n\nFile v0.1.0:SKILL.md\n\n---\nname: adversarial-code-review\ndescription: \"Multi-perspective adversarial code review with git-isolated worktrees. Two reviewers (Architect + Inspector), cross-validation, and synthesis report. The synthesis is the final arbiter — its verdict takes priority over individual reviewer outputs.\"\ntags: [adversarial, code-review, multi-model, parallel, review-only, persona, git]\nversion: 1.10.0\nlicense: 0BSD\n---\n\n# adversarial-code-review\n\nMulti-perspective adversarial review of a diff or codebase. Two independent\nreviewers (**Architect** + **Inspector**) run concurrently and each produce JSON\nfindings, two **cross-review** passes (A reviews B's findings, B reviews A's\nfindings) pressure-test them, and a **synthesis** rapporteur collapses everything\ninto a single ranked report.\n\nThe review engine, subprocess runner, and personas live in the sibling\n`adversarial-common` skill — this skill only wires the review flow and the\nsource-gathering modes.\n\n## Installation\n\nRequires the `adversarial-common` sibling repo (shared engine). One-line install:\n\ncurl -fsSL https://raw.githubusercontent.com/chpomob/adversarial-code-review/main/scripts/install.sh | bash\n\nor, from an existing checkout:\n\nbash scripts/install.sh\n\nBoth place adversarial-code-review and adversarial-common side by side under `~/.hermes/skills` (override the target with `$1` or `$HERMES_HOME`).\n\n## When to use\n\n- Before merging a feature branch (`--diff-git`).\n- On a standalone patch file (`--diff`).\n- On a whole directory or single file (`--dir`, `--file`).\n- On an existing project in place (`--project-dir`).\n\n## Usage\n\n```bash\npython3 scripts/adversarial_review.py <source> [options]\n```\n\nThe reviewer command defaults to the `claude-tmux` wrapper (no model pinned —\nthe CLI picks its own best). Override per-run with `--review-cmd` or persistently\nwith `$ACR_REVIEW_CMD`.\n\n### Sources (mutually exclusive)\n\n| Flag | Argument | Reviews |\n|------|----------|---------|\n| `--diff-git` | — | `<base>..HEAD` inside an isolated git worktree (dirty tree auto-stashed) |\n| `--diff` | `FILE` | a unified-diff file |\n| `--dir` | `DIR` | every file under a directory |\n| `--file` | `FILE` | a single file |\n| `--project-dir` | `DIR` | an existing project directory in place |\n\n### Options\n\n| Flag | Default | Purpose |\n|------|---------|---------|\n| `--a-cmd` | `--review-cmd` (or `$ACR_A_CMD`) | Architect model command (overrides `--review-cmd`) |\n| `--b-cmd` | `--review-cmd` (or `$ACR_B_CMD`) | Inspector model command (overrides `--review-cmd`) |\n| `--cross-a-cmd` | `--a-cmd` (or `$ACR_CROSS_A_CMD`) | Cross-review A model — Architect reviews Inspector's findings |\n| `--cross-b-cmd` | `--b-cmd` (or `$ACR_CROSS_B_CMD`) | Cross-review B model — Inspector reviews Architect's findings |\n| `--synth-cmd` | `--review-cmd` (or `$ACR_SYNTH_CMD`) | Synthesis model command |\n| `--base` | `$ACR_BASE`, then `main`, then `master` | base ref for `--diff-git` (tried in that order) |\n| `--feature` | current branch name | slug used for the worktree path `/tmp/review-<feature>-<N>` |\n| `--allow-fallback` | off | on `--diff-git` worktree failure, review the live workdir instead of exiting 2 |\n| `--out` | `.adversarial-review` | artifact directory |\n| `--review-cmd` | `$ACR_REVIEW_CMD`, then the claude wrapper | CLI that runs every reviewer pass (fallback for per-role flags) |\n| `--delegated` | off | orchestrator/worker pre-review for high-complexity inputs |\n| `--orchestrator-cmd` | `--synth-cmd` | delegation/decomposition model command |\n| `--worker-cmd` | `--b-cmd` | delegated worker model command |\n| `--max-agents` | `6` | cap parallel and delegated fan-out |\n| `--show-costs` | off | print per-model token/cost breakdown to stderr |\n| `--html` | off | write a self-contained `report.html` |\n| `--timeout` | `600` | per-phase timeout (seconds) |\n\n**Env vars:** `ACR_A_CMD`, `ACR_B_CMD`, `ACR_CROSS_A_CMD`, `ACR_CROSS_B_CMD`,\n`ACR_SYNTH_CMD`, `ACR_ORCHESTRATOR_CMD`, `ACR_WORKER_CMD` — each falls back to the\nresolved `--review-cmd` (or its env var `ACR_REVIEW_CMD`), except\n`ACR_CROSS_A_CMD` which falls back to `ACR_A_CMD` and `ACR_CROSS_B_CMD` which\nfalls back to `ACR_B_CMD`.\n\n### Example: review a single file with Codex Architect + Claude Inspector\n\n```bash\npython3 scripts/adversarial_review.py \\\n  --file /path/to/target.py \\\n  --a-cmd \"codex exec --skip-git-repo-check --sandbox read-only\" \\\n  --b-cmd \"python3 /path/to/claude-tmux.py --timeout 600 --hard-timeout 1200 --cwd /path/to/repo\" \\\n  --synth-cmd \"python3 /path/to/claude-tmux.py --timeout 600 --hard-timeout 1200 --cwd /path/to/repo\" \\\n  --out /tmp/acr-review\n```\n\n### Example: review the current branch against `main`\n\n```bash\npython3 scripts/adversarial_review.py --diff-git --base main --out .adversarial-review\n```\n\n### Example: review a full project directory — Codex + Claude with mutual cross-review\n\n```bash\npython3 scripts/adversarial_review.py \\\n  --project-dir /path/to/repo \\\n  --a-cmd \"codex exec --skip-git-repo-check --sandbox read-only\" \\\n  --b-cmd \"python3 /path/to/claude-tmux.py --timeout 600 --hard-timeout 1200 --cwd /path/to/repo\" \\\n  --cross-a-cmd \"codex exec --skip-git-repo-check --sandbox read-only\" \\\n  --cross-b-cmd \"python3 /path/to/claude-tmux.py --timeout 600 --hard-timeout 1200 --cwd /path/to/repo\" \\\n  --synth-cmd \"python3 /path/to/claude-tmux.py --timeout 600 --hard-timeout 1200 --cwd /path/to/repo\" \\\n  --out .adversarial-review --html --show-costs\n```\n\nWhen `--cross-a-cmd` and `--cross-b-cmd` are omitted, they default to `--a-cmd`\nand `--b-cmd` respectively. The targets remain symmetric: cross-review A uses\nthe Architect command to review the Inspector's findings, and cross-review B\nuses the Inspector command to review the Architect's findings. Override the\ncross commands only when those passes need different providers or settings.\n\n## Output\n\nArtifacts land under `--out` (default `.adversarial-review`):\n\n- `01_architect.txt`, `02_inspector.txt` — raw reviewer JSON\n- `03_cross_1.txt` — cross-review: A reviews B's (Inspector) findings\n- `04_cross_2.txt` — cross-review: B reviews A's (Architect) findings\n- `05_synthesis.txt` + `review.md` — the consolidated ranked report\n- `final.json` — machine-readable verdict, complexity, parallel mode, and cost ledger for CI/cron\n- `report.html` — optional self-contained report produced by `--html`\n\n`final.json` shape:\n\n```json\n{\n  \"verdict\": \"APPROVE|REQUEST_CHANGES|REJECT\",\n  \"summary\": \"first lines of the synthesis report\",\n  \"findings\": {\"blocker\": 1, \"major\": 2, \"minor\": 4},\n  \"report\": \".adversarial-review/review.md\",\n  \"source_diff\": true\n}\n```\n\n## Exit codes\n\n| Code | Meaning |\n|------|---------|\n| `0` | review complete |\n| `1` | pipeline / infrastructure failure (reviewer CLI crashed, git error) |\n| `2` | nothing to review or review setup cannot proceed (no files, missing base, `--diff-git` setup failure) |\n| `5` | `EXIT_CONTEXT_BLOCKED`: the preflight context gate rejected empty or insufficient input |\n\n## Personas\n\nLoaded from `../adversarial-common/personas/` — the single source of truth,\nnow **100% generic** (no embedded/hardware-specific references):\n\n- `architect.md` — architecture, security, concurrency, design\n- `inspector.md` — bugs, edge cases, error handling, quality\n- `cross_review.md` — devil's advocate: VALIDATE / CHALLENGE / ADD\n- `synthesis.md` — rapporteur: cross-validated / consensus / disputed\n\n## Model pairing rules\n\n- **Architect and Inspector MUST be different models** (never the same model for\n  both roles). The cross-reviews each default to one of the two (cross-A → A,\n  cross-B → B), while their targets are the other reviewer's findings: A reviews\n  B and B reviews A. The default is therefore a symmetric mutual cross-review.\n- **Never pin a specific Claude model** (`--model sonnet`, `--model best`, etc.)\n  unless the user explicitly asks for one — let the claude-tmux wrapper use its\n  default.\n- **Preferred pairing:** Codex (Architect) + Claude (Inspector + cross-B +\n  Synthesis). Codex does the structural/design analysis; Claude produces reliable\n  JSON output in the exact schema the pipeline expects.\n- **Alternative for Inspector:** GLM-5.2 — works but may output different JSON\n  keys (`category` instead of `file`, `issue` instead of `summary`).\n  See GLM-5.2 pitfall below.\n- **Synthesis should use the same model as Inspector** to avoid schema conflicts.\n\n## Pitfalls\n\n- **GLM-5.2 inspector may output a different JSON schema than expected in `--file` mode.** The pipeline expects findings with keys `{id, severity, file, line, summary, evidence}` plus a top-level `verdict`. GLM-5.2 may write prose with different keys like `{id, severity, category, location, issue, fix}` — a structural schema mismatch that `strip_json_wrapper` cannot fix. **Symptom:** `02_inspector.txt` exists but `Phase 'inspector' failed (exit 1)` with `invalid reviewer JSON: expected findings with id, severity, file, line, summary, and evidence`. **Diagnosis:** check `02_inspector.txt` — if the JSON keys don't match the pipeline schema, it's a schema mismatch, not a formatting issue. **Fix:** either (a) add the missing keys to the persona prompt in `personas/inspector.md`, or (b) switch the inspector to a model that reliably outputs the exact schema (Codex works; DeepSeek V4 Pro usually works). Validated 2026-07-14 on claude-tmux.py review.\n\n- **`_valid_line()` now accepts free-form string markers, not just integers.** Models sometimes emit non-numeric line markers like `\"(review request)\"` or `\"(global)\"` for findings that don't map to a specific line. Previously `_valid_line()` required `isinstance(line, int) or line.isdigit()`, which rejected these strings and caused the entire phase to fail with `invalid reviewer JSON`. **Fixed 2026-07-15:** `_valid_line()` now returns `True` for any non-empty string, preserving the original intent (integer preferred) while tolerating model-generated location markers. Validation still rejects empty strings and `None`. See `git log -1 -- scripts/adversarial_review.py` for the commit change.\n\n- **Full-project reviews (`--project-dir`, `--dir`) exceed the foreground timeout cap.** The\n  5-phase pipeline (Architect + Inspector + 2 cross-reviews + Synthesis) on a multi-file\n  codebase takes 5–30 minutes depending on model speed and file count. On Hermes CLI, the\n  foreground terminal timeout caps at 600s. **Always run `--project-dir` or `--dir` reviews\n  in background mode with `notify_on_complete=true`.** See example above.\n\n- **Synthesis phase times out when Claude quota is exhausted.**\n\n- **Cross-review is symmetric even when the cross command flags are omitted.**\n  Cross-review 1 runs the Architect command on the Inspector's findings;\n  cross-review 2 runs the Inspector command on the Architect's findings and\n  receives round 1 as additional context. The flags select commands, not review\n  targets.\n\n- **The `claude-tmux` wrapper rejects `--yolo`.** Do not add that option to\n  reviewer, cross-review, or synthesis commands.\n\n- **`~` in `--a-cmd`/`--b-cmd`/`--synth-cmd` mid-command breaks `resolve_role_cmd`.** `providers.resolve_role_cmd()` only calls `os.path.expanduser()` when the entire command starts with `~`. A command like `python3 ~/.hermes/skills/...` (tilde mid-string) never gets expanded, so the subprocess runner receives a literal `~` and fails with `Command not found`. **Fix (applied 2026-07-14):** split the command per-token with `shlex.split()`, expand each token, and re-join with `shlex.join()` before returning. This ensures `~` is resolved regardless of position in the command string. The fix lives in `adversarial_common/adversarial_common/providers.py:resolve_role_cmd`.\n\n- **Pre-publication reviews need a cleanup sweep, not just code defects.** Before publishing\n  any Hermes skill, run the full checklist in `references/pre-publication-cleanup.md`:\n  privacy scan, tracking audit (French files, pipeline artifacts, backup copies, personal\n  notes, OAuth bypass docs), .gitignore hygiene, and SKILL.md reference de-dangling.\n  The adversarial review finds code defects but does NOT check for leaked config,\n  language-mismatched content, or missing metadata — the orchestrator must run those\n  separately. **Validated 2026-07-16:** adversarial-code-loop had 9 French-language files\n  and 52 personal workflow references committed; adversarial-plan had pipeline artifacts\n  from 2 separate loop runs. All were git rm --cached + push-removed.\n\n- **Personas historically contained hardcoded hardware references (ESP32-S3, CC1101 at 433 MHz, BLE) that biased reviews of pure-software projects.** This was fixed 2026-07-17: all 4 persona files in `../adversarial-common/personas/` were rewritten to be generic. The old `architect.md` asked about DSP on ESP32-S3, noise floor, antenna gain, IRAM usage; the old `inspector.md` asked about CC1101 RSSI quantization, SPI bus speed, and BLE spectral scans. If you encounter any remaining hardware-specific language in the personas, patch `../adversarial-common/personas/<file>.md` to remove it.\n\n- **`--diff-git` needs git ≥ 2.5** (worktree support). `gitops.ensure_git_available()`\n  guards git presence; older hosts should use `--diff` or `--project-dir`.\n- **Worktrees are created under `/tmp/review-<feature>-<N>`** and force-removed in a\n  `try/finally`, even when the applied patch leaves them dirty. A crash mid-review\n  can leave one behind — `git worktree prune` cleans stale metadata.\n- **A dirty working tree is auto-stashed and restored.** If `git stash pop` hits a\n  conflict (rare — the review does not touch the main workdir), the stash is kept and\n  a warning is printed; resolve and `git stash pop` manually.\n- **Base resolution is a fallback chain**, not strict: `--base` that does not resolve\n  keeps trying `$ACR_BASE` → `main` → `master`. Set `ACR_BASE` in CI to make the base\n  explicit and stable.\n- **An empty or insufficient diff exits 5 (`EXIT_CONTEXT_BLOCKED`)**, not 0 —\n  configure CI to handle a blocked preflight explicitly.\n- **Worktree creation failure exits 2 by default** (no silent fallback to the live\n  workdir, which could review the wrong tree). Pass `--allow-fallback` to instead\n  review the current working directory with a prominent stderr warning.\n- **`--diff-git` never moves the main workdir's branch** — the worktree is a separate\n  checkout at the merge-base. The original branch is restored defensively in cleanup.\n- The reviewer CLIs are invoked through `adversarial_common.runner.run_cli` (temp-file\n  IO, `start_new_session`, killpg on timeout) — a hung sandbox grandchild cannot\n  deadlock the pipeline.\n\nFile v0.1.0:README.md\n\n# adversarial-code-review\n\nMulti-perspective adversarial code review with git-isolated worktrees. Two independent reviewers (Architect + Inspector) each produce JSON findings, two cross-review passes pressure-test them, and a synthesis rapporteur collapses everything into a single ranked report.\n\nFor Hermes Agent, Claude Code, Codex, or any LLM CLI.\n\n## How it works\n\n```\nARCHITECT ──→ reviews code (architecture, security, concurrency)\nINSPECTOR ──→ reviews code (bugs, edge cases, error handling)\nCROSS_1 ────→ challenges INSPECTOR findings with ARCHITECT perspective\nCROSS_2 ────→ challenges ARCHITECT findings with INSPECTOR perspective\nSYNTHESIS ──→ ranks, cross-validates, produces final report\n```\n\n## Comparison\n\n| Feature | adversarial-code-review | adverse (addyosmani) | alecnielsen/adversarial-review | agent-review-panel |\n|---------|------------------------|---------------------|-------------------------------|-------------------|\n| Cross-model debate | ✅ Architect↔Inspector | ❌ Single-reviewer | ❌ Single-round | ✅ 4-6 panel |\n| Git worktree isolation | ✅ | ❌ | ❌ | ❌ |\n| Cross-review rounds | ✅ 2 rounds of devil's advocate | ❌ | ❌ | ❌ |\n| JSON findings with schema | ✅ | ❌ | ❌ | ❌ |\n| --project-dir mode | ✅ Review existing codebase | ❌ | ❌ | ❌ |\n\n## Quick start\n\n```bash\n# Review changes on a branch\npython3 scripts/adversarial_review.py --diff-git\n\n# Review a whole project directory\npython3 scripts/adversarial_review.py --project-dir /path/to/project \\\n  --a-cmd \"claude-tmux --model best\" \\\n  --b-cmd \"codex exec -C /path/to/project\"\n```\n\n## Output\n\nArtifacts land in `--out` (default `.adversarial-review`):\n\n- `final.json` — machine-readable verdict (`APPROVE|REQUEST_CHANGES|REJECT`)\n- `review.md` — ranked report with per-finding evidence\n- `01_architect.txt` … `05_synthesis.txt` — per-phase raw output\n\n## Dependencies\n\n- Python ≥ 3.11\n- Git ≥ 2.5\n- Two LLM CLIs (one for architect, one for inspector)\n\nUses `adversarial-common` as the shared engine.\n\n## License\n\n0BSD — see [LICENSE](LICENSE).\n\nFile v0.1.0:_meta.json\n\n{\n  \"ownerId\": \"kn7e26az9x7m8bgwfwg90q1wkh8bsqw0\",\n  \"slug\": \"adversarial-code-review-2\",\n  \"version\": \"0.1.0\",\n  \"publishedAt\": 1785780677731\n}\n\nFile v0.1.0:references/ai-quota-apis.md\n\n# AI CLI Quota APIs — Direct programmatic access\n\n**Updated 2026-07-31** — Standalone CLI startup and explicit endpoint inventory.\nAll five providers share the optional external adapter\n`~/.hermes/plugins/hermes-quota-status/quota_api.py`.\n\n## Architecture\n\n```\nquota_api.py  ←  shared module (token reading + API calls)\n   ├── check-ai-quota.py  ←  CLI script (human + JSON output)\n   └── hermes-quota-status/__init__.py  ←  Hermes TUI statusbar plugin\n```\n\nThe CLI imports the adapter lazily. Importing `check-ai-quota.py` and running\n`--help` therefore require only the Python standard library. A quota check still\nrequires the `hermes-quota-status` plugin; if it is absent, the CLI reports a\nper-provider error in its normal human or JSON output without a traceback.\n\nNo tmux scraping or URL-embedded API keys are used. Each direct request reveals\nthe caller's IP address, request timing, and association with the authenticated\naccount to the target provider. Credentials are sent in headers and are never\nintentionally printed. The provider-specific caveats below are additional to\nthat baseline disclosure.\n\n## Claude Code (Pro subscription)\n\n- **Token**: `~/.claude/.credentials.json` → `claudeAiOauth.accessToken`\n- **Endpoint**: `https://api.anthropic.com/api/oauth/usage`\n- **Status**: First-party Anthropic endpoint, but undocumented and\n  community-discovered; it is not a supported public API contract and may change.\n- **Auth**: `Authorization: Bearer <token>`\n- **Privacy**: Sends the Claude OAuth credential to Anthropic and requests\n  subscription utilization and reset times.\n- **Response**:\n  ```json\n  {\n    \"five_hour\": {\"utilization\": 27.0, \"resets_at\": \"2026-06-12T18:30:00Z\"},\n    \"seven_day\": {\"utilization\": 10.0, \"resets_at\": \"2026-06-19T09:00:00Z\"}\n  }\n  ```\n- **Note**: Claude returns utilization as percentages (0-100). The old code had a\n  scale=\"fraction\" bug that inflated sub-1% values to 80%. Fixed 2026-06-12.\n\n## Codex (ChatGPT Plus subscription)\n\n- **Token**: `~/.codex/auth.json` → `tokens.access_token`\n- **Endpoint**: `https://chatgpt.com/backend-api/wham/usage`\n- **Status**: First-party/official ChatGPT service endpoint used for Codex\n  account usage, but not a documented public developer API contract.\n- **Auth**: `Authorization: Bearer <token>`\n- **Privacy**: Sends the ChatGPT OAuth credential to OpenAI and requests account\n  rate-limit utilization and reset times.\n- **Response**:\n  ```json\n  {\n    \"rate_limit\": {\n      \"primary_window\": {\"used_percent\": 11, \"reset_at\": 1779762941},\n      \"secondary_window\": {\"used_percent\": 4, \"reset_at\": 1780313088}\n    }\n  }\n  ```\n\n## Gemini / agy (Google AI Studio)\n\n- **Key**: `GOOGLE_API_KEY` env var or `~/.hermes/.env` → `GOOGLE_API_KEY=...`\n- **Endpoints**:\n  - `https://generativelanguage.googleapis.com/v1beta/models`\n  - `https://generativelanguage.googleapis.com/v1beta/models/gemini-2.5-flash:generateContent`\n- **Status**: Both are official Google Generative Language API endpoints. They\n  are not quota-reporting endpoints; `models.list` validates the key and\n  `generateContent` is used only as a live availability probe.\n- **Auth**: `x-goog-api-key: <key>` (header, NOT URL query string)\n- **Privacy**: Sends the API key and a minimal prompt (`ok`) to Google. The probe\n  processes content and consumes a small amount of model quota and may incur cost.\n- **Quota**: Google does NOT expose a public quota API. We work around this with:\n  1. `models.list` to validate the key (step 1)\n  2. Minimal `generateContent(\"ok\")` probe to HTTP 200 vs 429 (step 2)\n- **Rate limits** (free tier, published):\n  - ~1500 requests/day per project\n  - ~60 requests/minute per project\n  - Resets at midnight Pacific time\n  - Applied per project, not per API key\n- **Probe response**:\n  ```json\n  {\n    \"key_valid\": true,\n    \"available_models\": [\"gemini-2.5-flash\", ...],\n    \"model_count\": 28,\n    \"probe\": {\"status\": 200, \"rate_limit_remaining\": null}\n  }\n  ```\n\n## GLM / Z.AI coding plan\n\n- **Key**: `GLM_API_KEY` env var, falling back to `ZHIPU_API_KEY`\n- **Endpoints** (tried in order):\n  - `https://api.z.ai/api/monitor/usage/quota/limit`\n  - `https://open.bigmodel.cn/api/monitor/usage/quota/limit`\n- **Status**: Undocumented, community-derived coding-plan monitoring endpoints;\n  neither is a supported public quota API contract.\n- **Auth**: `Authorization: <key>`\n- **Privacy**: Sends the API key and requests account quota data. If the primary\n  host fails, the same credential is sent to the Zhipu fallback host, so one\n  check can contact both domains.\n\n## DeepSeek (pay-as-you-go)\n\n- **Key**: `DEEPSEEK_API_KEY` env var\n- **Endpoint**: `https://api.deepseek.com/user/balance`\n- **Status**: Official, documented DeepSeek user-balance API.\n- **Auth**: `Authorization: Bearer <key>`\n- **Privacy**: Sends the API key to DeepSeek and requests all account balance\n  entries; the CLI selects and displays the USD balance.\n\n## Script\n\n`scripts/check-ai-quota.py` — CLI checker, human-readable + JSON output.\nUsage: `python3 scripts/check-ai-quota.py [--claude] [--codex] [--gemini] [--glm] [--deepseek] [--all] [--json]`\n\n## When to check\n\nBefore launching an expensive adversarial pipeline (REVIEW A + CROSS A→B + SYNTHESIS\n= 4+ LLM calls), always run the quota checker. If Claude 5h utilization ≥ 80%,\nconsider swapping roles: use Codex for the review phases and Claude Sonnet for\ncross-review.\n\n## Pitfalls\n\n- Claude token expires after ~7h of inactivity. HTTP 401 → start a `claude` session\n  then `/exit` to refresh, then retry.\n- Gemini has NO real-time quota API. The probe detects if you're rate-limited now\n  (HTTP 429), but can't tell remaining capacity.\n- Codex endpoint is official but tied to ChatGPT OAuth lifecycle.\n- Claude, Codex, and GLM endpoints are not documented public API contracts and\n  can change without notice.\n- GLM fallback behavior may disclose the configured credential to both listed\n  hosts.\n\nFile v0.1.0:references/api-input-limits.md\n\n# API Input Size Limits for Adversarial Review\n\nWhen using `--project-dir` or `--dir` modes, the review script sends ALL source\nfiles concatenated to the reviewer's stdin. API-based reviewers enforce strict\ninput size limits that cause silent failures if exceeded.\n\n## Limits by provider\n\n| Provider | Max input chars | Limit type | Failure mode |\n|----------|----------------|------------|-------------|\n| Codex (OpenAI) | 1,048,576 (1 MB) | Hard API limit | `turn/start failed: Input exceeds the maximum length` — exit 1, empty stdout, no visible error in truncated stderr |\n| Claude (Anthropic) | ~200K tokens (~800K chars) | Soft per-model | Model refuses with \"input too long\" |\n| Claude-tmux | Depends on model | Varies | Usually works up to ~2M chars with extended thinking |\n\nCodex is the most restrictive: 1 MB of input characters. A project with 85 source\nfiles averages ~600K chars (safe). Adding test files, build artifacts, or library\ndependencies pushes it over the limit.\n\n## Debugging checklist when Architect phase exits 1\n\n1. Check `01_architect.txt` in the output artifact directory — if it's 0 bytes,\n   Codex received no stdin or the input was rejected\n2. Look for `input_exceeds_maximum_length` or `input_too_large` in stderr\n   (it may be buried deep in the output — grep for it)\n3. Run `python3 -c \"\nimport os; SKIP={'.git','.venv','__pycache__','node_modules','.pytest_cache','.pio','build','target','test','unity'}; PREFIX={'.adversarial','.omnisense-'}; out=[]\nfor dp,dirs,files in os.walk('.'):\n  dirs[:]=[d for d in dirs if d not in SKIP and not any(d.startswith(p) for p in PREFIX)]\n  for n in files:\n    if n.startswith('.'): continue\n    out.append(os.path.relpath(os.path.join(dp,n),'.'))\n    if len(out)>=200: break\n  if len(out)>=200: break\ntotal=sum(os.path.getsize(f) for f in out)\nprint(f'{len(out)} files, {total:,} chars')\n\"` from the project root to measure the input size\n\n## Fixes\n\n1. Add `test`, `unity` (test framework dirs) to `_SKIP_DIRS`\n2. Add `.pio`, `build`, `target` (build artifact dirs) to `_SKIP_DIRS`  \n3. Ensure `_SKIP_DIR_PREFIX` catches `.adversarial-*` and `.omnisense-*` dot-dirs\n4. Verify dot-prefixed individual files (`.omnisense-*.md` specs) are filtered\n\nAim for ≤ 700K chars to leave headroom for the persona text (~1.5K per role).\n\nFile v0.1.0:references/codex-claude-hardware-review.md\n\n# Codex GPT-5.6-Sol + Claude — Hardware/Embedded Review\n\nValidated pairing for adversarial review of firmware/embedded projects.\n\n## Roles\n\n| Role | Model | Command |\n|------|-------|---------|\n| Architect | Codex GPT-5.6-Sol (reasoning=medium; bump to high if shallow) | `codex exec -C <project_dir> --skip-git-repo-check --dangerously-bypass-approvals-and-sandbox -c model='gpt-5.6-sol' -c model_reasoning_effort='medium'` |\n| Inspector | Claude Fable 5 (tmux) | `python3 /path/to/claude-tmux.py --yolo --model best --timeout 600 --hard-timeout 1200` |\n| Cross-review | Claude Fable 5 (same cmd) | Same as Inspector |\n| Synthesis | Claude Fable 5 | Same as Inspector |\n\n## Approaches\n\n### Approach A: `codex exec -C <dir>` (recommended for large projects)\n\nFor Codex as the reviewer, **do NOT use `adversarial_review.py --project-dir`** when the project has many files. The script concatenates ALL source into stdin, which:\n\n- Exceeds Codex's 1 MB input limit for projects > ~85 files\n- Floods the model with irrelevant test/build files\n- Prevents Codex from exploring files selectively\n\nInstead, pass a focused prompt via inline argument and let Codex explore with `cat`/`rg`/`sed`.\n**Prefer inline prompt over pipe** to avoid PTY buffer/deadlock issues:\n\n```bash\ncodex exec \\\n  -C /path/to/project \\\n  --skip-git-repo-check --dangerously-bypass-approvals-and-sandbox \\\n  -c model='gpt-5.6-sol' -c model_reasoning_effort='medium' \\\n  'Review the project. Explore src/ and lib/ files yourself via cat/rg/sed.\n   Focus on algorithmic correctness and hardware realism.\n   Output JSON findings with id/severity/file/line/summary/evidence and verdict.'\n```\n\nThe inline prompt (~200 chars) is kept intentionally short — Codex reads the\ntask from the prompt and explores files on its own. If the prompt contains\napostrophes or complex quoting that break single-quote shell syntax, write it\nto a file and use `\"$(< /tmp/prompt.txt)\"`:\n\n```bash\ncodex exec ... \"$(< /tmp/prompt.txt)\" 'short fallback instruction'\n```\n\nThe prompt should include: task description, hardware context, JSON output format,\nand a directive to `cat`/`rg` files from `src/` and `lib/`. The `-C` flag sets the\ncontext directory so `exec` shell commands run there.\n\n### Approach B: `adversarial_review.py --project-dir` (for small projects)\n\nFor projects with <50 source files and no build artifacts:\n\n```bash\npython3 scripts/adversarial_review.py --project-dir /path --a-cmd \"codex ...\"\n```\n\nBefore launch, verify `_SKIP_DIRS` excludes `.pio`, `build`, `target`, `test`, `unity` and `_SKIP_DIR_PREFIX` covers `.adversarial-*` and project-specific dot-dirs. Check input size < 700K chars using the script's own `_list_tree`:\n\n```bash\npython3 -c \"exec(open('scripts/adversarial_review.py').read().split('if __name__')[0]); import os; f=_list_tree('.'); print(f'{len(f)} files, {sum(os.path.getsize(p) for p in f):,} chars')\"\n```\n\n## Strengths\n\n- **Codex (GPT-5.6-Sol)** excels at algorithmic correctness analysis: DSP math (Goertzel, Kalman, FFT), lock-free concurrency patterns, signal-processing pipeline design, and hardware/physics realism of detection thresholds.\n- **Claude (Fable 5)** excels at edge-case hunting, error recovery, and hardware physical limits (RSSI quantization, settling times, noise floors, SPI bus timing).\n\n## When to use\n\n- Embedded firmware with DSP/sensing algorithms (ESP32, CC1101, BLE)\n- Projects where hardware physics limits matter (RF sensing, radar-like processing)\n- Dual-core concurrency review (lock-free, SPSC, atomic memory ordering)\n\n## Pitfalls\n\n- **`_SKIP_DIRS` must exclude `.pio`, `build`, `target`, `test`, `unity`** for PlatformIO projects. Without these, `_list_tree()` descends into compiled library directories (383 MB+) and lists only build artifacts — the model never sees the source code. Run the file-count check above before launch.\n- **`_SKIP_DIR_PREFIX`** filters dot-prefixed dirs (`.adversarial-*`, `.omnisense-*`). These are not caught by the plain `_SKIP_DIRS` set.\n- **Dot-prefixed files at the project root** (`.omnisense-*.md` specs, `.adversarial-*.json`) must be filtered by `name.startswith(\".\")` — they live in non-skipped directories.\n- **The `_fail_phase` private-method bug:** older `adversarial_review.py` called `runner._fail_phase()` but `runner.py` exports `fail_phase()` (no underscore). If you see `module 'adversarial_common.runner' has no attribute '_fail_phase'`, replace `runner._fail_phase(` with `runner.fail_phase(` in line 271.\n- **Codex uses stdin as prompt** in Approach B. No inline `exec \"...\"` argument is passed; the persona + file listing are piped via `communicate(input=...)`. Ensure the persona starts with a clear instruction.\n- **`--project-dir` includes ALL non-skipped files.** For large projects, always verify the `_list_tree` output before assuming the model received relevant files.\n- **Timeouts:** GPT-5.6-Sol reasoning=high can take 2-10 min per response. Start with reasoning=medium (near-instant thinking, still thorough). Bump to high or xhigh only if the findings are shallow.\n- **GPT-5.6-Sol reasoning=high may appear hung** for 4+ minutes with no output — the model is silently reasoning before the next action. The process is not hung if the PID is still alive (check `process(action='poll')`). With reasoning=medium, the model produces intermediate tool calls (cat, rg) within seconds.\n\n## Hardware context to include in persona\n\nWhen reviewing RF sensing firmware (433 MHz CC1101, 2.4 GHz WiFi CSI, BLE spectral scan):\n\n- CC1101 RSSI: ~6-bit quantization, -110 dBm noise floor, +10 dBm TX max\n- ESP32-S3: dual-core, single-precision FPU only, SPI bus shared with SD card\n- 433 MHz through walls: 3-10 dB attenuation per wall, multipath complex\n- WiFi CSI: requires sustained traffic (>10 pps), gain lock via undocumented Espressif PHY symbols\n- BLE spectral scan: 40 channels vs 3-channel fallback (HCI controller dependency)\n\nFile v0.1.0:references/cross-model-diversity.md\n\n# Cross-Model Diversity in Adversarial Reviews\n\n## Finding: Different Model Pairings Find Completely Different Bugs\n\nValidated 2026-07-01 on **omnisense firmware** (ESP32, C/embedded, ~450 source files, 52 files in scope):\n\n| Review | Architect | Inspector | Total Findings | Unique |\n|--------|-----------|-----------|----------------|--------|\n| V1 | GLM-5.2 | Claude (tmux best) | 18 | 18 |\n| V2 | Claude (tmux best) | GLM-5.2 | 17 | 17 |\n| **Union** | | | **35** | **35** |\n\n**ZERO findings overlapped between V1 and V2.** Not a single finding was independently reported by both pairings.\n\n### Implications\n\n1. **Single-model reviews leave blind spots.** A review by one pairing (even adversarial with 2 different models) misses ~50% of the findings another pairing would find.\n2. **Inverting roles matters as much as changing models.** V1 had GLM as Architect + Claude as Inspector; V2 swapped them. The different role focuses (Architect looks at structure/security, Inspector at bugs/edge cases) combined with different model strengths produced completely orthogonal results.\n3. **For critical code, run ≥2 adversarial reviews with different model pairings.** The marginal cost is low (the pipeline is automated) and the return is high (2× the findings).\n4. **Cross-validation is a flawed confidence metric.** Zero findings were cross-validated (independently found by both reviewers) — not because any finding was wrong, but because different reviews find different things. Absence of cross-validation does NOT mean findings are weak.\n\n### When to Use Multiple Pairings\n\n- **Safety-critical firmware** (medical, automotive, aerospace): run 3+ pairings\n- **Security-sensitive code** (auth, crypto, network): run 2+ pairings with different Architect models\n- **Public-facing web apps**: 1 high-quality pairing (Codex DEV + Claude REVIEW) is usually sufficient\n- **Internal tools / quick fixes**: 1 pairing is fine\n\nFile v0.1.0:references/cross-review-flags.md\n\n# Cross-review flags — `--cross-a-cmd` and `--cross-b-cmd`\n\nAdded 2026-07-17 to enable mutual adversarial cross-validation where\nA reviews B's findings and B reviews A's findings.\n\n## Background\n\nOriginally the pipeline ran both cross-review passes with `--b-cmd` (the\nInspector's model), so Claude validated its own findings — a self-review\nwith no genuine adversarial pressure.\n\n## Solution\n\nTwo new flags, each defaulting to its corresponding review model:\n\n- `--cross-a-cmd` → defaults to `--a-cmd` (Architect reviews Inspector's work)\n- `--cross-b-cmd` → defaults to `--b-cmd` (Inspector reviews Architect's work)\n\nWhen both are omitted, the defaults already produce mutual cross-review:\nthe Architect command reviews the Inspector's findings, and the Inspector\ncommand reviews the Architect's findings. Set a cross flag only to override\nthe provider or settings for that pass; flags select commands, not targets.\n\n## Pipeline flow (with mutual cross-review)\n\n```\nPhase 1: Architect  (A cmd)  → produces findings\nPhase 2: Inspector  (B cmd)  → produces findings\nPhase 3: Cross-1    (cross-a-cmd) → A reviews B's findings\nPhase 4: Cross-2    (cross-b-cmd) → B reviews A's findings + sees Cross-1\nPhase 5: Synthesis  (synth-cmd) → consolidated report\n```\n\n## Validated on (2026-07-17)\n\nchatter-javier review: `--cross-a-cmd=\"codex ...\" --cross-b-cmd=\"claude-tmux ...\"`\nVerdict: REQUEST_CHANGES, 1 blocker + 11 major + 3 minor + 1 nit.\nBoth cross-review passes produced substantive VALIDATE/CHALLENGE/ADD findings\nthat strengthened the final report. Cross-2 (B reviews A) specifically caught\nDocker binding issues and end-to-end failure chains that A's initial review\nhad flagged at lower confidence.\n\n## Env vars\n\n- `ACR_CROSS_A_CMD` — overrides `--cross-a-cmd`\n- `ACR_CROSS_B_CMD` — overrides `--cross-b-cmd`\n\nFile v0.1.0:references/debugging-review-pipeline.md\n\n# Debugging the adversarial review pipeline\n\n## Quick diagnostic workflow\n\nWhen the review fails or hangs, isolate the problem systematically:\n\n```\n1. Check exit code and stderr\n   → Exit 1 = pipeline/infrastructure failure\n   → Exit 2 = nothing to review or git setup failure\n   → No output after 60+ seconds = phase is slow or hung\n\n2. Check artifact directory (--out)\n   → 01_architect.txt size 0 bytes = Architect phase never finished\n   → Missing artifact files = earlier phase failed\n\n3. Test each CLI/tool directly\n   echo \"Reply TEST_OK\" | python3 ~/.../claude-tmux.py --yolo --timeout 30\n   → If that works, the tool itself is fine\n   → If it hangs, check tmux: tmux list-sessions\n```\n\n## Common failure patterns\n\n### `FileNotFoundError` with `~` in the path\n\n```\npython3: can't open file '/home/user/project/~/.hermes/skills/script.py'\n```\n\n**Cause:** `runner.run_cli` uses `subprocess.Popen(argv, shell=False)`. A literal `~`\nin argv is a filesystem character, not a home-directory reference. With `cwd=project`,\nit resolves relative to the project directory instead of `$HOME`.\n\n**Fix:** The command must go through `providers.resolve_role_cmd()` which applies\nper-token `os.path.expanduser()` via `shlex.split` + `shlex.join`. Verify by\nprinting the resolved command string from `resolve_role_cmd`.\n\n**Verify:** `python3 -c \"import shlex, os; print(' '.join(os.path.expanduser(w) for w in shlex.split(cmd)))\"`\n\n### `AttributeError: module 'runner' has no attribute '_fail_phase'`\n\n```\nAttributeError: module 'adversarial_common.runner' has no attribute '_fail_phase'\n```\n\n**Cause:** The function is named `fail_phase` (no underscore prefix), but the caller\nuses `runner._fail_phase()` (with underscore). This is a naming mismatch introduced\nby a refactoring that renamed the public function but missed one call site.\n\n**Fix:** Change `runner._fail_phase(...)` to `runner.fail_phase(...)`.\n\n### Phase runs for minutes with no stdout output\n\n```\n$ cat /tmp/acr-review.log\nbash: ... (harmless shell warnings)\n(no further output for 5+ minutes)\n```\n\n**Cause:** The phase subprocess (Claude-tmux or Codex) sends output to a tmux pane or\ntemp file, not to the subprocess stdout. `run_cli` only reads stdout after the process\nexits. This is normal — no output ≠ hung.\n\n**Mitigating factors:**\n- The code text for `--project-dir` mode can be 300K+ chars (81 files), requiring\n  significant LLM processing time per phase (3-5 minutes with Claude).\n- The review runs 5 phases sequentially, so total time can be 20-30 minutes.\n\n**Monitor progress:**\n- Check `ls -la <out>/01_architect.txt` — non-zero size means Architect completed\n- Run `PYTHONUNBUFFERED=1` and redirect to a log file for eventual output\n- Use `notify_on_complete=true` with the background process terminal tool\n\n### Claude-tmux hangs inside the pipeline but works when piped directly\n\n```\n# Direct test works (fast)\necho \"TEST_OK\" | python3 claude-tmux.py --yolo --timeout 30\n\n# Pipeline invocation (via run_cli) hangs for minutes\n```\n\n**Check:** The pipeline feeds the entire 300K-char codebase as stdin to Claude. This\nprompt is large and Claude needs time to process it. A 30-second timeout is too short\nfor a 300K-char code review prompt — use `--timeout 900` minimum.\n\n**Also check:** Run the claude-tmux command from `resolve_role_cmd` output directly\nwith a sample of the actual stdin to reproduce the timing.\n\n## Phase-by-phase timing expectations (adversarial-code-loop, 350K chars, 81 files)\n\n| Phase | Tool | Expected time | Notes |\n|-------|------|---------------|-------|\n| 01_architect | Claude-tmux | 3-5 min | First read of the full codebase |\n| 02_inspector | Codex CLI | 2-4 min | Second perspective |\n| 03_cross_1 | Claude-tmux | 3-5 min | Architect command reviews Inspector findings |\n| 04_cross_2 | Codex CLI | 1-2 min | Inspector command reviews Architect findings; receives Cross 1 as context |\n| 05_synthesis | Claude-tmux | 2-4 min | Consolidates all prior phases |\n\nTotal: ~11-20 minutes for a codebase of this size.\n\nFile v0.1.0:references/gemini-quota-research.md\n\n# Gemini / agy Quota Research\n\n**Status: NO public API for real-time quota checking.** Discovered 2026-06-12 during an\nadversarial review of the quota-checking tooling (Codex Architect + agy Inspector).\n\n## The problem\n\nGoogle Gemini API (used by agy, the Antigravity CLI) has quota limits per project:\n- Free tier: ~1500 req/day, ~60 req/min\n- Pay-as-you-go (Tier 1+): higher but based on billing tier\n- Resets at midnight Pacific time\n\nBut there is **no public API endpoint** to query remaining quota programmatically.\nConfirmed by Google collaborator (ryanjsalva) in GitHub discussion #3096.\n\n## What does NOT work\n\n- `models.list` only validates the key, doesn't return quota info\n- `/stats` in Gemini CLI is session-level only\n- No HTTP headers carry remaining quota (unlike some other APIs)\n- No `generativelanguage.googleapis.com` endpoint returns usage stats\n\n## What COULD work (options, ordered by feasibility)\n\n### Option A — Probe-based (recommended, low effort)\nSend a minimal `generateContent` request to a cheap model (`gemini-2.5-flash` with\n`{\"contents\":[{\"parts\":[{\"text\":\"ok\"}]}]}`). If the response is 200 → quota available.\nIf 429 → quota exceeded. Track 429 rate over time for a rough estimate.\n\n```python\nimport urllib.request, json\nreq = urllib.request.Request(\n    \"https://generativelanguage.googleapis.com/v1beta/models/gemini-2.5-flash:generateContent\",\n    data=json.dumps({\"contents\": [{\"parts\": [{\"text\": \"ok\"}]}]}).encode(),\n    headers={\"Content-Type\": \"application/json\", \"x-goog-api-key\": API_KEY}\n)\ntry:\n    urllib.request.urlopen(req, timeout=10)\n    print(\"Quota available\")\nexcept urllib.error.HTTPError as e:\n    if e.code == 429:\n        print(\"Quota exceeded\")\n```\n\n### Option B — Reactive counter (medium effort)\nLog every 429 error from agy (or direct Gemini API calls) into a persistent counter\nfile. Display \"X 429s in last 24h\" as a quota-health proxy. Works for any CLI that\nwraps Gemini (agy, google-gemini CLI, etc.).\n\n```python\n# In a file like ~/.hermes/data/quota-429-log.jsonl\n{\"ts\": 1718200000, \"provider\": \"gemini\", \"model\": \"gemini-2.5-flash\", \"code\": 429}\n```\n\n### Option C — Google Cloud Monitoring (high effort, accurate)\nRequires:\n1. A GCP project with billing enabled AND the Gemini API linked\n2. `gcloud` CLI installed and authenticated\n3. Query `cloudaicompanion.googleapis.com/usage/response_count` metric\n\nThis gives real numbers but is overkill for most use cases. Only worth it if the\nuser already has GCP billing set up.\n\n## How to integrate (for check-ai-quota.py)\n\nAdd a `gemini_quota()` function that:\n1. Reads `GOOGLE_API_KEY` from env (check `os.environ.get()` first, then ~/.hermes/.env)\n2. Validates key via `models.list` (returns model availability)\n3. Sends probe to `gemini-2.5-flash:generateContent` (checks if 429)\n4. Optionally reads the reactive 429 counter\n\nJSON output shape:\n```json\n{\n  \"provider\": \"gemini\",\n  \"key_valid\": true,\n  \"models_available\": [\"gemini-2.5-flash\", \"gemini-3.1-pro-preview\", ...],\n  \"probe_status\": \"ok\",  // \"ok\" | \"rate_limited\" | \"error\"\n  \"recent_429s\": 0       // from counter file\n}\n```\n\n## Related files\n\n- `scripts/check-ai-quota.py` — the script that should get this function\n- `references/ai-quota-apis.md` — shared API reference doc\n- `~/.hermes/plugins/hermes-quota-status/__init__.py` — statusbar plugin (also needs Gemini)\n\nFile v0.1.0:references/github-push-protection-bypass.md\n\n> **Scope: this bypasses a GitHub security control. Only use it for a\n> confirmed false positive you can point to in the \"bypass vs fix\" table\n> below. Never use it to push a real secret — rotate and fix instead.**\n\n# GitHub Push Protection Bypass for False Positives\n\nWhen publishing reviewed code to GitHub, secret scanning's push protection\nmay reject the push for credentials that are **not actual secrets** (installed-app\nOAuth client secrets, public test API keys, demo tokens, etc.).\n\n## Diagnosis\n\nThe rejection message includes:\n\n```\nremote:       —— Google OAuth Client Secret ————————————————————————\nremote:        locations:\nremote:          - commit: <sha>\nremote:            path: gemini_cloudcode.py:31\nremote:\nremote:        (?) To push, remove secret from commit(s) or follow this URL\nremote:            to allow the secret.\nremote:            https://github.com/<owner>/<repo>/security/secret-scanning/unblock-secret/<PLACEHOLDER_ID>\n```\n\nThe `PLACEHOLDER_ID` in the URL is the key — it identifies the detected secret\nto the bypass API.\n\n## Per-value verification (required before bypass)\n\n**Category membership in the \"bypass vs. fix\" table below is not enough by\nitself.** The table says which *class* of credential is safe to bypass, but\na real, rotated secret can land in the flagged file under the same class\nlabel (e.g. someone pastes a live client secret into the same variable that\nnormally holds the public installed-app one). Before calling the bypass API,\ndiff the *exact* flagged value against a known-public constant sourced\n**independently of the flagged commit** for that credential class.\n\nExtract the exact value using the `path` and line number GitHub's rejection\nmessage cited (not a guessed filename or a grep for default-variable names —\neither of those can silently match a different assignment than the one\nflagged, letting a live secret elsewhere in the file, or under the same\nvariable name, pass unchecked):\n\n```bash\n# Use the exact path and line from the rejection's `path: <path>:<line>`:\ngit show <sha>:<path-from-rejection> | sed -n '<line-from-rejection>p'\n```\n\nThen compare it byte-for-byte against a known-public value from a source\n**independent of the flagged file/commit** — e.g. the vendor's published\nsource (upstream GitHub repo, the installed pip/npm package cache on disk,\nofficial docs) or a commit predating any point where the value could have\nbeen tampered with. Do **not** treat another occurrence of the same variable\nin the same working tree as independent verification — if the flagged commit\nis where a live value was pasted in, a second read of that same file/commit\nwill just confirm the tampered value against itself.\n\n```bash\ndiff <(echo \"<value flagged by GitHub>\") <(echo \"<value from independent source>\")\n```\n\nOnly bypass if that diff is empty. Any difference at all — a changed\ncharacter, a different length, a rotated timestamp — means this is not the\nknown-public value: stop, do not bypass, and rotate/remove the credential\ninstead.\n\n**If no independent known-public source can be located, treat the value as\nunverified: do not bypass.** `references/pre-publication-cleanup.md` §2 does\n**not** provide this baseline — it records only truncated placeholders\n(`_CLIENT_ID_DEFAULT = \"...\"`), not the vetted bytes, precisely so it can't be\nmistaken for ground truth here.\n\n## Bypass via REST API\n\nUse the `gh api` or `python3` with the user's GitHub token:\n\n```python\nimport json, urllib.request, os\n\ntoken = os.popen(\"gh auth token\").read().strip()\nheaders = {\n    \"Authorization\": f\"Bearer {token}\",\n    \"Accept\": \"application/json\",\n    \"Content-Type\": \"application/json\",\n}\n\n# One bypass per detected secret type. Replace PLACEHOLDER_ID from the URL.\ndata = {\n    \"secret_type\": \"google_oauth_client_id\",   # or google_oauth_client_secret\n    \"reason\": \"false_positive\",\n    \"placeholder_id\": \"PLACEHOLDER_ID\",\n}\n\nreq = urllib.request.Request(\n    \"https://api.github.com/repos/<owner>/<repo>/secret-scanning/push-protection-bypasses\",\n    data=json.dumps(data).encode(), headers=headers, method=\"POST\",\n)\nres = urllib.request.urlopen(req)\nprint(res.status, json.loads(res.read()))\n```\n\n**Required fields:**\n- `secret_type` — one of the secret types GitHub detected (found in the error\n  message heading, e.g. `google_oauth_client_secret`).\n- `reason` — must be one of: `false_positive`, `used_in_tests`, `will_fix_later`.\n  For installed-app credentials, always `false_positive`.\n- `placeholder_id` — the alphanumeric ID extracted from the unblock URL.\n\n**Response (200):**\n```json\n{\"reason\": \"false_positive\", \"expire_at\": \"2026-07-14T04:07:26.544-07:00\", \"token_type\": \"GOOGLE_OAUTH_CLIENT_ID\"}\n```\n\nThe bypass expires after ~24h. After creating the bypass, retry the push.\n\n## When to bypass vs. fix\n\n| Scenario | Action |\n|----------|--------|\n| Installed-app OAuth client secret (Google Cloud Code, etc.) | **Bypass** — these are public by design, not confidential |\n| Personal API keys, tokens, passwords | **Fix** — replace with env vars or remove from source |\n| Test credentials (obviously fake, like `sk-test`) | **Bypass** — use `used_in_tests` reason |\n| Demo/tutorial credentials published by the vendor | **Bypass** — use `false_positive` reason |\n| Production credentials | **Fix immediately** — rotate the credential, remove from git history |\n\n## Validated example (2026-07-13)\n\nProject: `hermes-quota-status` — `gemini_cloudcode.py` contained Google Cloud\nCode OAuth client ID/secret (installed-app flow, public by design).\n\n**Error:** Push rejected for `google_oauth_client_id` and `google_oauth_client_secret`.\n\n**Solution:** Two API calls (one per secret type) with `reason: false_positive`.\nAfter bypass, push succeeded. The values were already behind `os.environ.get()`\nwith defaults — the hardcoded fallbacks remain for out-of-the-box usability.\n\n**Alternative (if you prefer no bypass):** Remove the hardcoded values entirely\nand require `GOOGLE_CLIENT_ID`/`GOOGLE_CLIENT_SECRET` env vars. This breaks\nout-of-the-box OAuth for users who don't set the env vars, which is a worse\nexperience than the bypass.\n\nFile v0.1.0:references/glm5-adversarial-perf.md\n\n# GLM-5.2 Performance in Adversarial Pipelines\n\n## As REVIEWER (adversarial-code-review or adversarial-code-loop CRITIQUE/VERIFY)\n\n- **Finding quality**: Excellent. Returns structured JSON with id, severity, file, line, description, suggestion. Typical review: 5-10KB JSON, 5-15 findings with concrete runnable probes.\n- **Speed**: 3-6 minutes per phase on a 35-file C/C++ embedded project. Comparable to Claude Fable 5, slower than Codex.\n- **Thoroughness**: Finds more bugs than Claude or Codex when in Inspector role (observed: 11 findings vs 7 for Claude on same codebase).\n- **Tendency to find out-of-scope bugs**: HIGH. GLM-5.2 as reviewer routinely finds pre-existing bugs in files unrelated to the spec scope (pitfall #22 in adversarial-code-loop). Always verify code on disk after REJECT.\n\n## As DEV (adversarial-code-loop BUILD/FIX)\n\n- **Reliability**: Good with provider-specific personas (v1.2.0+). 9/9 steps APPROVED cycle #1 in validated session (omnisense firmware).\n- **Without pi-specific personas**: FAILS — overwrites source files with prose reports (`<<<SEE BELOW>>>` placeholder). See `references/glm5-pi-prose-behavior.md`.\n- **Speed**: BUILD ~3 min per step, FIX ~6-8 min per cycle.\n\n## Known Limitations\n\n| Issue | Workaround |\n|-------|-----------|\n| **Sentinel file protocol unsupported** | Use `--dir` or `--stdin` mode, never `--project-dir` |\n| **Timeout on large files (>800 lines)** | Use `--dir` to scope smaller, increase `--timeout` to 1800 |\n| **Quota limits (Z.AI Lite ~80/5h)** | Track usage between steps, fall back to Codex when exhausted |\n| **Prone to out-of-scope findings** | Check code on disk after REJECT before discarding changes |","readmeExcerpt":"Skill: Adversarial Code Review Owner: chpomob Summary: Multi-perspective adversarial code review with git-isolated worktrees. Two reviewers (Architect + Inspector), cross-validation, and synthesis report. The synthesis is the final arbiter — its verdict takes priority over individual reviewer outputs. Tags: latest:0.1.0 Version history: v0.1.0 | 2026-08-03T18:11:17.731Z | auto Initial release of adversarial-code-revi","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"python3 scripts/adversarial_review.py <source> [options]"},{"language":"bash","snippet":"python3 scripts/adversarial_review.py \\\n  --file /path/to/target.py \\\n  --a-cmd \"codex exec --skip-git-repo-check --sandbox read-only\" \\\n  --b-cmd \"python3 /path/to/claude-tmux.py --timeout 600 --hard-timeout 1200 --cwd /path/to/repo\" \\\n  --synth-cmd \"python3 /path/to/claude-tmux.py --timeout 600 --hard-timeout 1200 --cwd /path/to/repo\" \\\n  --out /tmp/acr-review"},{"language":"bash","snippet":"python3 scripts/adversarial_review.py --diff-git --base main --out .adversarial-review"},{"language":"bash","snippet":"python3 scripts/adversarial_review.py \\\n  --project-dir /path/to/repo \\\n  --a-cmd \"codex exec --skip-git-repo-check --sandbox read-only\" \\\n  --b-cmd \"python3 /path/to/claude-tmux.py --timeout 600 --hard-timeout 1200 --cwd /path/to/repo\" \\\n  --cross-a-cmd \"codex exec --skip-git-repo-check --sandbox read-only\" \\\n  --cross-b-cmd \"python3 /path/to/claude-tmux.py --timeout 600 --hard-timeout 1200 --cwd /path/to/repo\" \\\n  --synth-cmd \"python3 /path/to/claude-tmux.py --timeout 600 --hard-timeout 1200 --cwd /path/to/repo\" \\\n  --out .adversarial-review --html --show-costs"},{"language":"json","snippet":"{\n  \"verdict\": \"APPROVE|REQUEST_CHANGES|REJECT\",\n  \"summary\": \"first lines of the synthesis report\",\n  \"findings\": {\"blocker\": 1, \"major\": 2, \"minor\": 4},\n  \"report\": \".adversarial-review/review.md\",\n  \"source_diff\": true\n}"},{"language":"text","snippet":"ARCHITECT ──→ reviews code (architecture, security, concurrency)\nINSPECTOR ──→ reviews code (bugs, edge cases, error handling)\nCROSS_1 ────→ challenges INSPECTOR findings with ARCHITECT perspective\nCROSS_2 ────→ challenges ARCHITECT findings with INSPECTOR perspective\nSYNTHESIS ──→ ranks, cross-validates, produces final report"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: adversarial-code-review\ndescription: \"Multi-perspective adversarial code review with git-isolated worktrees. Two reviewers (Architect + Inspector), cross-validation, and synthesis report. The synthesis is the final arbiter — its verdict takes priority over individual reviewer outputs.\"\ntags: [adversarial, code-review, multi-model, parallel, review-only, persona, git]\nversion: 1.10.0\nlicense: 0BSD\n---\n\n# adversarial-code-review\n\nMulti-perspective adversarial review of a diff or codebase. Two independent\nreviewers (**Architect** + **Inspector**) run concurrently and each produce JSON\nfindings, two **cross-review** passes (A reviews B's findings, B reviews A's\nfindings) pressure-test them, and a **synthesis** rapporteur collapses everything\ninto a single ranked report.\n\nThe review engine, subprocess runner, and personas live in the sibling\n`adversarial-common` skill — this skill only wires the review flow and the\nsource-gathering modes.\n\n## Installation\n\nRequires the `adversarial-common` sibling repo (shared engine). One-line install:\n\ncurl -fsSL https://raw.githubusercontent.com/chpomob/adversarial-code-review/main/scripts/install.sh | bash\n\nor, from an existing checkout:\n\nbash scripts/install.sh\n\nBoth place adversarial-code-review and adversarial-common side by side under `~/.hermes/skills` (override the target with `$1` or `$HERMES_HOME`).\n\n## When to use\n\n- Before merging a feature branch (`--diff-git`).\n- On a standalone patch file (`--diff`).\n- On a whole directory or single file (`--dir`, `--file`).\n- On an existing project in place (`--project-dir`).\n\n## Usage\n\n```bash\npython3 scripts/adversarial_review.py <source> [options]\n```\n\nThe reviewer command defaults to the `claude-tmux` wrapper (no model pinned —\nthe CLI picks its own best). Override per-run with `--review-cmd` or persistently\nwith `$ACR_REVIEW_CMD`.\n\n### Sources (mutually exclusive)\n\n| Flag | Argument | Reviews |\n|------|----------|---------|\n| `--diff-git` | — | `<base>..HEAD` inside an isolated git worktree (dirty tree auto-stashed) |\n| `--diff` | `FILE` | a unified-diff file |\n| `--dir` | `DIR` | every file under a directory |\n| `--file` | `FILE` | a single file |\n| `--project-dir` | `DIR` | an existing project directory in place |\n\n### Options\n\n| Flag | Default | Purpose |\n|------|---------|---------|\n| `--a-cmd` | `--review-cmd` (or `$ACR_A_CMD`) | Architect model command (overrides `--review-cmd`) |\n| `--b-cmd` | `--review-cmd` (or `$ACR_B_CMD`) | Inspector model command (overrides `--review-cmd`) |\n| `--cross-a-cmd` | `--a-cmd` (or `$ACR_CROSS_A_CMD`) | Cross-review A model — Architect reviews Inspector's findings |\n| `--cross-b-cmd` | `--b-cmd` (or `$ACR_CROSS_B_CMD`) | Cross-review B model — Inspector reviews Architect's findings |\n| `--synth-cmd` | `--review-cmd` (or `$ACR_SYNTH_CMD`) | Synthesis model command |\n| `--base` | `$ACR_BASE`, then `main`, then `master` | base ref for `--diff-git` (tried in that order) |\n| `--feature` | current branch name | slug use"},{"path":"README.md","content":"# adversarial-code-review\n\nMulti-perspective adversarial code review with git-isolated worktrees. Two independent reviewers (Architect + Inspector) each produce JSON findings, two cross-review passes pressure-test them, and a synthesis rapporteur collapses everything into a single ranked report.\n\nFor Hermes Agent, Claude Code, Codex, or any LLM CLI.\n\n## How it works\n\n```\nARCHITECT ──→ reviews code (architecture, security, concurrency)\nINSPECTOR ──→ reviews code (bugs, edge cases, error handling)\nCROSS_1 ────→ challenges INSPECTOR findings with ARCHITECT perspective\nCROSS_2 ────→ challenges ARCHITECT findings with INSPECTOR perspective\nSYNTHESIS ──→ ranks, cross-validates, produces final report\n```\n\n## Comparison\n\n| Feature | adversarial-code-review | adverse (addyosmani) | alecnielsen/adversarial-review | agent-review-panel |\n|---------|------------------------|---------------------|-------------------------------|-------------------|\n| Cross-model debate | ✅ Architect↔Inspector | ❌ Single-reviewer | ❌ Single-round | ✅ 4-6 panel |\n| Git worktree isolation | ✅ | ❌ | ❌ | ❌ |\n| Cross-review rounds | ✅ 2 rounds of devil's advocate | ❌ | ❌ | ❌ |\n| JSON findings with schema | ✅ | ❌ | ❌ | ❌ |\n| --project-dir mode | ✅ Review existing codebase | ❌ | ❌ | ❌ |\n\n## Quick start\n\n```bash\n# Review changes on a branch\npython3 scripts/adversarial_review.py --diff-git\n\n# Review a whole project directory\npython3 scripts/adversarial_review.py --project-dir /path/to/project \\\n  --a-cmd \"claude-tmux --model best\" \\\n  --b-cmd \"codex exec -C /path/to/project\"\n```\n\n## Output\n\nArtifacts land in `--out` (default `.adversarial-review`):\n\n- `final.json` — machine-readable verdict (`APPROVE|REQUEST_CHANGES|REJECT`)\n- `review.md` — ranked report with per-finding evidence\n- `01_architect.txt` … `05_synthesis.txt` — per-phase raw output\n\n## Dependencies\n\n- Python ≥ 3.11\n- Git ≥ 2.5\n- Two LLM CLIs (one for architect, one for inspector)\n\nUses `adversarial-common` as the shared engine.\n\n## License\n\n0BSD — see [LICENSE](LICENSE)."},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn7e26az9x7m8bgwfwg90q1wkh8bsqw0\",\n  \"slug\": \"adversarial-code-review-2\",\n  \"version\": \"0.1.0\",\n  \"publishedAt\": 1785780677731\n}"},{"path":"references/ai-quota-apis.md","content":"# AI CLI Quota APIs — Direct programmatic access\n\n**Updated 2026-07-31** — Standalone CLI startup and explicit endpoint inventory.\nAll five providers share the optional external adapter\n`~/.hermes/plugins/hermes-quota-status/quota_api.py`.\n\n## Architecture\n\n```\nquota_api.py  ←  shared module (token reading + API calls)\n   ├── check-ai-quota.py  ←  CLI script (human + JSON output)\n   └── hermes-quota-status/__init__.py  ←  Hermes TUI statusbar plugin\n```\n\nThe CLI imports the adapter lazily. Importing `check-ai-quota.py` and running\n`--help` therefore require only the Python standard library. A quota check still\nrequires the `hermes-quota-status` plugin; if it is absent, the CLI reports a\nper-provider error in its normal human or JSON output without a traceback.\n\nNo tmux scraping or URL-embedded API keys are used. Each direct request reveals\nthe caller's IP address, request timing, and association with the authenticated\naccount to the target provider. Credentials are sent in headers and are never\nintentionally printed. The provider-specific caveats below are additional to\nthat baseline disclosure.\n\n## Claude Code (Pro subscription)\n\n- **Token**: `~/.claude/.credentials.json` → `claudeAiOauth.accessToken`\n- **Endpoint**: `https://api.anthropic.com/api/oauth/usage`\n- **Status**: First-party Anthropic endpoint, but undocumented and\n  community-discovered; it is not a supported public API contract and may change.\n- **Auth**: `Authorization: Bearer <token>`\n- **Privacy**: Sends the Claude OAuth credential to Anthropic and requests\n  subscription utilization and reset times.\n- **Response**:\n  ```json\n  {\n    \"five_hour\": {\"utilization\": 27.0, \"resets_at\": \"2026-06-12T18:30:00Z\"},\n    \"seven_day\": {\"utilization\": 10.0, \"resets_at\": \"2026-06-19T09:00:00Z\"}\n  }\n  ```\n- **Note**: Claude returns utilization as percentages (0-100). The old code had a\n  scale=\"fraction\" bug that inflated sub-1% values to 80%. Fixed 2026-06-12.\n\n## Codex (ChatGPT Plus subscription)\n\n- **Token**: `~/.codex/auth.json` → `tokens.access_token`\n- **Endpoint**: `https://chatgpt.com/backend-api/wham/usage`\n- **Status**: First-party/official ChatGPT service endpoint used for Codex\n  account usage, but not a documented public developer API contract.\n- **Auth**: `Authorization: Bearer <token>`\n- **Privacy**: Sends the ChatGPT OAuth credential to OpenAI and requests account\n  rate-limit utilization and reset times.\n- **Response**:\n  ```json\n  {\n    \"rate_limit\": {\n      \"primary_window\": {\"used_percent\": 11, \"reset_at\": 1779762941},\n      \"secondary_window\": {\"used_percent\": 4, \"reset_at\": 1780313088}\n    }\n  }\n  ```\n\n## Gemini / agy (Google AI Studio)\n\n- **Key**: `GOOGLE_API_KEY` env var or `~/.hermes/.env` → `GOOGLE_API_KEY=...`\n- **Endpoints**:\n  - `https://generativelanguage.googleapis.com/v1beta/models`\n  - `https://generativelanguage.googleapis.com/v1beta/models/gemini-2.5-flash:generateContent`\n- **Status**: Both are official Google Generative Language API endpoints. They\n  are not "},{"path":"references/api-input-limits.md","content":"# API Input Size Limits for Adversarial Review\n\nWhen using `--project-dir` or `--dir` modes, the review script sends ALL source\nfiles concatenated to the reviewer's stdin. API-based reviewers enforce strict\ninput size limits that cause silent failures if exceeded.\n\n## Limits by provider\n\n| Provider | Max input chars | Limit type | Failure mode |\n|----------|----------------|------------|-------------|\n| Codex (OpenAI) | 1,048,576 (1 MB) | Hard API limit | `turn/start failed: Input exceeds the maximum length` — exit 1, empty stdout, no visible error in truncated stderr |\n| Claude (Anthropic) | ~200K tokens (~800K chars) | Soft per-model | Model refuses with \"input too long\" |\n| Claude-tmux | Depends on model | Varies | Usually works up to ~2M chars with extended thinking |\n\nCodex is the most restrictive: 1 MB of input characters. A project with 85 source\nfiles averages ~600K chars (safe). Adding test files, build artifacts, or library\ndependencies pushes it over the limit.\n\n## Debugging checklist when Architect phase exits 1\n\n1. Check `01_architect.txt` in the output artifact directory — if it's 0 bytes,\n   Codex received no stdin or the input was rejected\n2. Look for `input_exceeds_maximum_length` or `input_too_large` in stderr\n   (it may be buried deep in the output — grep for it)\n3. Run `python3 -c \"\nimport os; SKIP={'.git','.venv','__pycache__','node_modules','.pytest_cache','.pio','build','target','test','unity'}; PREFIX={'.adversarial','.omnisense-'}; out=[]\nfor dp,dirs,files in os.walk('.'):\n  dirs[:]=[d for d in dirs if d not in SKIP and not any(d.startswith(p) for p in PREFIX)]\n  for n in files:\n    if n.startswith('.'): continue\n    out.append(os.path.relpath(os.path.join(dp,n),'.'))\n    if len(out)>=200: break\n  if len(out)>=200: break\ntotal=sum(os.path.getsize(f) for f in out)\nprint(f'{len(out)} files, {total:,} chars')\n\"` from the project root to measure the input size\n\n## Fixes\n\n1. Add `test`, `unity` (test framework dirs) to `_SKIP_DIRS`\n2. Add `.pio`, `build`, `target` (build artifact dirs) to `_SKIP_DIRS`  \n3. Ensure `_SKIP_DIR_PREFIX` catches `.adversarial-*` and `.omnisense-*` dot-dirs\n4. Verify dot-prefixed individual files (`.omnisense-*.md` specs) are filtered\n\nAim for ≤ 700K chars to leave headroom for the persona text (~1.5K per role)."}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Multi-perspective adversarial code review with git-isolated worktrees. Two reviewers (Architect + Inspector), cross-validation, and synthesis report. The synthesis is the final arbiter — its verdict takes priority over individual reviewer outputs. Skill: Adversarial Code Review Owner: chpomob Summary: Multi-perspective adversarial code review with git-isolated worktrees. Two reviewers (Architect + Inspector), cross-validation, and synthesis report. The synthesis is the final arbiter — its verdict takes priority over individual reviewer outputs. Tags: latest:0.1.0 Version history: v0.1.0 | 2026-08-03T18:11:17.731Z | auto Initial release of adversarial-code-revi","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1681,"uniquenessScore":46,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-09T11:52:36.428Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-09T11:52:36.428Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-09T23:09:42.044Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}