{"id":"eac17fd3-5686-4b33-8230-16f0751293b5","entityType":"agent","slug":"clawhub-yinghaojia-deep-code-review","name":"Code Review — Multi-Dimensional Audit","canonicalUrl":"https://www.xpersona.co/agent/clawhub-yinghaojia-deep-code-review","canonicalPath":"/agent/clawhub-yinghaojia-deep-code-review","generatedAt":"2026-10-11T20:57:02.904Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-11T16:28:51.797Z","emptyReason":null},"description":"Multi-dimensional code audit using structured subagent delegation. Use when reviewing a GitHub release, PR, or codebase. Systematically inspects security, co... Skill: Code Review — Multi-Dimensional Audit Owner: yinghaojia Summary: Multi-dimensional code audit using structured subagent delegation. Use when reviewing a GitHub release, PR, or codebase. Systematically inspects security, co... Tags: audit:1.1.1, code-review:1.1.1, concurrency:1.1.1, latest:1.1.1, security:1.1.1, simplicity:1.1.1 Version history: v1.1.1 | 2026-05-15T05:18:14.034Z | user v1.1.1: Added /deep-code-","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1K downloads reported by the source. Last updated 10/11/2026.","installCommand":"clawhub skill install s178ctw1yw1mpyqe81gqd7y1cn84b9ms:deep-code-review","sourceUrl":"https://clawhub.ai/yinghaojia/deep-code-review","homepage":"https://clawhub.ai/yinghaojia/skills/deep-code-review","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/yinghaojia/deep-code-review","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/yinghaojia/skills/deep-code-review","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":60,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Multi-dimensional code audit using structured subagent delegation. Use when reviewing a GitHub release, PR, or codebase. Systematically inspects security, co..."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-11T16:28:51.797Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T16:28:51.797Z","emptyReason":null},"stars":null,"forks":null,"downloads":1028,"packageName":null,"latestVersion":"1.1.1","tractionLabel":"1K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T16:28:51.731Z","emptyReason":null},"lastUpdatedAt":"2026-10-11T16:28:51.797Z","lastCrawledAt":"2026-10-11T16:28:51.731Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-12T16:28:51.731Z","lastVerifiedAt":null,"highlights":[{"version":"1.1.1","createdAt":"2026-05-15T05:18:14.034Z","changelog":"v1.1.1: Added /deep-code-review, /code-review, /review-code trigger phrases for direct invocation.","fileCount":8,"zipByteSize":15669},{"version":"1.1.0","createdAt":"2026-05-15T05:12:46.933Z","changelog":"v1.1.0: Added Phase 0 Pre-Review Gate (Danger-inspired), Simplicity & Over-Engineering dimension (from HN community experience), Four-Eyes Cross-Verification protocol (Bavota & Russo 2015), and Harness Compatibility guide.","fileCount":7,"zipByteSize":14318},{"version":"1.0.0","createdAt":"2026-05-15T05:05:55.285Z","changelog":"Initial release: Introduces a multi-dimensional code audit skill with structured subagent-based review and strict evidence requirements. - Systematically audits codebases across security, concurrency, UX/logic, and test quality dimensions. - Spawns parallel subagents for each audit dimension, each verifying findings via actual source files with line citations. - Enforces strict classification for findings: Confirmed / Mitigated / False Alarm; no speculative reporting allowed. - Synthesizes findings into a severity- and priority-ranked matrix for actionable review. - Provides detailed heuristics and anti-patterns for consistent, high-quality code review processes.","fileCount":6,"zipByteSize":8767}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s178ctw1yw1mpyqe81gqd7y1cn84b9ms:deep-code-review","setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-yinghaojia-deep-code-review/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-yinghaojia-deep-code-review/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-yinghaojia-deep-code-review/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-yinghaojia-deep-code-review/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-yinghaojia-deep-code-review/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-yinghaojia-deep-code-review/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-11T20:57:02.902Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-yinghaojia-deep-code-review/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-yinghaojia-deep-code-review/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-yinghaojia-deep-code-review/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-yinghaojia-deep-code-review/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-11T16:28:51.797Z","emptyReason":null},"readme":"Skill: Code Review — Multi-Dimensional Audit\n\nOwner: yinghaojia\n\nSummary: Multi-dimensional code audit using structured subagent delegation. Use when reviewing a GitHub release, PR, or codebase. Systematically inspects security, co...\n\nTags: audit:1.1.1, code-review:1.1.1, concurrency:1.1.1, latest:1.1.1, security:1.1.1, simplicity:1.1.1\n\nVersion history:\n\nv1.1.1 | 2026-05-15T05:18:14.034Z | user\n\nv1.1.1: Added /deep-code-review, /code-review, /review-code trigger phrases for direct invocation.\n\nv1.1.0 | 2026-05-15T05:12:46.933Z | user\n\nv1.1.0: Added Phase 0 Pre-Review Gate (Danger-inspired), Simplicity & Over-Engineering dimension (from HN community experience), Four-Eyes Cross-Verification protocol (Bavota & Russo 2015), and Harness Compatibility guide.\n\nv1.0.0 | 2026-05-15T05:05:55.285Z | auto\n\nInitial release: Introduces a multi-dimensional code audit skill with structured subagent-based review and strict evidence requirements.\n\n- Systematically audits codebases across security, concurrency, UX/logic, and test quality dimensions.\n- Spawns parallel subagents for each audit dimension, each verifying findings via actual source files with line citations.\n- Enforces strict classification for findings: Confirmed / Mitigated / False Alarm; no speculative reporting allowed.\n- Synthesizes findings into a severity- and priority-ranked matrix for actionable review.\n- Provides detailed heuristics and anti-patterns for consistent, high-quality code review processes.\n\nArchive index:\n\nArchive v1.1.1: 8 files, 15669 bytes\n\nFiles: references/audit-dimensions.md (5624b), references/four-eyes.md (2397b), references/output-format.md (2131b), references/severity-rubric.md (2495b), references/subagent-templates.md (3229b), skill-card.md (2498b), SKILL.md (10390b), _meta.json (135b)\n\nFile v1.1.1:SKILL.md\n\n---\nname: code-review\ndescription: >\n  Multi-dimensional code audit using structured subagent delegation.\n  Use when reviewing a GitHub release, PR, or codebase.\n  Systematically inspects security, concurrency/state-machine safety, UX/implementation logic, test quality, and simplicity/over-engineering.\n  Spawns parallel subagents for deep verification with Four-Eyes cross-validation on critical findings.\n  Synthesizes findings into a Confirmed/Critical-to-Low priority matrix.\n  Trigger phrases: review this release, audit this codebase, check this PR for issues, 代码审查, review 代码, 审查这个版本, /deep-code-review, /code-review, /review-code\n---\n\n# Code Review — Multi-Dimensional Audit Methodology\n\nSystematically audit a codebase release through five dimensions, using parallel subagent delegation for deep verification.\nInspired by: Modern Code Review taxonomy research (Bavota & Russo 2015 \"Four Eyes Are Better Than Two\"), reviewdog's tool-agnostic harness pattern, Danger's pre-review gate philosophy, and community experience with AI-generated code quality issues.\n\n## Core Principles\n\n1. **Real code, not release notes.** Every finding must be verified against actual source files by fetching them. The only acceptable evidence is `file:line` citations. The only acceptable conclusion labels are `Confirmed / Mitigated / False Alarm`.\n\n2. **Four Eyes on every Critical.** Any finding classified as Critical severity MUST be independently verified by a second subagent before appearing in the final report. This is the \"Four Eyes\" principle from Bavota & Russo (2015): multiple reviewers independently examining the same issue catch 60%+ more real bugs than a single reviewer. See [four-eyes.md](references/four-eyes.md).\n\n3. **Simplicity is a first-class dimension.** AI-generated code often produces \"massive overkill\" — hundreds of lines for what should be a two-method change. Always ask: \"Does the complexity of this solution match the complexity of the problem?\" This dimension is inspired by community experience on Hacker News and Reddit (2025 State of AI Code Quality discussions).\n\n## Workflow\n\n### Phase 0: Pre-Review Gate (in main session, <2 min)\n\nRun these quick checks before committing to a full audit. Inspired by Danger's \"automated pre-review\" philosophy.\n\n1. **PR/Diff size check**: if the change exceeds 400 lines, flag it as high-risk and recommend splitting\n2. **Missing artifacts**: is there a CHANGELOG entry? Updated README if API changed? Migration guide if schema changed?\n3. **File-level red flags**: any committed `.env`, credentials, large binary files?\n4. **Test presence**: does this change include or update tests? If zero test changes on a >100 line diff, flag.\n\n**Output**: Gate report (pass/warn/fail) + recommended audit depth.\n\n### Phase 1: Surface Scan (in main session)\n\nRead these in order — enough to understand architecture and identify candidate issues:\n\n1. **Release notes / CHANGELOG** — what the authors claim changed\n2. **README** — project purpose, architecture diagram, on-disk layout\n3. **ARCHITECTURE.md** or equivalent — module decomposition, API contracts\n4. **Directory tree** (via GitHub tree view) — file listing to map modules\n5. **Key source files** — entry point, core state machine, critical paths (read ~3-8 files)\n\n**Output**: A list of 10-20 candidate issues, categorized by dimension:\n- **Security** (SSRF, injection, auth, path traversal, credential leaks)\n- **Concurrency & State Machine** (race conditions, missing locks, TOCTOU, state corruption)\n- **UX & Implementation Logic** (feature semantics, error messages, recovery paths, access control)\n- **Test Quality** (mock fidelity, integration gaps, signature mismatches, coverage blind spots)\n- **Simplicity & Over-Engineering** (complexity-vs-problem mismatch, unnecessary abstraction, AI-bloat patterns)\n\n### Phase 2: Deep Audit (via subagents)\n\nFor each non-trivial dimension, spawn an isolated subagent. Each subagent:\n\n1. **Fetches every relevant source file** via `web_fetch` — never infers from docs\n2. **Verifies each issue against actual code** — cites specific lines\n3. **Constructs exploit scenarios** (security) or **race timelines** (concurrency)\n4. **Returns structured findings** with: Conclusion / Severity / Source Evidence / Risk / Fix\n\nSee [subagent-templates.md](references/subagent-templates.md) for the exact prompt template.\nSee [audit-dimensions.md](references/audit-dimensions.md) for dimension-specific question probes.\n\n**Model guidance**: Use the same model for all subagents to ensure consistent judgment. Prefer high-reasoning models for complex audits.\n\n### Phase 3: Four-Eyes Cross-Verification (critical findings only)\n\nFor every finding classified as **Critical** by a subagent:\n\n1. Spawn a **second, independent subagent** (different dimension focus) with the exact same issue prompt\n2. If both confirm → **Confirmed.** The issue enters the final report with a `👁️ Four-Eyes Verified` badge.\n3. If they disagree → **Flag as \"Disputed\"** in the report with both conclusions quoted.\n4. If the second finds the issue is Mitigated/False Alarm while the first said Critical → **The second wins.** But keep both in an appendix.\n\n### Phase 4: Synthesis (in main session)\n\nWhen all subagent reports return:\n\n1. **Merge findings** — deduplicate across dimensions, re-classify severity\n2. **Build summary table** — all issues with conclusion + severity + source dimension + root cause\n3. **Build priority matrix** — P0 (drop everything) through P5 (nice to have), with estimated work and blast radius\n4. **Write executive summary** — overall quality assessment + top 3 action items\n5. **Add Simplicity Score** — a subjective score 1-5 on whether the codebase's complexity matches its problem domain. 5 = elegantly simple, 1 = massively over-engineered.\n\nSee [output-format.md](references/output-format.md) for table and emoji conventions.\nSee [severity-rubric.md](references/severity-rubric.md) for severity classification rules.\n\n## Key Heuristics\n\n### Security Scan Heuristics\n\n- **Every URL fetch path must be checked for SSRF**: trace from user input → URL parsing → DNS resolution → HTTP request → redirect handling → response reading. Flag any step that skips IP validation.\n- **Every subprocess call must be checked for injection**: is `shell=True` used? Are user-controlled strings concatenated into the command? Are file paths sanitized?\n- **Every external API call must be checked for credential leaks**: are tokens/secrets logged? Do error messages include request bodies?\n\n### Concurrency Scan Heuristics\n\n- **For every `.json` / `.jsonl` write**: check if it uses tmp-rename atomic pattern or flock. Direct overwrite without either = bug.\n- **For every `load → modify → save` pattern**: check if the entire block is lock-protected. If load happens outside the lock, it's a TOCTOU bug.\n- **For every state machine transition**: check if two concurrent events can both see the same \"before\" state and both advance. If yes, state corruption possible.\n- **For every append-only log**: verify flock(LOCK_EX) covers the full append operation.\n\n### UX/Logic Scan Heuristics\n\n- **For every feature flag/mode**: trace all branches. Does \"mode=review-only\" actually prevent non-review actions? Don't trust the name — verify the code.\n- **For every error message**: read it as a user would. Does it tell you what went wrong AND how to fix it? If it only says \"X failed\", flag it.\n- **For every multi-step workflow**: is there an undo/backtrack/revisit path? If not, flag it.\n- **For every access-control check**: look for what's NOT checked. Does a group chat require @-mention? Does a rate limit exist?\n\n### Test Quality Heuristics\n\n- **Mocks that match wrong signatures**: if a test monkeypatches `call_llm` with a fake that takes `**kw` and reads `kw.get(\"old_param\")`, it will never catch a production code change to `new_param`. Flag these.\n- **No integration test in CI**: if CI only runs unit tests with mocks and there's no end-to-end smoke test, flag it.\n\n### Simplicity & Over-Engineering Heuristics (NEW — v1.1.0)\n\n- **AI-bloat detection**: does the change introduce a new service class, background worker, or framework dependency for what should be 1-2 methods in an existing file? (Inspired by the HN \"batching = 2 methods, not 200 lines\" incident.)\n- **Abstraction without justification**: does the code introduce interfaces, factories, or dependency injection where direct calls would suffice? For each abstraction layer, ask: \"What concrete problem does this solve today?\"\n- **Dead code or \"future-proofing\"**: are there code paths, config options, or extension points that are not used by any existing feature?\n- **Config sprawl**: does this change add new config keys, env vars, or CLI flags? Is each one justified by a real use case?\n- **Copy-paste detection**: are there blocks of >10 lines that could be extracted? Flag both directions — missing DRY AND forced DRY where the two copies have different evolution paths.\n\n## Anti-Patterns (avoid)\n\n- ❌ Filing an issue based on release notes alone (always verify against source)\n- ❌ Accepting a docstring claim without checking the implementation\n- ❌ Using \"I think\" / \"probably\" / \"seems like\" — every finding is Confirmed or it's not a finding\n- ❌ Leaving severity as \"TBD\" — classify immediately using the rubric\n- ❌ Mentioning an issue in prose without filing it in the structured output table\n- ❌ Trusting a single subagent's Critical finding without Four-Eyes cross-verification\n\n## Harness Compatibility\n\nThis skill is designed to work **with** existing code review tooling, not replace it. The recommended stack:\n\n| Layer | Tool | What it catches |\n|---|---|---|\n| **Lint/formatter** | ruff, eslint, gofmt | Style, basic bugs |\n| **Static analysis** | SonarQube, Semgrep, CodeQL | Security vulns, code smells |\n| **Diff harness** | reviewdog, Danger | Runs the above, posts inline comments |\n| **🆕 Deep audit** | **This skill** | Cross-cutting: concurrency, over-engineering, UX logic, test gaps |\n| **Human review** | Your team | Architecture, trade-offs, domain knowledge |\n\nThe Pre-Review Gate (Phase 0) picks up what reviewdog/SonarQube would catch, so you don't waste subagent time on style issues. Subagents focus on what static analysis **can't** see.\n\nFile v1.1.1:_meta.json\n\n{\n  \"ownerId\": \"kn7b56gp3gfpjr0n2jxr3hkfyn82cnkm\",\n  \"slug\": \"deep-code-review\",\n  \"version\": \"1.1.1\",\n  \"publishedAt\": 1778822294034\n}\n\nFile v1.1.1:references/audit-dimensions.md\n\n# Audit Dimensions\n\nEach codebase audit spans these five dimensions. For each dimension, spawn a dedicated subagent that fetches source files and verifies issues against actual code.\n\n## 1. Security (安全审计)\n\n**Focus**: Vulnerabilities that allow unauthorized access, data exfiltration, or resource abuse.\n\n**Key questions to probe**:\n- Are there SSRF attack vectors? Check URL fetching/redirect logic, DNS resolution, IP filtering\n- Are there command injection risks? Search for `subprocess`, `shell=True`, unchecked file path concatenation\n- Are there auth token leaks? Check logging, error messages, HTTP response handling\n- Are there path traversal risks? Check file operations with user-controlled paths\n- Are third-party dependencies validated? Check version pinning, integrity checks\n\n**Verification method**: Trace the full attack path from user input to exploit. For each defense claim in the code, verify it actually works at the call site.\n\n## 2. Concurrency & State Machine Safety (并发与状态机安全)\n\n**Focus**: Race conditions, data corruption, inconsistent state in multi-thread/multi-process environments.\n\n**Key questions to probe**:\n- Are file locks (flock/fcntl) used for shared state? Is the lock scope correct (covers read-modify-write)?\n- Are state transitions atomic? Check `load → modify → save` and `load → transition → save` patterns\n- Are there TOCTOU (Time-of-check-time-of-use) gaps? Gap between check and action\n- Are there duplicate definitions of the same named exception/class?\n- Is there any lock-free concurrent write to the same file?\n- For append-only logs: is `append_jsonl` properly flocked?\n- For json writes: is tmp-rename used for atomicity?\n\n**Verification method**: Construct a timeline with two concurrent operations, trace each thread's read and write points.\n\n## 3. UX & Implementation Logic (用户体验与实现逻辑)\n\n**Focus**: Whether the code actually implements what the docs/release-notes claim, and whether user-facing behavior is correct.\n\n**Key questions to probe**:\n- Do feature flags/modes actually enforce their claimed semantics? (e.g., \"review-only\" truly limits to reviews)\n- Are there dead code paths that silently swallow errors?\n- Are error messages actionable? Do they tell the user how to fix the problem?\n- Can users recover from mistakes? Is there undo/backtrack/revisit?\n- Are there implicit access control gaps? (e.g., no @-mention check in group chats)\n- Do test mocks hide real bugs? Check if mocked interfaces match actual call signatures\n- Are there misleading code comments that describe behavior not implemented?\n\n**Verification method**: Trace each code branch and confirm the actual behavior against the documented claim. For each error path, read the error message and assess if a user could act on it.\n\n## 4. Test Quality & Coverage Blind Spots (测试质量)\n\n**Focus**: Whether tests actually catch the bugs they claim to prevent.\n\n**Key questions to probe**:\n- Do integration tests use real dependencies or mocks? If mocked, do mocks validate call signatures?\n- Are there end-to-end tests in CI? If not, what's the smoke test strategy?\n- What bug classes are invisible to the current test suite?\n- Do test fixtures match production call signatures?\n\n**Verification method**: Read a representative sample of test files. Check if a wrong call signature or wrong return type would cause any test to fail.\n\n## 5. Simplicity & Over-Engineering (简洁性与过度工程) — NEW v1.1.0\n\n**Focus**: Whether the code's complexity is justified by the problem it solves. AI-generated code is prone to \"massive overkill\" — hundreds of lines with new services, patterns, and abstractions for what should be a small incremental change.\n\n**Background**: Community experience (Hacker News 2025 State of AI Code Quality discussion, Reddit r/ExperiencedDevs) consistently reports that AI coding agents over-engineer solutions. A \"batching\" request that should be 2 methods becomes a new service class + background worker + suite of tests. This dimension explicitly hunts for these patterns.\n\n**Key questions to probe**:\n- **AI-bloat check**: For each changed file, compute the ratio of \"conceptual complexity needed\" to \"lines of code added\". Does this PR add a new class/module/abstraction where a function would suffice?\n- **Abstraction audit**: Does the code introduce interfaces, factories, strategies, or dependency injection where direct calls would work? For each, ask: \"What concrete problem does this abstraction solve today, not hypothetically?\"\n- **Dead or speculative code**: Are there code paths, config keys, or extension points with zero current consumers? Flag \"future-proofing\" that isn't tied to a known roadmap item.\n- **Config sprawl**: Count new env vars, CLI flags, and config keys. Is each justified by a real user-facing need?\n- **Dependency footprint**: Count new third-party imports. Does every new dependency carry its weight?\n- **Rube Goldberg detection**: Could the same result be achieved with a simpler approach using existing infrastructure?\n\n**Verification method**: For each module changed, write a one-sentence description of \"what this module actually does.\" If that sentence is surprisingly short compared to the code volume, flag it.\n\n**Simplicity Score**: At the end of the Simplicity audit, assign a 1-5 score:\n- 5 = Elegantly minimal — every line earns its place\n- 4 = Clean — minor nitpicks but nothing egregious\n- 3 = Acceptable — some bloat but functional\n- 2 = Over-engineered — substantial unnecessary complexity\n- 1 = Massively bloated — requires refactoring before merge\n\nFile v1.1.1:references/four-eyes.md\n\n# Four-Eyes Cross-Verification Protocol\n\n## Origin\n\nBavota & Russo (2015), \"Four Eyes Are Better Than Two: On the Impact of Code Reviews on Software Quality\", found that multiple independent reviewers examining the same code catch 60%+ more real defects than a single reviewer. This protocol operationalizes that finding for AI subagent review.\n\n## Protocol\n\n### When to activate\n\nFour-Eyes cross-verification is **mandatory** for:\n- Any finding classified as **Critical** by a primary subagent\n- Any finding where the primary subagent reports \"confirmed with medium confidence\" or similar hedging\n\n### How to execute\n\n1. **Spawn a second subagent** with the **exact same issue prompt** as the primary\n2. The second subagent must have a **different dimension focus** than the primary (to avoid groupthink)\n3. The second subagent sees only the issue prompt — NOT the primary's findings (to ensure independence)\n4. Both reports are compared after completion\n\n### Resolution rules\n\n| Primary says | Second says | Final conclusion |\n|---|---|---|\n| Critical | Critical | **Confirmed** + `👁️ Four-Eyes Verified` badge |\n| Critical | High/Medium/Low | **Confirmed** at the higher severity. Add note: \"Second reviewer rated lower at [severity].\" |\n| Critical | Mitigated | **Mitigated** (the second found a defense the first missed). Keep both reports in appendix. |\n| Critical | False Alarm | **False Alarm** (the second found the issue doesn't exist). Keep both reports in appendix. |\n| Critical | Disputed (disagrees but can't reclassify) | **Disputed** — flag for human reviewer. Include both conclusions. |\n\n### Reporting format\n\nIn the final report, Critical findings carry:\n\n```\n### 🔴 Critical: [Issue Title]\n👁️ Four-Eyes Verified by [subagent-2-name]\n```\n\nDisputed findings carry:\n\n```\n### ⚠️ Disputed: [Issue Title]\n🔴 Subagent 1 ([dimension]): Critical — [summary]\n🟢 Subagent 2 ([dimension]): Mitigated — [defense found]\n→ Human review recommended.\n```\n\n## Cost/benefit guidance\n\n- **Small codebase (<20 files)**: Four-Eyes only on findings the auditor is uncertain about\n- **Medium codebase (20-100 files)**: Four-Eyes on all Critical findings\n- **Large/mission-critical codebase**: Four-Eyes on Critical + High findings\n\nThe additional cost is ~25-40% more subagent compute, but the false-positive reduction makes it worthwhile for release-blocking decisions.\n\nFile v1.1.1:references/output-format.md\n\n# Output Format Specification\n\n## Per-Issue Format\n\nEvery issue found must be reported with:\n\n```\n### 🔴/🟡/🟢 [Severity]: [Issue Title]\n\n**结论**: Confirmed / Mitigated / False Alarm / Disputed\n[If Critical + Four-Eyes verified: 👁️ Four-Eyes Verified by [subagent-name]]\n\n**源码证据**: [file:line range — the specific code that proves the issue]\n[Optional: quote the relevant code block]\n\n**风险场景**: [Concrete scenario — user does X → system does Y → consequence Z]\n\n**修复建议**: [Actionable, specific fix — not \"consider improving\"]\n```\n\n## Synthesis Format (after all subagent reports)\n\nAggregate findings into:\n\n### 1. Summary Table by Severity\n\n```\n| # | 问题 | 来源(审计维度) | 根因(一行) | 👁️ |\n|---|------|--------------|----------|-----|\n```\n`👁️` column: ✅ if Four-Eyes verified, blank otherwise.\n\n### 2. Priority Matrix\n\n```\n| 优先级 | 问题 | 工作量 | 影响范围 |\n|--------|------|--------|----------|\n| P0 🔴 | ... | 1行代码 | 所有用户 |\n| P1 🔴 | ... | ~30行 | 特定场景 |\n```\n\n### 3. Simplicity Score (NEW v1.1.0)\n\nA 1-5 rating of how well the codebase's complexity matches its problem domain:\n- **5/5** 🏆 — Elegantly minimal. Every line earns its place.\n- **4/5** ✅ — Clean. Minor nitpicks.\n- **3/5** ⚠️ — Acceptable. Some bloat but functional.\n- **2/5** ❌ — Over-engineered. Needs refactoring.\n- **1/5** 🚨 — Massively bloated. Do not merge.\n\n### 4. Executive Summary\n\nOne paragraph that captures the overall quality assessment + top 3 action items.\n\n## Emoji Convention\n\n- 🔴 Critical — System violates core guarantees\n- 🔴 High — Significant impact, fix before next release\n- 🟡 Medium — Degrades UX or creates operational risk\n- 🟢 Low — Cosmetic, theoretical, or well-mitigated\n- ✅ Mitigated — Concern exists but defended elsewhere\n- ❌ False Alarm — Concern does not exist\n- ⚠️ Disputed — Four-Eyes reviewers disagree (needs human)\n- 👁️ Four-Eyes Verified — Independently confirmed by a second subagent\n\nUse the emoji in the severity tag, not the conclusion tag.\n\nFile v1.1.1:references/severity-rubric.md\n\n# Severity Classification Rubric\n\nEvery confirmed issue gets exactly one severity level.\n\n## Critical\n\n**Definition**: A bug that causes the system to violate its core security or correctness guarantees, with no workaround.\n\n**Examples**:\n- Security boundary bypass that leaks private data\n- Feature flag that doesn't actually limit the feature it's named after\n- Data corruption that silently destroys user data\n- **Over-engineering so severe it makes the codebase unmaintainable** (e.g., 500 lines of new abstraction for a one-line bugfix)\n\n**Litmus test**: Would a reasonable user be shocked to learn this exists in production?\n\n## High\n\n**Definition**: A bug with significant security, correctness, or reliability impact, but with a partial workaround or limited blast radius.\n\n**Examples**:\n- SSRF vulnerability that requires attacker-controlled input\n- Missing @-mention check in group chats (abuse requires insider knowledge)\n- Concurrent race that requires specific timing to trigger\n\n**Litmus test**: Would you fix this before the next release?\n\n## Medium\n\n**Definition**: A bug that degrades user experience, creates operational risk, or has been partially mitigated.\n\n**Examples**:\n- Missing backtrack/revisit mechanism in a multi-step workflow\n- Error message that doesn't guide the user to the fix\n- Flock implemented but with gaps (correct on happy path, missing on edge path)\n- Race condition that wastes resources but doesn't corrupt data\n\n**Litmus test**: Would you prioritize fixing this in the next sprint?\n\n## Low\n\n**Definition**: Cosmetic issues, theoretical concerns, or already well-mitigated risks.\n\n**Examples**:\n- DNS rebinding theory window with high exploit difficulty\n- Token refresh margin that's overly conservative (no harm)\n- Function rename needed for code clarity\n\n**Litmus test**: Would you fix this only if you happened to be editing that file anyway?\n\n## Mitigated\n\n**Definition**: The reported concern exists in some form but is defended against by another mechanism.\n\n**Examples**:\n- Claimed SSRF vulnerability that's actually handled by a wrapper\n- Alleged race condition where an upstream lock serializes access\n- Reported missing check that actually happens at a different layer\n\n## False Alarm\n\n**Definition**: The reported concern does not exist in the codebase. No defense is needed because no attack path exists.\n\n**Examples**:\n- Reported command injection where no subprocess is used\n- Claimed token leak where no logging of sensitive data exists\n\nFile v1.1.1:references/subagent-templates.md\n\n# Subagent Audit Templates\n\nEach dimension gets its own subagent. Use these template structures when spawning.\n\n## Template Structure\n\n```markdown\n你是 [项目名] [版本号] 的 [审计维度]审计员。请仔细审计以下问题，从源码层面 double-check 每个问题是否真实存在。每个问题必须给出明确结论（Confirmed / Mitigated / False Alarm）。\n\n**背景**: [项目简介、运行环境、关键约束]\n\n## 问题 #N: [问题标题]\n\n[问题描述 — 包含具体的怀疑、预期 vs 实际、源码引用]\n\n请核实:\n- 阅读 [具体源文件 URL]\n- [具体的检查点 1]\n- [具体的检查点 2]\n- [判断标准]\n\n## 输出格式\n\n对每个问题给出:\n1. **结论**: Confirmed / Mitigated / False Alarm\n2. **严重程度**: Critical / High / Medium / Low\n3. **源码证据**: 引用具体行/函数/文件\n4. **风险描述**: 简明说明\n5. **修复建议**: 具体可操作的方案\n\n用中文输出。必须先 fetch 每个相关源文件确认实际代码。\n```\n\n## Template for Simplicity & Over-Engineering Subagent (NEW v1.1.0)\n\n```markdown\n你是 [项目名] [版本号] 的简洁性审计员。你的工作不是找 bug——而是判断代码是否\"过度工程化\"。\n\n**核心问题**: 这段代码的复杂度是否匹配它要解决的问题？\n\n**审计方法**:\n1. 读取每个被修改的文件\n2. 用一句话总结\"这个模块实际做了什么\"\n3. 如果一句话总结过于简短而代码量巨大 → 这是过度工程\n4. 检查每一个新引入的抽象层（接口/工厂/策略模式），问：它今天解决什么具体问题？\n5. 检查新依赖、新配置项、新CLI参数——每个都需要具体的使用场景\n6. 检查是否有\"未来扩展点\"没有当前消费者\n\n**输出**:\n1. **简洁性评分**: 1-5 (5=极简优雅, 1=大规模过度工程)\n2. **问题清单**: 每个过度工程问题 + 具体的简化建议\n3. **对比**: \"这段代码实现了 [X]；用更简单的方式可以只用 [Y] 行\"\n\n用中文输出。引用具体文件行。\n```\n\n## Subagent Quality Rules\n\n1. **Every claim must cite a source line**. Never say \"the code does X\" without referencing a file:line.\n2. **Fetch before judging**. Always `web_fetch` the actual source file; never infer from release notes or documentation.\n3. **Default to Confirmed/Mitigated/False Alarm**. Never leave a finding ambiguous.\n4. **If truncated, re-fetch**. If a source file exceeds the fetch limit, fetch with offset.\n5. **Construct exploit scenarios**. For security bugs, trace the full attack path with concrete URLs/IPs.\n6. **Construct race timelines**. For concurrency bugs, show reads and writes from each thread at each time step.\n7. **Estimate complexity delta**. For over-engineering findings, show the line-count if done simply.\n\n## Subagent Count Guidelines\n\n- **Small codebase (<20 files, single module)**: 2-3 subagents (combine dimensions)\n- **Medium codebase (20-100 files, 2-5 modules)**: 3-4 subagents\n- **Large codebase (100+ files, monorepo)**: 4-5 subagents (one per dimension)\n\nAssign dimensions based on the PR's content — not every audit needs all five dimensions. Skip Simplicity for refactoring PRs. Skip Concurrency for single-threaded CLI tools.\n\nFile v1.1.1:skill-card.md\n\n## Description:\n\nMulti-dimensional code audit skill that uses structured subagent delegation to review GitHub releases, pull requests, or codebases across security, concurrency, implementation logic, test quality, and simplicity.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[yinghaojia](https://clawhub.ai/user/yinghaojia)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and engineering teams use this skill to run structured code audits before merging or releasing code. It helps organize findings into verified severity, priority, risk, and fix guidance.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The workflow reads source code and may use multiple subagents during review, which can expose private code beyond the main session.\n\nMitigation: Confirm the review scope and data handling expectations before using it on private repositories or sensitive code.\n\nRisk: Audit findings and fix recommendations may be incomplete, incorrect, or overly broad.\n\nMitigation: Require line-level evidence for each finding and have humans verify important remediation decisions before changing production code.\n\nRisk: Critical findings can create release-blocking decisions if a single reviewer misclassifies severity.\n\nMitigation: Use the skill's Four-Eyes cross-verification process for Critical findings and escalate disputed findings to a human reviewer.\n\n## Reference(s):\n\n- [ClawHub Skill Page](https://clawhub.ai/yinghaojia/skills/deep-code-review)\n- [Audit Dimensions](references/audit-dimensions.md)\n- [Four-Eyes Cross-Verification Protocol](references/four-eyes.md)\n- [Output Format Specification](references/output-format.md)\n- [Severity Classification Rubric](references/severity-rubric.md)\n- [Subagent Audit Templates](references/subagent-templates.md)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, guidance]\n\n**Output Format:** [Markdown report with structured findings, severity tables, priority matrix, and remediation guidance]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [May include Chinese-formatted report sections and source file citations.]\n\n## Skill Version(s):\n\n1.1.1 (source: server release evidence)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v1.1.0: 7 files, 14318 bytes\n\nFiles: references/audit-dimensions.md (5624b), references/four-eyes.md (2397b), references/output-format.md (2131b), references/severity-rubric.md (2495b), references/subagent-templates.md (3229b), SKILL.md (10343b), _meta.json (135b)\n\nFile v1.1.0:SKILL.md\n\n---\nname: code-review\ndescription: >\n  Multi-dimensional code audit using structured subagent delegation.\n  Use when reviewing a GitHub release, PR, or codebase.\n  Systematically inspects security, concurrency/state-machine safety, UX/implementation logic, test quality, and simplicity/over-engineering.\n  Spawns parallel subagents for deep verification with Four-Eyes cross-validation on critical findings.\n  Synthesizes findings into a Confirmed/Critical-to-Low priority matrix.\n  Trigger phrases: review this release, audit this codebase, check this PR for issues, 代码审查, review 代码, 审查这个版本\n---\n\n# Code Review — Multi-Dimensional Audit Methodology\n\nSystematically audit a codebase release through five dimensions, using parallel subagent delegation for deep verification.\nInspired by: Modern Code Review taxonomy research (Bavota & Russo 2015 \"Four Eyes Are Better Than Two\"), reviewdog's tool-agnostic harness pattern, Danger's pre-review gate philosophy, and community experience with AI-generated code quality issues.\n\n## Core Principles\n\n1. **Real code, not release notes.** Every finding must be verified against actual source files by fetching them. The only acceptable evidence is `file:line` citations. The only acceptable conclusion labels are `Confirmed / Mitigated / False Alarm`.\n\n2. **Four Eyes on every Critical.** Any finding classified as Critical severity MUST be independently verified by a second subagent before appearing in the final report. This is the \"Four Eyes\" principle from Bavota & Russo (2015): multiple reviewers independently examining the same issue catch 60%+ more real bugs than a single reviewer. See [four-eyes.md](references/four-eyes.md).\n\n3. **Simplicity is a first-class dimension.** AI-generated code often produces \"massive overkill\" — hundreds of lines for what should be a two-method change. Always ask: \"Does the complexity of this solution match the complexity of the problem?\" This dimension is inspired by community experience on Hacker News and Reddit (2025 State of AI Code Quality discussions).\n\n## Workflow\n\n### Phase 0: Pre-Review Gate (in main session, <2 min)\n\nRun these quick checks before committing to a full audit. Inspired by Danger's \"automated pre-review\" philosophy.\n\n1. **PR/Diff size check**: if the change exceeds 400 lines, flag it as high-risk and recommend splitting\n2. **Missing artifacts**: is there a CHANGELOG entry? Updated README if API changed? Migration guide if schema changed?\n3. **File-level red flags**: any committed `.env`, credentials, large binary files?\n4. **Test presence**: does this change include or update tests? If zero test changes on a >100 line diff, flag.\n\n**Output**: Gate report (pass/warn/fail) + recommended audit depth.\n\n### Phase 1: Surface Scan (in main session)\n\nRead these in order — enough to understand architecture and identify candidate issues:\n\n1. **Release notes / CHANGELOG** — what the authors claim changed\n2. **README** — project purpose, architecture diagram, on-disk layout\n3. **ARCHITECTURE.md** or equivalent — module decomposition, API contracts\n4. **Directory tree** (via GitHub tree view) — file listing to map modules\n5. **Key source files** — entry point, core state machine, critical paths (read ~3-8 files)\n\n**Output**: A list of 10-20 candidate issues, categorized by dimension:\n- **Security** (SSRF, injection, auth, path traversal, credential leaks)\n- **Concurrency & State Machine** (race conditions, missing locks, TOCTOU, state corruption)\n- **UX & Implementation Logic** (feature semantics, error messages, recovery paths, access control)\n- **Test Quality** (mock fidelity, integration gaps, signature mismatches, coverage blind spots)\n- **Simplicity & Over-Engineering** (complexity-vs-problem mismatch, unnecessary abstraction, AI-bloat patterns)\n\n### Phase 2: Deep Audit (via subagents)\n\nFor each non-trivial dimension, spawn an isolated subagent. Each subagent:\n\n1. **Fetches every relevant source file** via `web_fetch` — never infers from docs\n2. **Verifies each issue against actual code** — cites specific lines\n3. **Constructs exploit scenarios** (security) or **race timelines** (concurrency)\n4. **Returns structured findings** with: Conclusion / Severity / Source Evidence / Risk / Fix\n\nSee [subagent-templates.md](references/subagent-templates.md) for the exact prompt template.\nSee [audit-dimensions.md](references/audit-dimensions.md) for dimension-specific question probes.\n\n**Model guidance**: Use the same model for all subagents to ensure consistent judgment. Prefer high-reasoning models for complex audits.\n\n### Phase 3: Four-Eyes Cross-Verification (critical findings only)\n\nFor every finding classified as **Critical** by a subagent:\n\n1. Spawn a **second, independent subagent** (different dimension focus) with the exact same issue prompt\n2. If both confirm → **Confirmed.** The issue enters the final report with a `👁️ Four-Eyes Verified` badge.\n3. If they disagree → **Flag as \"Disputed\"** in the report with both conclusions quoted.\n4. If the second finds the issue is Mitigated/False Alarm while the first said Critical → **The second wins.** But keep both in an appendix.\n\n### Phase 4: Synthesis (in main session)\n\nWhen all subagent reports return:\n\n1. **Merge findings** — deduplicate across dimensions, re-classify severity\n2. **Build summary table** — all issues with conclusion + severity + source dimension + root cause\n3. **Build priority matrix** — P0 (drop everything) through P5 (nice to have), with estimated work and blast radius\n4. **Write executive summary** — overall quality assessment + top 3 action items\n5. **Add Simplicity Score** — a subjective score 1-5 on whether the codebase's complexity matches its problem domain. 5 = elegantly simple, 1 = massively over-engineered.\n\nSee [output-format.md](references/output-format.md) for table and emoji conventions.\nSee [severity-rubric.md](references/severity-rubric.md) for severity classification rules.\n\n## Key Heuristics\n\n### Security Scan Heuristics\n\n- **Every URL fetch path must be checked for SSRF**: trace from user input → URL parsing → DNS resolution → HTTP request → redirect handling → response reading. Flag any step that skips IP validation.\n- **Every subprocess call must be checked for injection**: is `shell=True` used? Are user-controlled strings concatenated into the command? Are file paths sanitized?\n- **Every external API call must be checked for credential leaks**: are tokens/secrets logged? Do error messages include request bodies?\n\n### Concurrency Scan Heuristics\n\n- **For every `.json` / `.jsonl` write**: check if it uses tmp-rename atomic pattern or flock. Direct overwrite without either = bug.\n- **For every `load → modify → save` pattern**: check if the entire block is lock-protected. If load happens outside the lock, it's a TOCTOU bug.\n- **For every state machine transition**: check if two concurrent events can both see the same \"before\" state and both advance. If yes, state corruption possible.\n- **For every append-only log**: verify flock(LOCK_EX) covers the full append operation.\n\n### UX/Logic Scan Heuristics\n\n- **For every feature flag/mode**: trace all branches. Does \"mode=review-only\" actually prevent non-review actions? Don't trust the name — verify the code.\n- **For every error message**: read it as a user would. Does it tell you what went wrong AND how to fix it? If it only says \"X failed\", flag it.\n- **For every multi-step workflow**: is there an undo/backtrack/revisit path? If not, flag it.\n- **For every access-control check**: look for what's NOT checked. Does a group chat require @-mention? Does a rate limit exist?\n\n### Test Quality Heuristics\n\n- **Mocks that match wrong signatures**: if a test monkeypatches `call_llm` with a fake that takes `**kw` and reads `kw.get(\"old_param\")`, it will never catch a production code change to `new_param`. Flag these.\n- **No integration test in CI**: if CI only runs unit tests with mocks and there's no end-to-end smoke test, flag it.\n\n### Simplicity & Over-Engineering Heuristics (NEW — v1.1.0)\n\n- **AI-bloat detection**: does the change introduce a new service class, background worker, or framework dependency for what should be 1-2 methods in an existing file? (Inspired by the HN \"batching = 2 methods, not 200 lines\" incident.)\n- **Abstraction without justification**: does the code introduce interfaces, factories, or dependency injection where direct calls would suffice? For each abstraction layer, ask: \"What concrete problem does this solve today?\"\n- **Dead code or \"future-proofing\"**: are there code paths, config options, or extension points that are not used by any existing feature?\n- **Config sprawl**: does this change add new config keys, env vars, or CLI flags? Is each one justified by a real use case?\n- **Copy-paste detection**: are there blocks of >10 lines that could be extracted? Flag both directions — missing DRY AND forced DRY where the two copies have different evolution paths.\n\n## Anti-Patterns (avoid)\n\n- ❌ Filing an issue based on release notes alone (always verify against source)\n- ❌ Accepting a docstring claim without checking the implementation\n- ❌ Using \"I think\" / \"probably\" / \"seems like\" — every finding is Confirmed or it's not a finding\n- ❌ Leaving severity as \"TBD\" — classify immediately using the rubric\n- ❌ Mentioning an issue in prose without filing it in the structured output table\n- ❌ Trusting a single subagent's Critical finding without Four-Eyes cross-verification\n\n## Harness Compatibility\n\nThis skill is designed to work **with** existing code review tooling, not replace it. The recommended stack:\n\n| Layer | Tool | What it catches |\n|---|---|---|\n| **Lint/formatter** | ruff, eslint, gofmt | Style, basic bugs |\n| **Static analysis** | SonarQube, Semgrep, CodeQL | Security vulns, code smells |\n| **Diff harness** | reviewdog, Danger | Runs the above, posts inline comments |\n| **🆕 Deep audit** | **This skill** | Cross-cutting: concurrency, over-engineering, UX logic, test gaps |\n| **Human review** | Your team | Architecture, trade-offs, domain knowledge |\n\nThe Pre-Review Gate (Phase 0) picks up what reviewdog/SonarQube would catch, so you don't waste subagent time on style issues. Subagents focus on what static analysis **can't** see.\n\nFile v1.1.0:_meta.json\n\n{\n  \"ownerId\": \"kn7b56gp3gfpjr0n2jxr3hkfyn82cnkm\",\n  \"slug\": \"deep-code-review\",\n  \"version\": \"1.1.0\",\n  \"publishedAt\": 1778821966933\n}\n\nFile v1.1.0:references/audit-dimensions.md\n\n# Audit Dimensions\n\nEach codebase audit spans these five dimensions. For each dimension, spawn a dedicated subagent that fetches source files and verifies issues against actual code.\n\n## 1. Security (安全审计)\n\n**Focus**: Vulnerabilities that allow unauthorized access, data exfiltration, or resource abuse.\n\n**Key questions to probe**:\n- Are there SSRF attack vectors? Check URL fetching/redirect logic, DNS resolution, IP filtering\n- Are there command injection risks? Search for `subprocess`, `shell=True`, unchecked file path concatenation\n- Are there auth token leaks? Check logging, error messages, HTTP response handling\n- Are there path traversal risks? Check file operations with user-controlled paths\n- Are third-party dependencies validated? Check version pinning, integrity checks\n\n**Verification method**: Trace the full attack path from user input to exploit. For each defense claim in the code, verify it actually works at the call site.\n\n## 2. Concurrency & State Machine Safety (并发与状态机安全)\n\n**Focus**: Race conditions, data corruption, inconsistent state in multi-thread/multi-process environments.\n\n**Key questions to probe**:\n- Are file locks (flock/fcntl) used for shared state? Is the lock scope correct (covers read-modify-write)?\n- Are state transitions atomic? Check `load → modify → save` and `load → transition → save` patterns\n- Are there TOCTOU (Time-of-check-time-of-use) gaps? Gap between check and action\n- Are there duplicate definitions of the same named exception/class?\n- Is there any lock-free concurrent write to the same file?\n- For append-only logs: is `append_jsonl` properly flocked?\n- For json writes: is tmp-rename used for atomicity?\n\n**Verification method**: Construct a timeline with two concurrent operations, trace each thread's read and write points.\n\n## 3. UX & Implementation Logic (用户体验与实现逻辑)\n\n**Focus**: Whether the code actually implements what the docs/release-notes claim, and whether user-facing behavior is correct.\n\n**Key questions to probe**:\n- Do feature flags/modes actually enforce their claimed semantics? (e.g., \"review-only\" truly limits to reviews)\n- Are there dead code paths that silently swallow errors?\n- Are error messages actionable? Do they tell the user how to fix the problem?\n- Can users recover from mistakes? Is there undo/backtrack/revisit?\n- Are there implicit access control gaps? (e.g., no @-mention check in group chats)\n- Do test mocks hide real bugs? Check if mocked interfaces match actual call signatures\n- Are there misleading code comments that describe behavior not implemented?\n\n**Verification method**: Trace each code branch and confirm the actual behavior against the documented claim. For each error path, read the error message and assess if a user could act on it.\n\n## 4. Test Quality & Coverage Blind Spots (测试质量)\n\n**Focus**: Whether tests actually catch the bugs they claim to prevent.\n\n**Key questions to probe**:\n- Do integration tests use real dependencies or mocks? If mocked, do mocks validate call signatures?\n- Are there end-to-end tests in CI? If not, what's the smoke test strategy?\n- What bug classes are invisible to the current test suite?\n- Do test fixtures match production call signatures?\n\n**Verification method**: Read a representative sample of test files. Check if a wrong call signature or wrong return type would cause any test to fail.\n\n## 5. Simplicity & Over-Engineering (简洁性与过度工程) — NEW v1.1.0\n\n**Focus**: Whether the code's complexity is justified by the problem it solves. AI-generated code is prone to \"massive overkill\" — hundreds of lines with new services, patterns, and abstractions for what should be a small incremental change.\n\n**Background**: Community experience (Hacker News 2025 State of AI Code Quality discussion, Reddit r/ExperiencedDevs) consistently reports that AI coding agents over-engineer solutions. A \"batching\" request that should be 2 methods becomes a new service class + background worker + suite of tests. This dimension explicitly hunts for these patterns.\n\n**Key questions to probe**:\n- **AI-bloat check**: For each changed file, compute the ratio of \"conceptual complexity needed\" to \"lines of code added\". Does this PR add a new class/module/abstraction where a function would suffice?\n- **Abstraction audit**: Does the code introduce interfaces, factories, strategies, or dependency injection where direct calls would work? For each, ask: \"What concrete problem does this abstraction solve today, not hypothetically?\"\n- **Dead or speculative code**: Are there code paths, config keys, or extension points with zero current consumers? Flag \"future-proofing\" that isn't tied to a known roadmap item.\n- **Config sprawl**: Count new env vars, CLI flags, and config keys. Is each justified by a real user-facing need?\n- **Dependency footprint**: Count new third-party imports. Does every new dependency carry its weight?\n- **Rube Goldberg detection**: Could the same result be achieved with a simpler approach using existing infrastructure?\n\n**Verification method**: For each module changed, write a one-sentence description of \"what this module actually does.\" If that sentence is surprisingly short compared to the code volume, flag it.\n\n**Simplicity Score**: At the end of the Simplicity audit, assign a 1-5 score:\n- 5 = Elegantly minimal — every line earns its place\n- 4 = Clean — minor nitpicks but nothing egregious\n- 3 = Acceptable — some bloat but functional\n- 2 = Over-engineered — substantial unnecessary complexity\n- 1 = Massively bloated — requires refactoring before merge\n\nFile v1.1.0:references/four-eyes.md\n\n# Four-Eyes Cross-Verification Protocol\n\n## Origin\n\nBavota & Russo (2015), \"Four Eyes Are Better Than Two: On the Impact of Code Reviews on Software Quality\", found that multiple independent reviewers examining the same code catch 60%+ more real defects than a single reviewer. This protocol operationalizes that finding for AI subagent review.\n\n## Protocol\n\n### When to activate\n\nFour-Eyes cross-verification is **mandatory** for:\n- Any finding classified as **Critical** by a primary subagent\n- Any finding where the primary subagent reports \"confirmed with medium confidence\" or similar hedging\n\n### How to execute\n\n1. **Spawn a second subagent** with the **exact same issue prompt** as the primary\n2. The second subagent must have a **different dimension focus** than the primary (to avoid groupthink)\n3. The second subagent sees only the issue prompt — NOT the primary's findings (to ensure independence)\n4. Both reports are compared after completion\n\n### Resolution rules\n\n| Primary says | Second says | Final conclusion |\n|---|---|---|\n| Critical | Critical | **Confirmed** + `👁️ Four-Eyes Verified` badge |\n| Critical | High/Medium/Low | **Confirmed** at the higher severity. Add note: \"Second reviewer rated lower at [severity].\" |\n| Critical | Mitigated | **Mitigated** (the second found a defense the first missed). Keep both reports in appendix. |\n| Critical | False Alarm | **False Alarm** (the second found the issue doesn't exist). Keep both reports in appendix. |\n| Critical | Disputed (disagrees but can't reclassify) | **Disputed** — flag for human reviewer. Include both conclusions. |\n\n### Reporting format\n\nIn the final report, Critical findings carry:\n\n```\n### 🔴 Critical: [Issue Title]\n👁️ Four-Eyes Verified by [subagent-2-name]\n```\n\nDisputed findings carry:\n\n```\n### ⚠️ Disputed: [Issue Title]\n🔴 Subagent 1 ([dimension]): Critical — [summary]\n🟢 Subagent 2 ([dimension]): Mitigated — [defense found]\n→ Human review recommended.\n```\n\n## Cost/benefit guidance\n\n- **Small codebase (<20 files)**: Four-Eyes only on findings the auditor is uncertain about\n- **Medium codebase (20-100 files)**: Four-Eyes on all Critical findings\n- **Large/mission-critical codebase**: Four-Eyes on Critical + High findings\n\nThe additional cost is ~25-40% more subagent compute, but the false-positive reduction makes it worthwhile for release-blocking decisions.\n\nFile v1.1.0:references/output-format.md\n\n# Output Format Specification\n\n## Per-Issue Format\n\nEvery issue found must be reported with:\n\n```\n### 🔴/🟡/🟢 [Severity]: [Issue Title]\n\n**结论**: Confirmed / Mitigated / False Alarm / Disputed\n[If Critical + Four-Eyes verified: 👁️ Four-Eyes Verified by [subagent-name]]\n\n**源码证据**: [file:line range — the specific code that proves the issue]\n[Optional: quote the relevant code block]\n\n**风险场景**: [Concrete scenario — user does X → system does Y → consequence Z]\n\n**修复建议**: [Actionable, specific fix — not \"consider improving\"]\n```\n\n## Synthesis Format (after all subagent reports)\n\nAggregate findings into:\n\n### 1. Summary Table by Severity\n\n```\n| # | 问题 | 来源(审计维度) | 根因(一行) | 👁️ |\n|---|------|--------------|----------|-----|\n```\n`👁️` column: ✅ if Four-Eyes verified, blank otherwise.\n\n### 2. Priority Matrix\n\n```\n| 优先级 | 问题 | 工作量 | 影响范围 |\n|--------|------|--------|----------|\n| P0 🔴 | ... | 1行代码 | 所有用户 |\n| P1 🔴 | ... | ~30行 | 特定场景 |\n```\n\n### 3. Simplicity Score (NEW v1.1.0)\n\nA 1-5 rating of how well the codebase's complexity matches its problem domain:\n- **5/5** 🏆 — Elegantly minimal. Every line earns its place.\n- **4/5** ✅ — Clean. Minor nitpicks.\n- **3/5** ⚠️ — Acceptable. Some bloat but functional.\n- **2/5** ❌ — Over-engineered. Needs refactoring.\n- **1/5** 🚨 — Massively bloated. Do not merge.\n\n### 4. Executive Summary\n\nOne paragraph that captures the overall quality assessment + top 3 action items.\n\n## Emoji Convention\n\n- 🔴 Critical — System violates core guarantees\n- 🔴 High — Significant impact, fix before next release\n- 🟡 Medium — Degrades UX or creates operational risk\n- 🟢 Low — Cosmetic, theoretical, or well-mitigated\n- ✅ Mitigated — Concern exists but defended elsewhere\n- ❌ False Alarm — Concern does not exist\n- ⚠️ Disputed — Four-Eyes reviewers disagree (needs human)\n- 👁️ Four-Eyes Verified — Independently confirmed by a second subagent\n\nUse the emoji in the severity tag, not the conclusion tag.\n\nFile v1.1.0:references/severity-rubric.md\n\n# Severity Classification Rubric\n\nEvery confirmed issue gets exactly one severity level.\n\n## Critical\n\n**Definition**: A bug that causes the system to violate its core security or correctness guarantees, with no workaround.\n\n**Examples**:\n- Security boundary bypass that leaks private data\n- Feature flag that doesn't actually limit the feature it's named after\n- Data corruption that silently destroys user data\n- **Over-engineering so severe it makes the codebase unmaintainable** (e.g., 500 lines of new abstraction for a one-line bugfix)\n\n**Litmus test**: Would a reasonable user be shocked to learn this exists in production?\n\n## High\n\n**Definition**: A bug with significant security, correctness, or reliability impact, but with a partial workaround or limited blast radius.\n\n**Examples**:\n- SSRF vulnerability that requires attacker-controlled input\n- Missing @-mention check in group chats (abuse requires insider knowledge)\n- Concurrent race that requires specific timing to trigger\n\n**Litmus test**: Would you fix this before the next release?\n\n## Medium\n\n**Definition**: A bug that degrades user experience, creates operational risk, or has been partially mitigated.\n\n**Examples**:\n- Missing backtrack/revisit mechanism in a multi-step workflow\n- Error message that doesn't guide the user to the fix\n- Flock implemented but with gaps (correct on happy path, missing on edge path)\n- Race condition that wastes resources but doesn't corrupt data\n\n**Litmus test**: Would you prioritize fixing this in the next sprint?\n\n## Low\n\n**Definition**: Cosmetic issues, theoretical concerns, or already well-mitigated risks.\n\n**Examples**:\n- DNS rebinding theory window with high exploit difficulty\n- Token refresh margin that's overly conservative (no harm)\n- Function rename needed for code clarity\n\n**Litmus test**: Would you fix this only if you happened to be editing that file anyway?\n\n## Mitigated\n\n**Definition**: The reported concern exists in some form but is defended against by another mechanism.\n\n**Examples**:\n- Claimed SSRF vulnerability that's actually handled by a wrapper\n- Alleged race condition where an upstream lock serializes access\n- Reported missing check that actually happens at a different layer\n\n## False Alarm\n\n**Definition**: The reported concern does not exist in the codebase. No defense is needed because no attack path exists.\n\n**Examples**:\n- Reported command injection where no subprocess is used\n- Claimed token leak where no logging of sensitive data exists\n\nFile v1.1.0:references/subagent-templates.md\n\n# Subagent Audit Templates\n\nEach dimension gets its own subagent. Use these template structures when spawning.\n\n## Template Structure\n\n```markdown\n你是 [项目名] [版本号] 的 [审计维度]审计员。请仔细审计以下问题，从源码层面 double-check 每个问题是否真实存在。每个问题必须给出明确结论（Confirmed / Mitigated / False Alarm）。\n\n**背景**: [项目简介、运行环境、关键约束]\n\n## 问题 #N: [问题标题]\n\n[问题描述 — 包含具体的怀疑、预期 vs 实际、源码引用]\n\n请核实:\n- 阅读 [具体源文件 URL]\n- [具体的检查点 1]\n- [具体的检查点 2]\n- [判断标准]\n\n## 输出格式\n\n对每个问题给出:\n1. **结论**: Confirmed / Mitigated / False Alarm\n2. **严重程度**: Critical / High / Medium / Low\n3. **源码证据**: 引用具体行/函数/文件\n4. **风险描述**: 简明说明\n5. **修复建议**: 具体可操作的方案\n\n用中文输出。必须先 fetch 每个相关源文件确认实际代码。\n```\n\n## Template for Simplicity & Over-Engineering Subagent (NEW v1.1.0)\n\n```markdown\n你是 [项目名] [版本号] 的简洁性审计员。你的工作不是找 bug——而是判断代码是否\"过度工程化\"。\n\n**核心问题**: 这段代码的复杂度是否匹配它要解决的问题？\n\n**审计方法**:\n1. 读取每个被修改的文件\n2. 用一句话总结\"这个模块实际做了什么\"\n3. 如果一句话总结过于简短而代码量巨大 → 这是过度工程\n4. 检查每一个新引入的抽象层（接口/工厂/策略模式），问：它今天解决什么具体问题？\n5. 检查新依赖、新配置项、新CLI参数——每个都需要具体的使用场景\n6. 检查是否有\"未来扩展点\"没有当前消费者\n\n**输出**:\n1. **简洁性评分**: 1-5 (5=极简优雅, 1=大规模过度工程)\n2. **问题清单**: 每个过度工程问题 + 具体的简化建议\n3. **对比**: \"这段代码实现了 [X]；用更简单的方式可以只用 [Y] 行\"\n\n用中文输出。引用具体文件行。\n```\n\n## Subagent Quality Rules\n\n1. **Every claim must cite a source line**. Never say \"the code does X\" without referencing a file:line.\n2. **Fetch before judging**. Always `web_fetch` the actual source file; never infer from release notes or documentation.\n3. **Default to Confirmed/Mitigated/False Alarm**. Never leave a finding ambiguous.\n4. **If truncated, re-fetch**. If a source file exceeds the fetch limit, fetch with offset.\n5. **Construct exploit scenarios**. For security bugs, trace the full attack path with concrete URLs/IPs.\n6. **Construct race timelines**. For concurrency bugs, show reads and writes from each thread at each time step.\n7. **Estimate complexity delta**. For over-engineering findings, show the line-count if done simply.\n\n## Subagent Count Guidelines\n\n- **Small codebase (<20 files, single module)**: 2-3 subagents (combine dimensions)\n- **Medium codebase (20-100 files, 2-5 modules)**: 3-4 subagents\n- **Large codebase (100+ files, monorepo)**: 4-5 subagents (one per dimension)\n\nAssign dimensions based on the PR's content — not every audit needs all five dimensions. Skip Simplicity for refactoring PRs. Skip Concurrency for single-threaded CLI tools.\n\nArchive v1.0.0: 6 files, 8767 bytes\n\nFiles: references/audit-dimensions.md (3411b), references/output-format.md (1426b), references/severity-rubric.md (2366b), references/subagent-templates.md (1971b), SKILL.md (5779b), _meta.json (135b)\n\nFile v1.0.0:SKILL.md\n\n---\nname: code-review\ndescription: >\n  Multi-dimensional code audit using structured subagent delegation.\n  Use when reviewing a GitHub release, PR, or codebase.\n  Systematically inspects security, concurrency/state-machine safety, UX/implementation logic, and test quality.\n  Spawns parallel subagents for deep verification, then synthesizes findings into a Confirmed/Critical-to-Low priority matrix.\n  Trigger phrases: review this release, audit this codebase, check this PR for issues, 代码审查, review 代码, 审查这个版本\n---\n\n# Code Review — Multi-Dimensional Audit Methodology\n\nSystematically audit a codebase release through four dimensions, using parallel subagent delegation for deep verification.\n\n## Core Principle\n\n**Real code, not release notes.** Every finding must be verified against actual source files by fetching them. The only acceptable evidence is `file:line` citations. The only acceptable conclusion labels are `Confirmed / Mitigated / False Alarm`.\n\n## Workflow\n\n### Phase 1: Surface Scan (in main session)\n\nRead these in order — enough to understand architecture and identify candidate issues:\n\n1. **Release notes / CHANGELOG** — what the authors claim changed\n2. **README** — project purpose, architecture diagram, on-disk layout\n3. **ARCHITECTURE.md** or equivalent — module decomposition, API contracts\n4. **Directory tree** (via GitHub tree view) — file listing to map modules\n5. **Key source files** — entry point, core state machine, critical paths (read ~3-8 files)\n\n**Output**: A list of 10-20 candidate issues, categorized by dimension:\n- Security (SSRF, injection, auth, path traversal)\n- Concurrency & State Machine (race conditions, missing locks, TOCTOU)\n- UX & Implementation Logic (feature semantics, error messages, recovery paths)\n- Test Quality (mock fidelity, integration gaps, signature mismatches)\n\n### Phase 2: Deep Audit (via subagents)\n\nFor each non-trivial dimension, spawn an isolated subagent. Each subagent:\n\n1. **Fetches every relevant source file** via `web_fetch` — never infers from docs\n2. **Verifies each issue against actual code** — cites specific lines\n3. **Constructs exploit scenarios** (security) or **race timelines** (concurrency)\n4. **Returns structured findings** with: Conclusion / Severity / Source Evidence / Risk / Fix\n\nSee [subagent-templates.md](references/subagent-templates.md) for the exact prompt template.\nSee [audit-dimensions.md](references/audit-dimensions.md) for dimension-specific question probes.\n\n**Model guidance**: Use the same model for all subagents to ensure consistent judgment. Prefer high-reasoning models for complex audits.\n\n### Phase 3: Synthesis (in main session)\n\nWhen all subagent reports return:\n\n1. **Merge findings** — deduplicate across dimensions, re-classify severity\n2. **Build summary table** — all issues with conclusion + severity + source dimension + root cause\n3. **Build priority matrix** — P0 (drop everything) through P5 (nice to have), with estimated work and blast radius\n4. **Write executive summary** — overall quality assessment + top 3 action items\n\nSee [output-format.md](references/output-format.md) for table and emoji conventions.\nSee [severity-rubric.md](references/severity-rubric.md) for severity classification rules.\n\n## Key Heuristics\n\n### Security Scan Heuristics\n\n- **Every URL fetch path must be checked for SSRF**: trace from user input → URL parsing → DNS resolution → HTTP request → redirect handling → response reading. Flag any step that skips IP validation.\n- **Every subprocess call must be checked for injection**: is `shell=True` used? Are user-controlled strings concatenated into the command? Are file paths sanitized?\n- **Every external API call must be checked for credential leaks**: are tokens/secrets logged? Do error messages include request bodies?\n\n### Concurrency Scan Heuristics\n\n- **For every `.json` / `.jsonl` write**: check if it uses tmp-rename atomic pattern or flock. Direct overwrite without either = bug.\n- **For every `load → modify → save` pattern**: check if the entire block is lock-protected. If load happens outside the lock, it's a TOCTOU bug.\n- **For every state machine transition**: check if two concurrent events can both see the same \"before\" state and both advance. If yes, state corruption possible.\n- **For every append-only log**: verify flock(LOCK_EX) covers the full append operation.\n\n### UX/Logic Scan Heuristics\n\n- **For every feature flag/mode**: trace all branches. Does \"mode=review-only\" actually prevent non-review actions? Don't trust the name — verify the code.\n- **For every error message**: read it as a user would. Does it tell you what went wrong AND how to fix it? If it only says \"X failed\", flag it.\n- **For every multi-step workflow**: is there an undo/backtrack/revisit path? If not, flag it.\n- **For every access-control check**: look for what's NOT checked. Does a group chat require @-mention? Does a rate limit exist?\n\n### Test Quality Heuristics\n\n- **Mocks that match wrong signatures**: if a test monkeypatches `call_llm` with a fake that takes `**kw` and reads `kw.get(\"old_param\")`, it will never catch a production code change to `new_param`. Flag these.\n- **No integration test in CI**: if CI only runs unit tests with mocks and there's no end-to-end smoke test, flag it.\n\n## Anti-Patterns (avoid)\n\n- ❌ Filing an issue based on release notes alone (always verify against source)\n- ❌ Accepting a docstring claim without checking the implementation\n- ❌ Using \"I think\" / \"probably\" / \"seems like\" — every finding is Confirmed or it's not a finding\n- ❌ Leaving severity as \"TBD\" — classify immediately using the rubric\n- ❌ Mentioning an issue in prose without filing it in the structured output table\n\nFile v1.0.0:_meta.json\n\n{\n  \"ownerId\": \"kn7b56gp3gfpjr0n2jxr3hkfyn82cnkm\",\n  \"slug\": \"deep-code-review\",\n  \"version\": \"1.0.0\",\n  \"publishedAt\": 1778821555285\n}\n\nFile v1.0.0:references/audit-dimensions.md\n\n# Audit Dimensions\n\nEach codebase audit spans these four dimensions. For each dimension, spawn a dedicated subagent that fetches source files and verifies issues against actual code.\n\n## 1. Security (安全审计)\n\n**Focus**: Vulnerabilities that allow unauthorized access, data exfiltration, or resource abuse.\n\n**Key questions to probe**:\n- Are there SSRF attack vectors? Check URL fetching/redirect logic, DNS resolution, IP filtering\n- Are there command injection risks? Search for `subprocess`, `shell=True`, unchecked file path concatenation\n- Are there auth token leaks? Check logging, error messages, HTTP response handling\n- Are there path traversal risks? Check file operations with user-controlled paths\n- Are third-party dependencies validated? Check version pinning, integrity checks\n\n**Verification method**: Trace the full attack path from user input to exploit. For each defense claim in the code, verify it actually works at the call site.\n\n## 2. Concurrency & State Machine Safety (并发与状态机安全)\n\n**Focus**: Race conditions, data corruption, inconsistent state in multi-thread/multi-process environments.\n\n**Key questions to probe**:\n- Are file locks (flock/fcntl) used for shared state? Is the lock scope correct (covers read-modify-write)?\n- Are state transitions atomic? Check `load → modify → save` and `load → transition → save` patterns\n- Are there TOCTOU (Time-of-check-time-of-use) gaps? Gap between check and action\n- Are there duplicate definitions of the same named exception/class?\n- Is there any lock-free concurrent write to the same file?\n- For append-only logs: is `append_jsonl` properly flocked?\n- For json writes: is tmp-rename used for atomicity?\n\n**Verification method**: Construct a timeline with two concurrent operations, trace each thread's read and write points.\n\n## 3. UX & Implementation Logic (用户体验与实现逻辑)\n\n**Focus**: Whether the code actually implements what the docs/release-notes claim, and whether user-facing behavior is correct.\n\n**Key questions to probe**:\n- Do feature flags/modes actually enforce their claimed semantics? (e.g., \"review-only\" truly limits to reviews)\n- Are there dead code paths that silently swallow errors?\n- Are error messages actionable? Do they tell the user how to fix the problem?\n- Can users recover from mistakes? Is there undo/backtrack/revisit?\n- Are there implicit access control gaps? (e.g., no @-mention check in group chats)\n- Do test mocks hide real bugs? Check if mocked interfaces match actual call signatures\n- Are there misleading code comments that describe behavior not implemented?\n\n**Verification method**: Trace each code branch and confirm the actual behavior against the documented claim. For each error path, read the error message and assess if a user could act on it.\n\n## 4. Test Quality & Coverage Blind Spots (测试质量)\n\n**Focus**: Whether tests actually catch the bugs they claim to prevent.\n\n**Key questions to probe**:\n- Do integration tests use real dependencies or mocks? If mocked, do mocks validate call signatures?\n- Are there end-to-end tests in CI? If not, what's the smoke test strategy?\n- What bug classes are invisible to the current test suite?\n- Do test fixtures match production call signatures?\n\n**Verification method**: Read a representative sample of test files. Check if a wrong call signature or wrong return type would cause any test to fail.\n\nFile v1.0.0:references/output-format.md\n\n# Output Format Specification\n\n## Per-Issue Format\n\nEvery issue found must be reported with:\n\n```\n### 🔴/🟡/🟢 [Severity]: [Issue Title]\n\n**结论**: Confirmed / Mitigated / False Alarm\n\n**源码证据**: [file:line range — the specific code that proves the issue]\n[Optional: quote the relevant code block]\n\n**风险场景**: [Concrete scenario — user does X → system does Y → consequence Z]\n\n**修复建议**: [Actionable, specific fix — not \"consider improving\"]\n```\n\n## Synthesis Format (after all subagent reports)\n\nAggregate findings into:\n\n### 1. Summary Table by Severity\n\n```\n| # | 问题 | 来源(审计维度) | 根因(一行) |\n|---|------|--------------|----------|\n```\n\n### 2. Priority Matrix\n\n```\n| 优先级 | 问题 | 工作量 | 影响范围 |\n|--------|------|--------|----------|\n| P0 🔴 | ... | 1行代码 | 所有用户 |\n| P1 🔴 | ... | ~30行 | 特定场景 |\n```\n\n### 3. Executive Summary\n\nOne paragraph that captures the overall quality assessment + top 3 action items.\n\n## Emoji Convention\n\n- 🔴 Critical — System violates core guarantees\n- 🔴 High — Significant impact, fix before next release\n- 🟡 Medium — Degrades UX or creates operational risk\n- 🟢 Low — Cosmetic, theoretical, or well-mitigated\n- ✅ Mitigated — Concern exists but defended elsewhere\n- ❌ False Alarm — Concern does not exist\n\nUse the emoji in the severity tag, not the conclusion tag.\n\nFile v1.0.0:references/severity-rubric.md\n\n# Severity Classification Rubric\n\nEvery confirmed issue gets exactly one severity level.\n\n## Critical\n\n**Definition**: A bug that causes the system to violate its core security or correctness guarantees, with no workaround.\n\n**Examples**:\n- Security boundary bypass that leaks private data\n- Feature flag that doesn't actually limit the feature it's named after\n- Data corruption that silently destroys user data\n\n**Litmus test**: Would a reasonable user be shocked to learn this exists in production?\n\n## High\n\n**Definition**: A bug with significant security, correctness, or reliability impact, but with a partial workaround or limited blast radius.\n\n**Examples**:\n- SSRF vulnerability that requires attacker-controlled input\n- Missing @-mention check in group chats (abuse requires insider knowledge)\n- Concurrent race that requires specific timing to trigger\n\n**Litmus test**: Would you fix this before the next release?\n\n## Medium\n\n**Definition**: A bug that degrades user experience, creates operational risk, or has been partially mitigated.\n\n**Examples**:\n- Missing backtrack/revisit mechanism in a multi-step workflow\n- Error message that doesn't guide the user to the fix\n- Flock implemented but with gaps (correct on happy path, missing on edge path)\n- Race condition that wastes resources but doesn't corrupt data\n\n**Litmus test**: Would you prioritize fixing this in the next sprint?\n\n## Low\n\n**Definition**: Cosmetic issues, theoretical concerns, or already well-mitigated risks.\n\n**Examples**:\n- DNS rebinding theory window with high exploit difficulty\n- Token refresh margin that's overly conservative (no harm)\n- Function rename needed for code clarity\n\n**Litmus test**: Would you fix this only if you happened to be editing that file anyway?\n\n## Mitigated\n\n**Definition**: The reported concern exists in some form but is defended against by another mechanism.\n\n**Examples**:\n- Claimed SSRF vulnerability that's actually handled by a wrapper\n- Alleged race condition where an upstream lock serializes access\n- Reported missing check that actually happens at a different layer\n\n## False Alarm\n\n**Definition**: The reported concern does not exist in the codebase. No defense is needed because no attack path exists.\n\n**Examples**:\n- Reported command injection where no subprocess is used\n- Claimed token leak where no logging of sensitive data exists\n\nFile v1.0.0:references/subagent-templates.md\n\n# Subagent Audit Templates\n\nEach dimension gets its own subagent. Use these template structures when spawning.\n\n## Template Structure\n\n```markdown\n你是 [项目名] [版本号] 的 [审计维度]审计员。请仔细审计以下问题，从源码层面 double-check 每个问题是否真实存在。每个问题必须给出明确结论（Confirmed / Mitigated / False Alarm）。\n\n**背景**: [项目简介、运行环境、关键约束]\n\n## 问题 #N: [问题标题]\n\n[问题描述 — 包含具体的怀疑、预期 vs 实际、源码引用]\n\n请核实:\n- 阅读 [具体源文件 URL]\n- [具体的检查点 1]\n- [具体的检查点 2]\n- [判断标准]\n\n## 输出格式\n\n对每个问题给出:\n1. **结论**: Confirmed / Mitigated / False Alarm\n2. **严重程度**: Critical / High / Medium / Low\n3. **源码证据**: 引用具体行/函数/文件\n4. **风险描述**: 简明说明\n5. **修复建议**: 具体可操作的方案\n\n用中文输出。必须先 fetch 每个相关源文件确认实际代码。\n```\n\n## Subagent Quality Rules\n\n1. **Every claim must cite a source line**. Never say \"the code does X\" without referencing a file:line.\n2. **Fetch before judging**. Always `web_fetch` the actual source file; never infer from release notes or documentation.\n3. **Default to Confirmed/Fixed/Mitigated/False Alarm**. Never leave a finding ambiguous.\n4. **If truncated, re-fetch**. If a source file exceeds the fetch limit, fetch with offset.\n5. **Construct exploit scenarios**. For security bugs, trace the full attack path with concrete URLs/IPs.\n6. **Construct race timelines**. For concurrency bugs, show reads and writes from each thread at each time step.\n\n## Subagent Count Guidelines\n\n- **Small codebase (<20 files, single module)**: 1-2 subagents\n- **Medium codebase (20-100 files, 2-5 modules)**: 2-3 subagents\n- **Large codebase (100+ files, monorepo)**: 3-4 subagents\n\nSplit by dimension, not by module. Each subagent should have a clear thematic focus.","readmeExcerpt":"Skill: Code Review — Multi-Dimensional Audit Owner: yinghaojia Summary: Multi-dimensional code audit using structured subagent delegation. Use when reviewing a GitHub release, PR, or codebase. Systematically inspects security, co... Tags: audit:1.1.1, code-review:1.1.1, concurrency:1.1.1, latest:1.1.1, security:1.1.1, simplicity:1.1.1 Version history: v1.1.1 | 2026-05-15T05:18:14.034Z | user v1.1.1: Added /deep-code-","codeSnippets":[],"executableExamples":[{"language":"text","snippet":"### 🔴 Critical: [Issue Title]\n👁️ Four-Eyes Verified by [subagent-2-name]"},{"language":"text","snippet":"### ⚠️ Disputed: [Issue Title]\n🔴 Subagent 1 ([dimension]): Critical — [summary]\n🟢 Subagent 2 ([dimension]): Mitigated — [defense found]\n→ Human review recommended."},{"language":"text","snippet":"### 🔴/🟡/🟢 [Severity]: [Issue Title]\n\n**结论**: Confirmed / Mitigated / False Alarm / Disputed\n[If Critical + Four-Eyes verified: 👁️ Four-Eyes Verified by [subagent-name]]\n\n**源码证据**: [file:line range — the specific code that proves the issue]\n[Optional: quote the relevant code block]\n\n**风险场景**: [Concrete scenario — user does X → system does Y → consequence Z]\n\n**修复建议**: [Actionable, specific fix — not \"consider improving\"]"},{"language":"text","snippet":"| # | 问题 | 来源(审计维度) | 根因(一行) | 👁️ |\n|---|------|--------------|----------|-----|"},{"language":"text","snippet":"| 优先级 | 问题 | 工作量 | 影响范围 |\n|--------|------|--------|----------|\n| P0 🔴 | ... | 1行代码 | 所有用户 |\n| P1 🔴 | ... | ~30行 | 特定场景 |"},{"language":"markdown","snippet":"你是 [项目名] [版本号] 的 [审计维度]审计员。请仔细审计以下问题，从源码层面 double-check 每个问题是否真实存在。每个问题必须给出明确结论（Confirmed / Mitigated / False Alarm）。\n\n**背景**: [项目简介、运行环境、关键约束]\n\n## 问题 #N: [问题标题]\n\n[问题描述 — 包含具体的怀疑、预期 vs 实际、源码引用]\n\n请核实:\n- 阅读 [具体源文件 URL]\n- [具体的检查点 1]\n- [具体的检查点 2]\n- [判断标准]\n\n## 输出格式\n\n对每个问题给出:\n1. **结论**: Confirmed / Mitigated / False Alarm\n2. **严重程度**: Critical / High / Medium / Low\n3. **源码证据**: 引用具体行/函数/文件\n4. **风险描述**: 简明说明\n5. **修复建议**: 具体可操作的方案\n\n用中文输出。必须先 fetch 每个相关源文件确认实际代码。"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: code-review\ndescription: >\n  Multi-dimensional code audit using structured subagent delegation.\n  Use when reviewing a GitHub release, PR, or codebase.\n  Systematically inspects security, concurrency/state-machine safety, UX/implementation logic, test quality, and simplicity/over-engineering.\n  Spawns parallel subagents for deep verification with Four-Eyes cross-validation on critical findings.\n  Synthesizes findings into a Confirmed/Critical-to-Low priority matrix.\n  Trigger phrases: review this release, audit this codebase, check this PR for issues, 代码审查, review 代码, 审查这个版本, /deep-code-review, /code-review, /review-code\n---\n\n# Code Review — Multi-Dimensional Audit Methodology\n\nSystematically audit a codebase release through five dimensions, using parallel subagent delegation for deep verification.\nInspired by: Modern Code Review taxonomy research (Bavota & Russo 2015 \"Four Eyes Are Better Than Two\"), reviewdog's tool-agnostic harness pattern, Danger's pre-review gate philosophy, and community experience with AI-generated code quality issues.\n\n## Core Principles\n\n1. **Real code, not release notes.** Every finding must be verified against actual source files by fetching them. The only acceptable evidence is `file:line` citations. The only acceptable conclusion labels are `Confirmed / Mitigated / False Alarm`.\n\n2. **Four Eyes on every Critical.** Any finding classified as Critical severity MUST be independently verified by a second subagent before appearing in the final report. This is the \"Four Eyes\" principle from Bavota & Russo (2015): multiple reviewers independently examining the same issue catch 60%+ more real bugs than a single reviewer. See [four-eyes.md](references/four-eyes.md).\n\n3. **Simplicity is a first-class dimension.** AI-generated code often produces \"massive overkill\" — hundreds of lines for what should be a two-method change. Always ask: \"Does the complexity of this solution match the complexity of the problem?\" This dimension is inspired by community experience on Hacker News and Reddit (2025 State of AI Code Quality discussions).\n\n## Workflow\n\n### Phase 0: Pre-Review Gate (in main session, <2 min)\n\nRun these quick checks before committing to a full audit. Inspired by Danger's \"automated pre-review\" philosophy.\n\n1. **PR/Diff size check**: if the change exceeds 400 lines, flag it as high-risk and recommend splitting\n2. **Missing artifacts**: is there a CHANGELOG entry? Updated README if API changed? Migration guide if schema changed?\n3. **File-level red flags**: any committed `.env`, credentials, large binary files?\n4. **Test presence**: does this change include or update tests? If zero test changes on a >100 line diff, flag.\n\n**Output**: Gate report (pass/warn/fail) + recommended audit depth.\n\n### Phase 1: Surface Scan (in main session)\n\nRead these in order — enough to understand architecture and identify candidate issues:\n\n1. **Release notes / CHANGELOG** — what the authors claim changed\n2. **README** — project purpos"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn7b56gp3gfpjr0n2jxr3hkfyn82cnkm\",\n  \"slug\": \"deep-code-review\",\n  \"version\": \"1.1.1\",\n  \"publishedAt\": 1778822294034\n}"},{"path":"references/audit-dimensions.md","content":"# Audit Dimensions\n\nEach codebase audit spans these five dimensions. For each dimension, spawn a dedicated subagent that fetches source files and verifies issues against actual code.\n\n## 1. Security (安全审计)\n\n**Focus**: Vulnerabilities that allow unauthorized access, data exfiltration, or resource abuse.\n\n**Key questions to probe**:\n- Are there SSRF attack vectors? Check URL fetching/redirect logic, DNS resolution, IP filtering\n- Are there command injection risks? Search for `subprocess`, `shell=True`, unchecked file path concatenation\n- Are there auth token leaks? Check logging, error messages, HTTP response handling\n- Are there path traversal risks? Check file operations with user-controlled paths\n- Are third-party dependencies validated? Check version pinning, integrity checks\n\n**Verification method**: Trace the full attack path from user input to exploit. For each defense claim in the code, verify it actually works at the call site.\n\n## 2. Concurrency & State Machine Safety (并发与状态机安全)\n\n**Focus**: Race conditions, data corruption, inconsistent state in multi-thread/multi-process environments.\n\n**Key questions to probe**:\n- Are file locks (flock/fcntl) used for shared state? Is the lock scope correct (covers read-modify-write)?\n- Are state transitions atomic? Check `load → modify → save` and `load → transition → save` patterns\n- Are there TOCTOU (Time-of-check-time-of-use) gaps? Gap between check and action\n- Are there duplicate definitions of the same named exception/class?\n- Is there any lock-free concurrent write to the same file?\n- For append-only logs: is `append_jsonl` properly flocked?\n- For json writes: is tmp-rename used for atomicity?\n\n**Verification method**: Construct a timeline with two concurrent operations, trace each thread's read and write points.\n\n## 3. UX & Implementation Logic (用户体验与实现逻辑)\n\n**Focus**: Whether the code actually implements what the docs/release-notes claim, and whether user-facing behavior is correct.\n\n**Key questions to probe**:\n- Do feature flags/modes actually enforce their claimed semantics? (e.g., \"review-only\" truly limits to reviews)\n- Are there dead code paths that silently swallow errors?\n- Are error messages actionable? Do they tell the user how to fix the problem?\n- Can users recover from mistakes? Is there undo/backtrack/revisit?\n- Are there implicit access control gaps? (e.g., no @-mention check in group chats)\n- Do test mocks hide real bugs? Check if mocked interfaces match actual call signatures\n- Are there misleading code comments that describe behavior not implemented?\n\n**Verification method**: Trace each code branch and confirm the actual behavior against the documented claim. For each error path, read the error message and assess if a user could act on it.\n\n## 4. Test Quality & Coverage Blind Spots (测试质量)\n\n**Focus**: Whether tests actually catch the bugs they claim to prevent.\n\n**Key questions to probe**:\n- Do integration tests use real dependencies or mocks? If mocked, do mocks validate call s"},{"path":"references/four-eyes.md","content":"# Four-Eyes Cross-Verification Protocol\n\n## Origin\n\nBavota & Russo (2015), \"Four Eyes Are Better Than Two: On the Impact of Code Reviews on Software Quality\", found that multiple independent reviewers examining the same code catch 60%+ more real defects than a single reviewer. This protocol operationalizes that finding for AI subagent review.\n\n## Protocol\n\n### When to activate\n\nFour-Eyes cross-verification is **mandatory** for:\n- Any finding classified as **Critical** by a primary subagent\n- Any finding where the primary subagent reports \"confirmed with medium confidence\" or similar hedging\n\n### How to execute\n\n1. **Spawn a second subagent** with the **exact same issue prompt** as the primary\n2. The second subagent must have a **different dimension focus** than the primary (to avoid groupthink)\n3. The second subagent sees only the issue prompt — NOT the primary's findings (to ensure independence)\n4. Both reports are compared after completion\n\n### Resolution rules\n\n| Primary says | Second says | Final conclusion |\n|---|---|---|\n| Critical | Critical | **Confirmed** + `👁️ Four-Eyes Verified` badge |\n| Critical | High/Medium/Low | **Confirmed** at the higher severity. Add note: \"Second reviewer rated lower at [severity].\" |\n| Critical | Mitigated | **Mitigated** (the second found a defense the first missed). Keep both reports in appendix. |\n| Critical | False Alarm | **False Alarm** (the second found the issue doesn't exist). Keep both reports in appendix. |\n| Critical | Disputed (disagrees but can't reclassify) | **Disputed** — flag for human reviewer. Include both conclusions. |\n\n### Reporting format\n\nIn the final report, Critical findings carry:\n\n```\n### 🔴 Critical: [Issue Title]\n👁️ Four-Eyes Verified by [subagent-2-name]\n```\n\nDisputed findings carry:\n\n```\n### ⚠️ Disputed: [Issue Title]\n🔴 Subagent 1 ([dimension]): Critical — [summary]\n🟢 Subagent 2 ([dimension]): Mitigated — [defense found]\n→ Human review recommended.\n```\n\n## Cost/benefit guidance\n\n- **Small codebase (<20 files)**: Four-Eyes only on findings the auditor is uncertain about\n- **Medium codebase (20-100 files)**: Four-Eyes on all Critical findings\n- **Large/mission-critical codebase**: Four-Eyes on Critical + High findings\n\nThe additional cost is ~25-40% more subagent compute, but the false-positive reduction makes it worthwhile for release-blocking decisions."},{"path":"references/output-format.md","content":"# Output Format Specification\n\n## Per-Issue Format\n\nEvery issue found must be reported with:\n\n```\n### 🔴/🟡/🟢 [Severity]: [Issue Title]\n\n**结论**: Confirmed / Mitigated / False Alarm / Disputed\n[If Critical + Four-Eyes verified: 👁️ Four-Eyes Verified by [subagent-name]]\n\n**源码证据**: [file:line range — the specific code that proves the issue]\n[Optional: quote the relevant code block]\n\n**风险场景**: [Concrete scenario — user does X → system does Y → consequence Z]\n\n**修复建议**: [Actionable, specific fix — not \"consider improving\"]\n```\n\n## Synthesis Format (after all subagent reports)\n\nAggregate findings into:\n\n### 1. Summary Table by Severity\n\n```\n| # | 问题 | 来源(审计维度) | 根因(一行) | 👁️ |\n|---|------|--------------|----------|-----|\n```\n`👁️` column: ✅ if Four-Eyes verified, blank otherwise.\n\n### 2. Priority Matrix\n\n```\n| 优先级 | 问题 | 工作量 | 影响范围 |\n|--------|------|--------|----------|\n| P0 🔴 | ... | 1行代码 | 所有用户 |\n| P1 🔴 | ... | ~30行 | 特定场景 |\n```\n\n### 3. Simplicity Score (NEW v1.1.0)\n\nA 1-5 rating of how well the codebase's complexity matches its problem domain:\n- **5/5** 🏆 — Elegantly minimal. Every line earns its place.\n- **4/5** ✅ — Clean. Minor nitpicks.\n- **3/5** ⚠️ — Acceptable. Some bloat but functional.\n- **2/5** ❌ — Over-engineered. Needs refactoring.\n- **1/5** 🚨 — Massively bloated. Do not merge.\n\n### 4. Executive Summary\n\nOne paragraph that captures the overall quality assessment + top 3 action items.\n\n## Emoji Convention\n\n- 🔴 Critical — System violates core guarantees\n- 🔴 High — Significant impact, fix before next release\n- 🟡 Medium — Degrades UX or creates operational risk\n- 🟢 Low — Cosmetic, theoretical, or well-mitigated\n- ✅ Mitigated — Concern exists but defended elsewhere\n- ❌ False Alarm — Concern does not exist\n- ⚠️ Disputed — Four-Eyes reviewers disagree (needs human)\n- 👁️ Four-Eyes Verified — Independently confirmed by a second subagent\n\nUse the emoji in the severity tag, not the conclusion tag."}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Multi-dimensional code audit using structured subagent delegation. Use when reviewing a GitHub release, PR, or codebase. Systematically inspects security, co... Skill: Code Review — Multi-Dimensional Audit Owner: yinghaojia Summary: Multi-dimensional code audit using structured subagent delegation. Use when reviewing a GitHub release, PR, or codebase. Systematically inspects security, co... Tags: audit:1.1.1, code-review:1.1.1, concurrency:1.1.1, latest:1.1.1, security:1.1.1, simplicity:1.1.1 Version history: v1.1.1 | 2026-05-15T05:18:14.034Z | user v1.1.1: Added /deep-code-","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1808,"uniquenessScore":47,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-11T16:28:51.797Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-11T16:28:51.797Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-11T20:57:02.904Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}