{"id":"1b798a02-6a94-4c76-9969-426bdf8fb471","entityType":"agent","slug":"clawhub-athola-nm-pensive-test-review","name":"test-review","canonicalUrl":"https://www.xpersona.co/agent/clawhub-athola-nm-pensive-test-review","canonicalPath":"/agent/clawhub-athola-nm-pensive-test-review","generatedAt":"2026-10-10T10:49:59.034Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-10T07:58:53.992Z","emptyReason":null},"description":"Evaluates test suites for coverage gaps, TDD/BDD compliance, and anti-patterns Skill: test-review Owner: athola Summary: Evaluates test suites for coverage gaps, TDD/BDD compliance, and anti-patterns Tags: latest:1.9.19 Version history: v1.9.19 | 2026-08-26T13:19:33.885Z | user Release v1.9.19 v1.9.17 | 2026-07-30T05:39:43.732Z | user Release v1.9.17 v1.9.16 | 2026-07-14T19:56:28.437Z | user Release v1.9.16 v1.9.14 | 2026-06-30T18:04:41.848Z | user Release v1.9.14 v1.9.13 | 2026-06-27T16:22:37.","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.6K downloads reported by the source. Last updated 10/10/2026.","installCommand":"clawhub skill install s17emme0e2m3cpf7k2jvp3a84984b8z9:nm-pensive-test-review","sourceUrl":"https://clawhub.ai/athola/nm-pensive-test-review","homepage":"https://clawhub.ai/athola/skills/nm-pensive-test-review","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/athola/nm-pensive-test-review","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/athola/skills/nm-pensive-test-review","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":64,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Evaluates test suites for coverage gaps, TDD/BDD compliance, and anti-patterns Skill: test-review Owner: athola Summary: Evaluates test suites for coverage gaps"},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-10T07:58:53.992Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T07:58:53.992Z","emptyReason":null},"stars":null,"forks":null,"downloads":1575,"packageName":null,"latestVersion":"1.9.19","tractionLabel":"1.6K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T07:58:53.992Z","emptyReason":null},"lastUpdatedAt":"2026-10-10T07:58:53.992Z","lastCrawledAt":"2026-10-10T07:58:53.992Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-11T07:58:53.992Z","lastVerifiedAt":null,"highlights":[{"version":"1.9.19","createdAt":"2026-08-26T13:19:33.885Z","changelog":"Release v1.9.19","fileCount":8,"zipByteSize":13601},{"version":"1.9.17","createdAt":"2026-07-30T05:39:43.732Z","changelog":"Release v1.9.17","fileCount":8,"zipByteSize":13647},{"version":"1.9.16","createdAt":"2026-07-14T19:56:28.437Z","changelog":"Release v1.9.16","fileCount":8,"zipByteSize":13557},{"version":"1.9.14","createdAt":"2026-06-30T18:04:41.848Z","changelog":"Release v1.9.14","fileCount":8,"zipByteSize":13547},{"version":"1.9.13","createdAt":"2026-06-27T16:22:37.642Z","changelog":"Release v1.9.13","fileCount":8,"zipByteSize":13688},{"version":"1.9.12","createdAt":"2026-06-19T03:17:53.084Z","changelog":"Release v1.9.12","fileCount":8,"zipByteSize":13626},{"version":"1.0.3","createdAt":"2026-06-18T15:16:51.939Z","changelog":"Release v1.9.12","fileCount":8,"zipByteSize":13748},{"version":"1.0.2","createdAt":"2026-05-09T02:19:32.036Z","changelog":"Release v1.9.5","fileCount":8,"zipByteSize":12571}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s17emme0e2m3cpf7k2jvp3a84984b8z9:nm-pensive-test-review","setupComplexity":"low","setupSteps":["Setup complexity is classified as HIGH. You must provision dedicated cloud infrastructure or an isolated VM. Do not run this directly on your local workstation.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-athola-nm-pensive-test-review/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-athola-nm-pensive-test-review/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-athola-nm-pensive-test-review/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-athola-nm-pensive-test-review/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-athola-nm-pensive-test-review/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-athola-nm-pensive-test-review/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-10T10:49:59.032Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-athola-nm-pensive-test-review/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-athola-nm-pensive-test-review/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-athola-nm-pensive-test-review/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-athola-nm-pensive-test-review/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-10T07:58:53.992Z","emptyReason":null},"readme":"Skill: test-review\n\nOwner: athola\n\nSummary: Evaluates test suites for coverage gaps, TDD/BDD compliance, and anti-patterns\n\nTags: latest:1.9.19\n\nVersion history:\n\nv1.9.19 | 2026-08-26T13:19:33.885Z | user\n\nRelease v1.9.19\n\nv1.9.17 | 2026-07-30T05:39:43.732Z | user\n\nRelease v1.9.17\n\nv1.9.16 | 2026-07-14T19:56:28.437Z | user\n\nRelease v1.9.16\n\nv1.9.14 | 2026-06-30T18:04:41.848Z | user\n\nRelease v1.9.14\n\nv1.9.13 | 2026-06-27T16:22:37.642Z | user\n\nRelease v1.9.13\n\nv1.9.12 | 2026-06-19T03:17:53.084Z | user\n\nRelease v1.9.12\n\nv1.0.3 | 2026-06-18T15:16:51.939Z | user\n\nRelease v1.9.12\n\nv1.0.2 | 2026-05-09T02:19:32.036Z | user\n\nRelease v1.9.5\n\nv1.0.1 | 2026-05-06T14:20:49.197Z | user\n\nRelease v1.9.4\n\nv1.0.0 | 2026-04-15T15:01:59.565Z | auto\n\nInitial release of the test-review skill—analyzes test suites for coverage, quality, and TDD/BDD compliance.\n\n- Provides a 5-step workflow: language detection, coverage inventory, scenario quality check, remediation planning, and evidence logging.\n- Includes quick start guide, troubleshooting, and condensed test quality checklist.\n- Outlines modular, progressive loading based on review depth.\n- Delivers output in a structured, standardized markdown format.\n- Integrates with night-market.pensive and imbue:proof-of-work for reproducibility and evidence capture.\n\nArchive index:\n\nArchive v1.9.19: 8 files, 13601 bytes\n\nFiles: modules/content-assertion-quality.md (2534b), modules/coverage-analysis.md (3330b), modules/framework-detection.md (2315b), modules/remediation-planning.md (4849b), modules/scenario-quality.md (5212b), skill-card.md (1928b), SKILL.md (8033b), _meta.json (142b)\n\nFile v1.9.19:SKILL.md\n\n---\nname: test-review\ndescription: Evaluates test suites for coverage gaps, TDD/BDD compliance, and anti-patterns\nversion: 1.9.8\ntriggers:\n  - testing\n  - tdd\n  - bdd\n  - coverage\n  - quality\n  - fixtures\n  - auditing test quality or before a major release\nmetadata: {\"openclaw\": {\"homepage\": \"https://github.com/athola/claude-night-market/tree/master/plugins/pensive\", \"emoji\": \"\\ud83e\\uddea\", \"requires\": {\"config\": [\"night-market.pensive:shared\", \"night-market.imbue:proof-of-work\"]}}}\nsource: claude-night-market\nsource_plugin: pensive\n---\n\n> **Night Market Skill** — ported from [claude-night-market/pensive](https://github.com/athola/claude-night-market/tree/master/plugins/pensive). For the full experience with agents, hooks, and commands, install the Claude Code plugin.\n\n\n## Table of Contents\n\n- [Quick Start](#quick-start)\n- [When to Use](#when-to-use)\n- [Required TodoWrite Items](#required-todowrite-items)\n- [Progressive Loading](#progressive-loading)\n- [Workflow](#workflow)\n- [Step 1: Detect Languages (`test-review:languages-detected`)](#step-1:-detect-languages-(test-review:languages-detected))\n- [Step 2: Inventory Coverage (`test-review:coverage-inventoried`)](#step-2:-inventory-coverage-(test-review:coverage-inventoried))\n- [Step 3: Assess Scenario Quality (`test-review:scenario-quality`)](#step-3:-assess-scenario-quality-(test-review:scenario-quality))\n- [Step 4: Plan Remediation (`test-review:gap-remediation`)](#step-4:-plan-remediation-(test-review:gap-remediation))\n- [Step 5: Log Evidence (`test-review:evidence-logged`)](#step-5:-log-evidence-(test-review:evidence-logged))\n- [Test Quality Checklist (Condensed)](#test-quality-checklist-(condensed))\n- [Output Format](#output-format)\n- [Summary](#summary)\n- [Framework Detection](#framework-detection)\n- [Coverage Analysis](#coverage-analysis)\n- [Quality Issues](#quality-issues)\n- [Remediation Plan](#remediation-plan)\n- [Recommendation](#recommendation)\n- [Integration Notes](#integration-notes)\n- [Exit Criteria](#exit-criteria)\n\n\n# Test Review Workflow\n\nEvaluate and improve test suites with TDD/BDD rigor.\n\n## Quick Start\n\n```bash\n/test-review\n```\n**Verification:** Run `pytest -v` to verify tests pass.\n\n## When To Use\n\n- Reviewing test suite quality\n- Analyzing coverage gaps\n- Before major releases\n- After test failures\n- Planning test improvements\n\n## When NOT To Use\n\n- Writing new tests - use parseltongue:python-testing\n- Updating existing tests - use sanctum:test-updates\n\n## Required TodoWrite Items\n\n1. `test-review:languages-detected`\n2. `test-review:coverage-inventoried`\n3. `test-review:scenario-quality`\n4. `test-review:invariant-preservation`\n5. `test-review:gap-remediation`\n6. `test-review:evidence-logged`\n\n## Progressive Loading\n\nLoad modules as needed based on review depth:\n\n- **Basic review**: Core workflow (this file)\n- **Framework detection**: Load `modules/framework-detection.md`\n- **Coverage analysis**: Load `modules/coverage-analysis.md`\n- **Quality assessment**: Load `modules/scenario-quality.md`\n- **Remediation planning**: Load `modules/remediation-planning.md`\n\n## Workflow\n\n### Step 1: Detect Languages (`test-review:languages-detected`)\n\nIdentify testing frameworks and version constraints.\n→ **See**: `modules/framework-detection.md`\n\nQuick check:\n```bash\nfind . -maxdepth 2 -name \"Cargo.toml\" -o -name \"pyproject.toml\" -o -name \"package.json\" -o -name \"go.mod\"\n```\n**Verification:** Run the command with `--help` flag to verify availability.\n\n### Step 2: Inventory Coverage (`test-review:coverage-inventoried`)\n\nRun coverage tools and identify gaps.\n→ **See**: `modules/coverage-analysis.md`\n\nQuick check:\n```bash\ngit diff --name-only | rg 'tests|spec|feature'\n```\n**Verification:** Run `pytest -v` to verify tests pass.\n\n### Step 3: Assess Scenario Quality (`test-review:scenario-quality`)\n\nEvaluate test quality using BDD patterns and assertion checks.\n→ **See**: `modules/scenario-quality.md`\n\nFocus on:\n- Given/When/Then clarity\n- Assertion specificity\n- Anti-patterns (dead waits, mocking internals, repeated boilerplate)\n\n### Step 4: Plan Remediation (`test-review:gap-remediation`)\n\nCreate concrete improvement plan with owners and dates.\n→ **See**: `modules/remediation-planning.md`\n\n### Step 5: Log Evidence (`test-review:evidence-logged`)\n\nRecord executed commands, outputs, and recommendations.\n→ **See**: `imbue:proof-of-work`\n\n## Test Quality Checklist (Condensed)\n\n- [ ] Clear test structure (Arrange-Act-Assert)\n- [ ] Critical paths covered (auth, validation, errors)\n- [ ] Specific assertions with context\n- [ ] No flaky tests (dead waits, order dependencies)\n- [ ] Reusable fixtures/factories\n- [ ] Invariant-encoding tests intact (see below)\n\n### Invariant-Encoding Tests\n\nTests do not just verify behavior — they encode design\ninvariants. A test that asserts \"module A never imports\nfrom module B\" encodes a layer boundary. A test that\nasserts \"this function is pure\" encodes a concurrency\nmodel. These tests are load-bearing in ways that\ncoverage metrics cannot capture.\n\n**During review, check:**\n\n1. **Were invariant-encoding tests removed or weakened?**\n   A test that enforced an architectural boundary,\n   data structure constraint, or API contract should\n   not be deleted without naming the invariant being\n   abandoned and escalating to human judgment.\n\n2. **Were test expectations changed to match a broken\n   implementation?** If an assertion value changed, ask:\n   did the *requirement* change, or did the agent change\n   the test to make its code pass? The latter is the\n   single most dangerous form of test tampering.\n\n3. **Are new invariants encoded as tests?** When a design\n   decision is made (choice of data structure, module\n   boundary, error strategy), there should be at least\n   one test whose failure would signal that the\n   invariant was violated.\n\n**Red flag patterns:**\n\n| Pattern | Risk |\n|---------|------|\n| `@pytest.mark.skip` added to a passing test | Invariant being silently dropped |\n| Assertion changed from specific to broad | Constraint being relaxed |\n| Test renamed to describe new behavior | Old invariant erased from history |\n| Test deleted \"because it tested old code\" | Invariant removed without replacement |\n\n**When invariant erosion is detected:**\n\nDo NOT approve. Flag as a BLOCKING quality issue and\npresent the three options to the human:\n\n1. **Preserve**: Revert the test change, fix the\n   implementation to satisfy the invariant\n2. **Layer**: Keep the invariant test, add the new\n   behavior alongside it (accepting inelegance)\n3. **Revise**: The invariant is genuinely wrong — remove\n   the old test AND write a new test encoding the\n   replacement invariant\n\nThis is a judgment call that models get wrong far too\noften. Default to option 1 (preserve) when no human is\navailable.\n\n## Output Format\n\n```markdown\n## Summary\n[Brief assessment]\n\n## Framework Detection\n- Languages: [list] | Frameworks: [list] | Versions: [constraints]\n\n## Coverage Analysis\n- Overall: X% | Critical: X% | Gaps: [list]\n\n## Quality Issues\n[Q1] [Issue] - Location - Fix\n\n## Remediation Plan\n1. [Action] - Owner - Date\n\n## Recommendation\nApprove / Approve with actions / Block\n```\n**Verification:** Run the command with `--help` flag to verify availability.\n\n## Integration Notes\n\n- Use `imbue:proof-of-work` for reproducible evidence capture\n- Reference `imbue:diff-analysis` for risk assessment\n- Format output using `imbue:structured-output` patterns\n\n## Exit Criteria\n\n- Frameworks detected and documented\n- Coverage analyzed and gaps identified\n- Scenario quality assessed\n- Remediation plan created with owners and dates\n- Evidence logged with citations\n## Troubleshooting\n\n### Common Issues\n\n**Tests not discovered**\nEnsure test files match pattern `test_*.py` or `*_test.py`. Run `pytest --collect-only` to verify.\n\n**Import errors**\nCheck that the module being tested is in `PYTHONPATH` or install with `pip install -e .`\n\n**Async tests failing**\nInstall pytest-asyncio and decorate test functions with `@pytest.mark.asyncio`\n\nFile v1.9.19:_meta.json\n\n{\n  \"ownerId\": \"kn7d107jg9jv602h9ytsegydq184a42s\",\n  \"slug\": \"nm-pensive-test-review\",\n  \"version\": \"1.9.19\",\n  \"publishedAt\": 1787750373885\n}\n\nFile v1.9.19:modules/content-assertion-quality.md\n\n# Content Assertion Quality\n\nScoring criteria for evaluating content assertion tests during test review. Extends the scenario quality assessment with a Content Depth dimension.\n\nReference: `leyline:testing-quality-standards/modules/content-assertion-levels.md`\n\n## Content Depth Scoring\n\nRate content assertion depth on a 1-5 scale:\n\n| Score | Level | Description |\n|---|---|---|\n| 1 | None | Tests only file existence or line count |\n| 2 | L1 | Keyword presence checks (`assert \"section\" in content`) |\n| 3 | L2 | Parses embedded examples, validates schema structure |\n| 4 | L3 | Cross-references, anti-patterns, decision framework contracts |\n| 5 | L3+ | Cross-plugin validation (version refs checked against other plugins' docs) |\n\n## When to Flag Missing Content Assertions\n\nDuring test review, flag as a content test gap when:\n\n- A skill has tests but all are L1 (keyword-only) and the skill contains JSON or YAML code blocks\n- A skill has version-gated features but no cross-reference validation\n- A skill defines behavioral guidance (decision trees, strategies) but no anti-pattern or completeness tests\n- A module documents forbidden behaviors but no test asserts their absence\n\n## Content Assertion Anti-Patterns\n\nAvoid these when reviewing content tests:\n\n| Anti-Pattern | Problem | Better Approach |\n|---|---|---|\n| Testing prose style | Brittle to rewording, overlaps with scribe:slop-detector | Test behavioral semantics |\n| Asserting exact wording | Breaks on any edit | Assert concepts (`\"version\" in content.lower()`) |\n| Checking line counts | Not behavioral | Check required sections exist |\n| Testing formatting | Not what Claude interprets | Test parseable structure |\n| Duplicating slop detection | Already handled by scribe | Focus on correctness, not style |\n\n## Review Checklist Addition\n\nAdd this item to the existing Test Quality Checklist when reviewing a plugin that has execution markdown:\n\n```markdown\n- [ ] Content assertion depth matches content complexity\n      (L1 for simple skills, L2+ for code examples, L3 for behavioral guidance)\n```\n\n## Remediation Guidance\n\nWhen content tests are missing or insufficient:\n\n1. **No content tests at all**: Generate L1 scaffolding using `sanctum:test-updates/modules/generation/content-test-templates.md`\n2. **L1 only, has code blocks**: Upgrade to L2 (add JSON/YAML parsing tests)\n3. **L2 only, has version gates**: Upgrade to L3 (add cross-reference validation)\n4. **L2 only, has behavioral guidance**: Upgrade to L3 (add anti-pattern and completeness tests)\n\nFile v1.9.19:modules/coverage-analysis.md\n\n---\nparent_skill: pensive:test-review\nname: coverage-analysis\ndescription: Coverage measurement and gap identification\ncategory: testing\ntags: [coverage, testing, gap-analysis]\nload_priority: 2\nestimated_tokens: 350\n---\n\n# Coverage Analysis\n\nMeasure test coverage and identify gaps.\n\n## Coverage Tools by Language\n\n### Rust\n```bash\n# Using tarpaulin\ncargo install cargo-tarpaulin\ncargo tarpaulin --out Html --output-dir coverage/\n\n# Using llvm-cov\ncargo install cargo-llvm-cov\ncargo llvm-cov --html\n```\n\n### Python\n```bash\n# Using pytest-cov\npytest --cov=src --cov-report=html --cov-report=term-missing\n\n# Using coverage.py\ncoverage run -m pytest\ncoverage html\ncoverage report --show-missing\n```\n\n### JavaScript/TypeScript\n```bash\n# Jest\nnpm test -- --coverage --coverageReporters=html text\n\n# Vitest\nvitest --coverage\n\n# Cypress (code coverage plugin)\ncypress run --env coverage=true\n```\n\n### Go\n```bash\n# Built-in coverage\ngo test -cover ./...\ngo test -coverprofile=coverage.out ./...\ngo tool cover -html=coverage.out\n\n# Detailed coverage\ngo test -covermode=count -coverprofile=coverage.out ./...\n```\n\n## Coverage Thresholds\n\n| Level | Coverage | Use Case |\n|-------|----------|----------|\n| Minimum | 60% | Legacy code, initial cleanup |\n| Standard | 80% | Normal development |\n| High | 90% | Critical systems, libraries |\n| detailed | 95%+ | Safety-critical, financial |\n\n## Gap Identification\n\n### Find impacted test files\n```bash\n# Tests affected by changes\ngit diff --name-only main...HEAD | rg 'tests|spec|feature'\n\n# Find related tests\ngit diff --name-only main...HEAD | while read file; do\n  basename \"$file\" .py | xargs -I {} find . \\\n    -not -path \"*/.venv/*\" -not -path \"*/__pycache__/*\" \\\n    -not -path \"*/node_modules/*\" -not -path \"*/.git/*\" \\\n    -name \"*test*{}*\"\ndone\n```\n\n### Identify uncovered code\n1. Run coverage tool with `--show-missing` flag\n2. Cross-reference with critical paths:\n   - Authentication/authorization\n   - Data validation\n   - Error handling\n   - API endpoints\n   - Database operations\n\n3. Map to requirements:\n   - Feature specifications\n   - User stories\n   - Bug reports\n   - Security requirements\n\n### Coverage Patterns\n\n**Critical paths** (should be 100%):\n- Security boundaries (auth, validation)\n- Data integrity operations\n- Error recovery logic\n- Public API surface\n\n**Lower priority** (can be <80%):\n- Internal helpers\n- Logging/debugging code\n- Trivial getters/setters\n- Deprecated code paths\n\n## Output Format\n\n```markdown\n## Coverage Analysis\n- **Overall**: 78%\n- **Critical paths**: 92%\n- **Changed files**: 85%\n\n### Gaps Identified\n1. **src/auth.py:45-60** - Token validation edge cases\n2. **src/api/routes.py:120-135** - Error handling for 400/500 codes\n3. **src/db/migrations.py** - Rollback scenarios untested\n\n### Test-to-Feature Mapping\n- Feature: User registration → `tests/test_registration.py` (95%)\n- Feature: Password reset → `tests/test_auth.py` (60%) [WARN]\n- Feature: Email validation → Missing tests [FAIL]\n```\n\n## Best Practices\n\n1. **Branch coverage** over line coverage when available\n2. **Mutation testing** for critical code (e.g., `cargo mutants`, `mutmut`)\n3. **Coverage trends**: Track over time, not just absolute values\n4. **Exclude generated code**: Focus on hand-written logic\n5. **Integration coverage**: Don't just unit test in isolation\n\nFile v1.9.19:modules/framework-detection.md\n\n---\nparent_skill: pensive:test-review\nname: framework-detection\ndescription: Language and test framework detection patterns\ncategory: testing\ntags: [testing, framework-detection, language-detection]\nload_priority: 1\nestimated_tokens: 250\n---\n\n# Framework Detection\n\nIdentify testing frameworks and tooling constraints.\n\n## Language Detection Patterns\n\n### Rust\n- **Framework**: cargo test (built-in)\n- **Commands**: `cargo test`, `cargo nextest run`\n- **Config files**: `Cargo.toml`, `Cargo.lock`\n- **Test patterns**: `#[test]`, `#[cfg(test)]`\n- **MSRV**: Check `rust-version` in Cargo.toml\n\n### Python\n- **Frameworks**: pytest, unittest, behave\n- **Commands**: `pytest`, `python -m pytest`, `behave`\n- **Config files**: `pytest.ini`, `pyproject.toml`, `tox.ini`\n- **Test patterns**: `test_*.py`, `*_test.py`, `tests/`\n- **Version**: Check `requires-python` in pyproject.toml\n\n### JavaScript/TypeScript\n- **Frameworks**: Jest, Mocha, Cypress, Vitest\n- **Commands**: `npm test`, `yarn test`, `cypress run`\n- **Config files**: `jest.config.js`, `vitest.config.ts`, `cypress.config.js`\n- **Test patterns**: `*.test.js`, `*.spec.ts`, `__tests__/`\n- **Version**: Check `engines.node` in package.json\n\n### Go\n- **Framework**: go test (built-in)\n- **Commands**: `go test ./...`, `go test -v`\n- **Config files**: `go.mod`, `go.sum`\n- **Test patterns**: `*_test.go`\n- **Version**: Check `go` directive in go.mod\n\n## Detection Workflow\n\n1. **Scan for config files**:\n```bash\nfind . -maxdepth 2 -name \"Cargo.toml\" -o -name \"pyproject.toml\" -o -name \"package.json\" -o -name \"go.mod\"\n```\n\n2. **Check test directories**:\n```bash\nfind . -type d -name \"tests\" -o -name \"__tests__\" -o -name \"test\"\n```\n\n3. **Identify test files**:\n```bash\nfind . -not -path \"*/.venv/*\" -not -path \"*/__pycache__/*\" \\\n  -not -path \"*/node_modules/*\" -not -path \"*/.git/*\" \\\n  \\( -name \"*test*\" -o -name \"*spec*\" \\) \\\n  | grep -E '\\.(rs|py|js|ts|go)$'\n```\n\n4. **Version constraints**:\n- Extract MSRV, Python version, Node version\n- Note if constraints affect tooling (e.g., async/await)\n- Document CI/CD version requirements\n\n## Output Format\n\n```markdown\n## Framework Detection\n- **Languages**: Rust, Python\n- **Frameworks**: cargo test, pytest\n- **Versions**:\n  - Rust MSRV: 1.70\n  - Python: >=3.8\n- **Config files**: Cargo.toml, pyproject.toml\n```\n\nFile v1.9.19:modules/remediation-planning.md\n\n---\nparent_skill: pensive:test-review\nname: remediation-planning\ndescription: Test improvement strategies and phased remediation\ncategory: testing\ntags: [remediation, test-improvement, refactoring]\nload_priority: 4\nestimated_tokens: 300\n---\n\n# Remediation Planning\n\nConcrete strategies for test improvement.\n\n## Test Improvement Patterns\n\n### 1. Add Missing Coverage\nTie tests to specific behaviors using Given/When/Then:\n\n```markdown\n### Gap: Authentication edge cases\n**Behavior**: Given missing auth token, When hitting /v1/resource, Then HTTP 401 returned\n**Test**: `tests/test_auth.py::test_missing_token_returns_401`\n**Priority**: High (security boundary)\n```\n\n### 2. Refactor Test Helpers\n\n**Before** (repeated setup):\n```python\ndef test_user_creation():\n    db = setup_database()\n    config = load_test_config()\n    user_data = {\"email\": \"alice@example.com\", \"role\": \"user\"}\n    ...\n\ndef test_user_deletion():\n    db = setup_database()\n    config = load_test_config()\n    user_data = {\"email\": \"bob@example.com\", \"role\": \"admin\"}\n    ...\n```\n\n**After** (fixtures):\n```python\n@pytest.fixture\ndef test_db():\n    db = setup_database()\n    yield db\n    db.teardown()\n\n@pytest.fixture\ndef test_config():\n    return load_test_config()\n\ndef test_user_creation(test_db, test_config):\n    user_data = user_factory(email=\"alice@example.com\")\n    ...\n```\n\n### 3. Data Builders and Factories\n\n**Factory pattern**:\n```python\n# conftest.py\ndef user_factory(**overrides):\n    defaults = {\n        \"email\": \"user@example.com\",\n        \"role\": \"user\",\n        \"verified\": True,\n        \"created_at\": datetime.now()\n    }\n    return User(**{**defaults, **overrides})\n\n# test file\ndef test_admin_access():\n    admin = user_factory(role=\"admin\")\n    assert admin.can_access_dashboard()\n```\n\n**Builder pattern** (Rust):\n```rust\nstruct UserBuilder {\n    email: String,\n    role: Role,\n    verified: bool,\n}\n\nimpl UserBuilder {\n    fn new() -> Self {\n        Self {\n            email: \"user@example.com\".to_string(),\n            role: Role::User,\n            verified: true,\n        }\n    }\n\n    fn with_role(mut self, role: Role) -> Self {\n        self.role = role;\n        self\n    }\n\n    fn build(self) -> User {\n        User { /* ... */ }\n    }\n}\n\n#[test]\nfn test_admin_permissions() {\n    let admin = UserBuilder::new().with_role(Role::Admin).build();\n    assert!(admin.can_delete_users());\n}\n```\n\n### 4. Improve Assertions\n\n**Replace magic values**:\n```python\n# Before\nassert response.status == 200\n\n# After\nfrom http import HTTPStatus\nassert response.status == HTTPStatus.OK\n```\n\n**Add context**:\n```python\n# Before\nassert result\n\n# After\nassert result.success, f\"Expected success, got error: {result.error}\"\n```\n\n### 5. Remove Brittle Patterns\n\n**Dead waits** → **Explicit conditions**:\n```python\n# Before\ntime.sleep(3)\nassert element.visible\n\n# After\nwait_for(element.to_be_visible, timeout=5)\n```\n\n**Mocking internals** → **Mock boundaries**:\n```python\n# Before: mocking private implementation\n@patch('service._internal_helper')\ndef test_service(mock):\n    ...\n\n# After: mock external dependency\n@patch('requests.post')\ndef test_service(mock_requests):\n    ...\n```\n\n## Phased Remediation\n\nFor major test suite rewrites:\n\n### Phase 1: Stabilize (Week 1-2)\n1. Fix flaky tests (eliminate dead waits, order dependencies)\n2. Remove duplicate tests\n3. Add missing critical path tests\n4. **Metric**: Flaky test rate < 1%, critical paths 100%\n\n### Phase 2: Acceptance Specs (Week 3-4)\n1. Add BDD scenarios for user-facing features\n2. Create feature-to-test mapping\n3. Document test strategy per component\n4. **Metric**: All features have acceptance tests\n\n### Phase 3: Enforce Quality (Week 5+)\n1. Set coverage budgets (80% standard, 100% critical)\n2. Add pre-commit hooks for coverage checks\n3. Integrate mutation testing for critical code\n4. **Metric**: Coverage trends upward, no regressions\n\n## Recommendation Template\n\n```markdown\n## Remediation Plan\n\n### Immediate Actions (This Sprint)\n1. **Fix flaky test**: `test_user_login_retries` - Replace sleep with explicit wait\n   - Owner: @alice\n   - Due: 2025-12-10\n\n2. **Add missing coverage**: Password reset flow (currently 0%)\n   - Tests needed: valid token, expired token, invalid token\n   - Owner: @bob\n   - Due: 2025-12-12\n\n### Short-term (Next Sprint)\n3. **Refactor fixtures**: Extract common setup in `tests/test_api.py`\n   - Pattern: Use pytest fixtures for DB, config\n   - Owner: @charlie\n   - Due: 2025-12-20\n\n### Long-term (Next Month)\n4. **BDD acceptance tests**: User registration feature\n   - Tool: Behave/Gherkin\n   - Owner: @diana\n   - Due: 2025-01-15\n```\n\n## Exit Criteria\n\n- [ ] All critical gaps have assigned owners and due dates\n- [ ] Recommendations tied to specific behaviors\n- [ ] Phased approach for large refactorings\n- [ ] Success metrics defined (coverage %, flaky rate, etc.)\n\nFile v1.9.19:modules/scenario-quality.md\n\n---\nparent_skill: pensive:test-review\nname: scenario-quality\ndescription: Test scenario quality assessment with BDD patterns\ncategory: testing\ntags: [bdd, scenario-quality, assertions, anti-patterns]\nload_priority: 3\nestimated_tokens: 350\n---\n\n# Scenario Quality Assessment\n\nEvaluate test quality using BDD principles and assertion patterns.\n\n## Given/When/Then Clarity\n\n### Good Examples\n\n**Rust:**\n```rust\n#[test]\nfn test_authenticated_user_can_access_profile() {\n    // Given: authenticated user\n    let user = create_authenticated_user(\"alice@example.com\");\n    let token = generate_token(&user);\n\n    // When: accessing profile endpoint\n    let response = get(\"/profile\", &token);\n\n    // Then: profile data returned\n    assert_eq!(response.status, 200);\n    assert_eq!(response.body[\"email\"], \"alice@example.com\");\n}\n```\n\n**Python:**\n```python\ndef test_invalid_credentials_rejected():\n    # Given: user with wrong password\n    user = User(email=\"bob@example.com\")\n    wrong_password = \"incorrect\"\n\n    # When: attempting authentication\n    result = authenticate(user.email, wrong_password)\n\n    # Then: authentication fails with 401\n    assert result.status_code == 401\n    assert \"invalid credentials\" in result.error_message\n```\n\n**Gherkin (BDD):**\n```gherkin\nScenario: Registered user logs in successfully\n  Given a registered user with email \"alice@example.com\"\n  When they submit valid credentials\n  Then they receive an authentication token\n  And the token expires in 24 hours\n```\n\n## Assertion Quality\n\n### Bad Assertions (vague, brittle)\n```python\n# Too vague\nassert result\n\n# Multiple unrelated assertions\nassert len(users) > 0 and users[0].active and config.debug\n\n# Magic numbers without context\nassert response.status == 200\n```\n\n### Good Assertions (specific, meaningful)\n```python\n# Specific outcome\nassert result.status_code == 200, \"Expected successful login\"\n\n# Named constants\nassert response.status == HTTP_OK\nassert user.role == UserRole.ADMIN\n\n# Structured assertions\nassert response.json() == {\n    \"user\": {\"email\": expected_email, \"verified\": True},\n    \"token\": {\"expires_at\": ANY_DATETIME}\n}\n```\n\n## Anti-Patterns to Flag\n\n### 1. Dead Waits\n```python\n# BAD: arbitrary sleep\ntime.sleep(5)\nassert element.is_visible()\n\n# GOOD: explicit wait with condition\nwait_until(lambda: element.is_visible(), timeout=5)\n```\n\n### 2. Mocking Internals\n```python\n# BAD: mocking implementation details\n@patch('module.internal._private_helper')\ndef test_feature(mock_helper):\n    ...\n\n# GOOD: mock external dependencies only\n@patch('requests.get')\ndef test_api_call(mock_get):\n    ...\n```\n\n### 3. Repeated Boilerplate\n```python\n# BAD: copy-pasted setup\ndef test_user_creation():\n    db = Database(\"test.db\")\n    db.connect()\n    user = User(\"alice\")\n    ...\n\ndef test_user_deletion():\n    db = Database(\"test.db\")\n    db.connect()\n    user = User(\"bob\")\n    ...\n\n# GOOD: fixture/helper\n@pytest.fixture\ndef db_session():\n    db = Database(\"test.db\")\n    db.connect()\n    yield db\n    db.close()\n```\n\n### 4. Order Dependencies\n```python\n# BAD: tests depend on execution order\ndef test_01_create_user():\n    global user_id\n    user_id = create_user()\n\ndef test_02_delete_user():\n    delete_user(user_id)  # Depends on test_01!\n\n# GOOD: isolated tests\ndef test_delete_user():\n    user_id = create_user()  # Self-contained\n    delete_user(user_id)\n    assert not user_exists(user_id)\n```\n\n### 5. Multiple Assertions Without Context\n```python\n# BAD: unclear which assertion failed\nassert user.active\nassert user.verified\nassert user.role == \"admin\"\n\n# GOOD: grouped with context or separate tests\nassert user.active, \"User should be active\"\nassert user.verified, \"User should be verified\"\nassert user.role == \"admin\", \"User should have admin role\"\n```\n\n## BDD Suite Quality\n\n### Reusable Step Definitions\n```python\n# Good: parameterized, reusable\n@given('a user with email \"{email}\"')\ndef create_user(context, email):\n    context.user = User(email=email)\n\n@when('they submit credentials with password \"{password}\"')\ndef submit_credentials(context, password):\n    context.response = authenticate(context.user.email, password)\n```\n\n### Background Context Sharing\n```gherkin\nFeature: User authentication\n\n  Background:\n    Given a clean database\n    And the authentication service is running\n\n  Scenario: Valid login\n    Given a registered user\n    ...\n```\n\n### Scenario Outlines for Edge Cases\n```gherkin\nScenario Outline: Password validation\n  Given a user registering with password \"<password>\"\n  When they submit the registration form\n  Then they receive response \"<outcome>\"\n\n  Examples:\n    | password    | outcome           |\n    | abc         | too_short         |\n    | password123 | no_special_chars  |\n    | P@ssw0rd!   | success           |\n```\n\n## Quality Scoring\n\nScore each test file 1-5 on:\n- **Clarity**: Given/When/Then structure evident\n- **Assertions**: Specific, meaningful checks\n- **Isolation**: No shared state or order dependencies\n- **Maintainability**: DRY, uses fixtures/helpers\n- **Coverage**: Tests behavior, not implementation\n\n**Overall quality**:\n- 4-5: Excellent, minimal changes needed\n- 3: Good, some improvements recommended\n- 1-2: Poor, significant refactoring required\n\nFile v1.9.19:skill-card.md\n\n## Description:\n\nEvaluates test suites for coverage gaps, TDD/BDD compliance, and anti-patterns.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[athola](https://clawhub.ai/user/athola)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and engineers use this skill to review existing test suites, identify coverage and scenario-quality gaps, flag fragile or misleading tests, and plan concrete remediation before releases or after test failures.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The skill may lead an agent to inspect repository contents and run test, coverage, or tool-install commands.\n\nMitigation: Use it only in trusted repositories or a sandbox, and require explicit approval before package installs or test execution.\n\nRisk: Coverage tools referenced by the artifact may be installed without pinned versions.\n\nMitigation: Prefer project-approved, pinned tooling and review install commands before execution.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/athola/skills/nm-pensive-test-review)\n- [OpenClaw homepage](https://github.com/athola/claude-night-market/tree/master/plugins/pensive)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, shell commands, guidance]\n\n**Output Format:** [Markdown with inline shell commands and structured review sections]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [May include framework detection, coverage findings, quality issues, remediation plans, recommendations, and evidence notes.]\n\n## Skill Version(s):\n\n1.9.19 (source: server release metadata; artifact frontmatter reports 1.9.8)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v1.9.17: 8 files, 13647 bytes\n\nFiles: modules/content-assertion-quality.md (2534b), modules/coverage-analysis.md (3330b), modules/framework-detection.md (2315b), modules/remediation-planning.md (4849b), modules/scenario-quality.md (5212b), skill-card.md (2164b), SKILL.md (8033b), _meta.json (142b)\n\nFile v1.9.17:SKILL.md\n\n---\nname: test-review\ndescription: Evaluates test suites for coverage gaps, TDD/BDD compliance, and anti-patterns\nversion: 1.9.8\ntriggers:\n  - testing\n  - tdd\n  - bdd\n  - coverage\n  - quality\n  - fixtures\n  - auditing test quality or before a major release\nmetadata: {\"openclaw\": {\"homepage\": \"https://github.com/athola/claude-night-market/tree/master/plugins/pensive\", \"emoji\": \"\\ud83e\\uddea\", \"requires\": {\"config\": [\"night-market.pensive:shared\", \"night-market.imbue:proof-of-work\"]}}}\nsource: claude-night-market\nsource_plugin: pensive\n---\n\n> **Night Market Skill** — ported from [claude-night-market/pensive](https://github.com/athola/claude-night-market/tree/master/plugins/pensive). For the full experience with agents, hooks, and commands, install the Claude Code plugin.\n\n\n## Table of Contents\n\n- [Quick Start](#quick-start)\n- [When to Use](#when-to-use)\n- [Required TodoWrite Items](#required-todowrite-items)\n- [Progressive Loading](#progressive-loading)\n- [Workflow](#workflow)\n- [Step 1: Detect Languages (`test-review:languages-detected`)](#step-1:-detect-languages-(test-review:languages-detected))\n- [Step 2: Inventory Coverage (`test-review:coverage-inventoried`)](#step-2:-inventory-coverage-(test-review:coverage-inventoried))\n- [Step 3: Assess Scenario Quality (`test-review:scenario-quality`)](#step-3:-assess-scenario-quality-(test-review:scenario-quality))\n- [Step 4: Plan Remediation (`test-review:gap-remediation`)](#step-4:-plan-remediation-(test-review:gap-remediation))\n- [Step 5: Log Evidence (`test-review:evidence-logged`)](#step-5:-log-evidence-(test-review:evidence-logged))\n- [Test Quality Checklist (Condensed)](#test-quality-checklist-(condensed))\n- [Output Format](#output-format)\n- [Summary](#summary)\n- [Framework Detection](#framework-detection)\n- [Coverage Analysis](#coverage-analysis)\n- [Quality Issues](#quality-issues)\n- [Remediation Plan](#remediation-plan)\n- [Recommendation](#recommendation)\n- [Integration Notes](#integration-notes)\n- [Exit Criteria](#exit-criteria)\n\n\n# Test Review Workflow\n\nEvaluate and improve test suites with TDD/BDD rigor.\n\n## Quick Start\n\n```bash\n/test-review\n```\n**Verification:** Run `pytest -v` to verify tests pass.\n\n## When To Use\n\n- Reviewing test suite quality\n- Analyzing coverage gaps\n- Before major releases\n- After test failures\n- Planning test improvements\n\n## When NOT To Use\n\n- Writing new tests - use parseltongue:python-testing\n- Updating existing tests - use sanctum:test-updates\n\n## Required TodoWrite Items\n\n1. `test-review:languages-detected`\n2. `test-review:coverage-inventoried`\n3. `test-review:scenario-quality`\n4. `test-review:invariant-preservation`\n5. `test-review:gap-remediation`\n6. `test-review:evidence-logged`\n\n## Progressive Loading\n\nLoad modules as needed based on review depth:\n\n- **Basic review**: Core workflow (this file)\n- **Framework detection**: Load `modules/framework-detection.md`\n- **Coverage analysis**: Load `modules/coverage-analysis.md`\n- **Quality assessment**: Load `modules/scenario-quality.md`\n- **Remediation planning**: Load `modules/remediation-planning.md`\n\n## Workflow\n\n### Step 1: Detect Languages (`test-review:languages-detected`)\n\nIdentify testing frameworks and version constraints.\n→ **See**: `modules/framework-detection.md`\n\nQuick check:\n```bash\nfind . -maxdepth 2 -name \"Cargo.toml\" -o -name \"pyproject.toml\" -o -name \"package.json\" -o -name \"go.mod\"\n```\n**Verification:** Run the command with `--help` flag to verify availability.\n\n### Step 2: Inventory Coverage (`test-review:coverage-inventoried`)\n\nRun coverage tools and identify gaps.\n→ **See**: `modules/coverage-analysis.md`\n\nQuick check:\n```bash\ngit diff --name-only | rg 'tests|spec|feature'\n```\n**Verification:** Run `pytest -v` to verify tests pass.\n\n### Step 3: Assess Scenario Quality (`test-review:scenario-quality`)\n\nEvaluate test quality using BDD patterns and assertion checks.\n→ **See**: `modules/scenario-quality.md`\n\nFocus on:\n- Given/When/Then clarity\n- Assertion specificity\n- Anti-patterns (dead waits, mocking internals, repeated boilerplate)\n\n### Step 4: Plan Remediation (`test-review:gap-remediation`)\n\nCreate concrete improvement plan with owners and dates.\n→ **See**: `modules/remediation-planning.md`\n\n### Step 5: Log Evidence (`test-review:evidence-logged`)\n\nRecord executed commands, outputs, and recommendations.\n→ **See**: `imbue:proof-of-work`\n\n## Test Quality Checklist (Condensed)\n\n- [ ] Clear test structure (Arrange-Act-Assert)\n- [ ] Critical paths covered (auth, validation, errors)\n- [ ] Specific assertions with context\n- [ ] No flaky tests (dead waits, order dependencies)\n- [ ] Reusable fixtures/factories\n- [ ] Invariant-encoding tests intact (see below)\n\n### Invariant-Encoding Tests\n\nTests do not just verify behavior — they encode design\ninvariants. A test that asserts \"module A never imports\nfrom module B\" encodes a layer boundary. A test that\nasserts \"this function is pure\" encodes a concurrency\nmodel. These tests are load-bearing in ways that\ncoverage metrics cannot capture.\n\n**During review, check:**\n\n1. **Were invariant-encoding tests removed or weakened?**\n   A test that enforced an architectural boundary,\n   data structure constraint, or API contract should\n   not be deleted without naming the invariant being\n   abandoned and escalating to human judgment.\n\n2. **Were test expectations changed to match a broken\n   implementation?** If an assertion value changed, ask:\n   did the *requirement* change, or did the agent change\n   the test to make its code pass? The latter is the\n   single most dangerous form of test tampering.\n\n3. **Are new invariants encoded as tests?** When a design\n   decision is made (choice of data structure, module\n   boundary, error strategy), there should be at least\n   one test whose failure would signal that the\n   invariant was violated.\n\n**Red flag patterns:**\n\n| Pattern | Risk |\n|---------|------|\n| `@pytest.mark.skip` added to a passing test | Invariant being silently dropped |\n| Assertion changed from specific to broad | Constraint being relaxed |\n| Test renamed to describe new behavior | Old invariant erased from history |\n| Test deleted \"because it tested old code\" | Invariant removed without replacement |\n\n**When invariant erosion is detected:**\n\nDo NOT approve. Flag as a BLOCKING quality issue and\npresent the three options to the human:\n\n1. **Preserve**: Revert the test change, fix the\n   implementation to satisfy the invariant\n2. **Layer**: Keep the invariant test, add the new\n   behavior alongside it (accepting inelegance)\n3. **Revise**: The invariant is genuinely wrong — remove\n   the old test AND write a new test encoding the\n   replacement invariant\n\nThis is a judgment call that models get wrong far too\noften. Default to option 1 (preserve) when no human is\navailable.\n\n## Output Format\n\n```markdown\n## Summary\n[Brief assessment]\n\n## Framework Detection\n- Languages: [list] | Frameworks: [list] | Versions: [constraints]\n\n## Coverage Analysis\n- Overall: X% | Critical: X% | Gaps: [list]\n\n## Quality Issues\n[Q1] [Issue] - Location - Fix\n\n## Remediation Plan\n1. [Action] - Owner - Date\n\n## Recommendation\nApprove / Approve with actions / Block\n```\n**Verification:** Run the command with `--help` flag to verify availability.\n\n## Integration Notes\n\n- Use `imbue:proof-of-work` for reproducible evidence capture\n- Reference `imbue:diff-analysis` for risk assessment\n- Format output using `imbue:structured-output` patterns\n\n## Exit Criteria\n\n- Frameworks detected and documented\n- Coverage analyzed and gaps identified\n- Scenario quality assessed\n- Remediation plan created with owners and dates\n- Evidence logged with citations\n## Troubleshooting\n\n### Common Issues\n\n**Tests not discovered**\nEnsure test files match pattern `test_*.py` or `*_test.py`. Run `pytest --collect-only` to verify.\n\n**Import errors**\nCheck that the module being tested is in `PYTHONPATH` or install with `pip install -e .`\n\n**Async tests failing**\nInstall pytest-asyncio and decorate test functions with `@pytest.mark.asyncio`\n\nFile v1.9.17:_meta.json\n\n{\n  \"ownerId\": \"kn7d107jg9jv602h9ytsegydq184a42s\",\n  \"slug\": \"nm-pensive-test-review\",\n  \"version\": \"1.9.17\",\n  \"publishedAt\": 1785389983732\n}\n\nFile v1.9.17:modules/content-assertion-quality.md\n\n# Content Assertion Quality\n\nScoring criteria for evaluating content assertion tests during test review. Extends the scenario quality assessment with a Content Depth dimension.\n\nReference: `leyline:testing-quality-standards/modules/content-assertion-levels.md`\n\n## Content Depth Scoring\n\nRate content assertion depth on a 1-5 scale:\n\n| Score | Level | Description |\n|---|---|---|\n| 1 | None | Tests only file existence or line count |\n| 2 | L1 | Keyword presence checks (`assert \"section\" in content`) |\n| 3 | L2 | Parses embedded examples, validates schema structure |\n| 4 | L3 | Cross-references, anti-patterns, decision framework contracts |\n| 5 | L3+ | Cross-plugin validation (version refs checked against other plugins' docs) |\n\n## When to Flag Missing Content Assertions\n\nDuring test review, flag as a content test gap when:\n\n- A skill has tests but all are L1 (keyword-only) and the skill contains JSON or YAML code blocks\n- A skill has version-gated features but no cross-reference validation\n- A skill defines behavioral guidance (decision trees, strategies) but no anti-pattern or completeness tests\n- A module documents forbidden behaviors but no test asserts their absence\n\n## Content Assertion Anti-Patterns\n\nAvoid these when reviewing content tests:\n\n| Anti-Pattern | Problem | Better Approach |\n|---|---|---|\n| Testing prose style | Brittle to rewording, overlaps with scribe:slop-detector | Test behavioral semantics |\n| Asserting exact wording | Breaks on any edit | Assert concepts (`\"version\" in content.lower()`) |\n| Checking line counts | Not behavioral | Check required sections exist |\n| Testing formatting | Not what Claude interprets | Test parseable structure |\n| Duplicating slop detection | Already handled by scribe | Focus on correctness, not style |\n\n## Review Checklist Addition\n\nAdd this item to the existing Test Quality Checklist when reviewing a plugin that has execution markdown:\n\n```markdown\n- [ ] Content assertion depth matches content complexity\n      (L1 for simple skills, L2+ for code examples, L3 for behavioral guidance)\n```\n\n## Remediation Guidance\n\nWhen content tests are missing or insufficient:\n\n1. **No content tests at all**: Generate L1 scaffolding using `sanctum:test-updates/modules/generation/content-test-templates.md`\n2. **L1 only, has code blocks**: Upgrade to L2 (add JSON/YAML parsing tests)\n3. **L2 only, has version gates**: Upgrade to L3 (add cross-reference validation)\n4. **L2 only, has behavioral guidance**: Upgrade to L3 (add anti-pattern and completeness tests)\n\nFile v1.9.17:modules/coverage-analysis.md\n\n---\nparent_skill: pensive:test-review\nname: coverage-analysis\ndescription: Coverage measurement and gap identification\ncategory: testing\ntags: [coverage, testing, gap-analysis]\nload_priority: 2\nestimated_tokens: 350\n---\n\n# Coverage Analysis\n\nMeasure test coverage and identify gaps.\n\n## Coverage Tools by Language\n\n### Rust\n```bash\n# Using tarpaulin\ncargo install cargo-tarpaulin\ncargo tarpaulin --out Html --output-dir coverage/\n\n# Using llvm-cov\ncargo install cargo-llvm-cov\ncargo llvm-cov --html\n```\n\n### Python\n```bash\n# Using pytest-cov\npytest --cov=src --cov-report=html --cov-report=term-missing\n\n# Using coverage.py\ncoverage run -m pytest\ncoverage html\ncoverage report --show-missing\n```\n\n### JavaScript/TypeScript\n```bash\n# Jest\nnpm test -- --coverage --coverageReporters=html text\n\n# Vitest\nvitest --coverage\n\n# Cypress (code coverage plugin)\ncypress run --env coverage=true\n```\n\n### Go\n```bash\n# Built-in coverage\ngo test -cover ./...\ngo test -coverprofile=coverage.out ./...\ngo tool cover -html=coverage.out\n\n# Detailed coverage\ngo test -covermode=count -coverprofile=coverage.out ./...\n```\n\n## Coverage Thresholds\n\n| Level | Coverage | Use Case |\n|-------|----------|----------|\n| Minimum | 60% | Legacy code, initial cleanup |\n| Standard | 80% | Normal development |\n| High | 90% | Critical systems, libraries |\n| detailed | 95%+ | Safety-critical, financial |\n\n## Gap Identification\n\n### Find impacted test files\n```bash\n# Tests affected by changes\ngit diff --name-only main...HEAD | rg 'tests|spec|feature'\n\n# Find related tests\ngit diff --name-only main...HEAD | while read file; do\n  basename \"$file\" .py | xargs -I {} find . \\\n    -not -path \"*/.venv/*\" -not -path \"*/__pycache__/*\" \\\n    -not -path \"*/node_modules/*\" -not -path \"*/.git/*\" \\\n    -name \"*test*{}*\"\ndone\n```\n\n### Identify uncovered code\n1. Run coverage tool with `--show-missing` flag\n2. Cross-reference with critical paths:\n   - Authentication/authorization\n   - Data validation\n   - Error handling\n   - API endpoints\n   - Database operations\n\n3. Map to requirements:\n   - Feature specifications\n   - User stories\n   - Bug reports\n   - Security requirements\n\n### Coverage Patterns\n\n**Critical paths** (should be 100%):\n- Security boundaries (auth, validation)\n- Data integrity operations\n- Error recovery logic\n- Public API surface\n\n**Lower priority** (can be <80%):\n- Internal helpers\n- Logging/debugging code\n- Trivial getters/setters\n- Deprecated code paths\n\n## Output Format\n\n```markdown\n## Coverage Analysis\n- **Overall**: 78%\n- **Critical paths**: 92%\n- **Changed files**: 85%\n\n### Gaps Identified\n1. **src/auth.py:45-60** - Token validation edge cases\n2. **src/api/routes.py:120-135** - Error handling for 400/500 codes\n3. **src/db/migrations.py** - Rollback scenarios untested\n\n### Test-to-Feature Mapping\n- Feature: User registration → `tests/test_registration.py` (95%)\n- Feature: Password reset → `tests/test_auth.py` (60%) [WARN]\n- Feature: Email validation → Missing tests [FAIL]\n```\n\n## Best Practices\n\n1. **Branch coverage** over line coverage when available\n2. **Mutation testing** for critical code (e.g., `cargo mutants`, `mutmut`)\n3. **Coverage trends**: Track over time, not just absolute values\n4. **Exclude generated code**: Focus on hand-written logic\n5. **Integration coverage**: Don't just unit test in isolation\n\nFile v1.9.17:modules/framework-detection.md\n\n---\nparent_skill: pensive:test-review\nname: framework-detection\ndescription: Language and test framework detection patterns\ncategory: testing\ntags: [testing, framework-detection, language-detection]\nload_priority: 1\nestimated_tokens: 250\n---\n\n# Framework Detection\n\nIdentify testing frameworks and tooling constraints.\n\n## Language Detection Patterns\n\n### Rust\n- **Framework**: cargo test (built-in)\n- **Commands**: `cargo test`, `cargo nextest run`\n- **Config files**: `Cargo.toml`, `Cargo.lock`\n- **Test patterns**: `#[test]`, `#[cfg(test)]`\n- **MSRV**: Check `rust-version` in Cargo.toml\n\n### Python\n- **Frameworks**: pytest, unittest, behave\n- **Commands**: `pytest`, `python -m pytest`, `behave`\n- **Config files**: `pytest.ini`, `pyproject.toml`, `tox.ini`\n- **Test patterns**: `test_*.py`, `*_test.py`, `tests/`\n- **Version**: Check `requires-python` in pyproject.toml\n\n### JavaScript/TypeScript\n- **Frameworks**: Jest, Mocha, Cypress, Vitest\n- **Commands**: `npm test`, `yarn test`, `cypress run`\n- **Config files**: `jest.config.js`, `vitest.config.ts`, `cypress.config.js`\n- **Test patterns**: `*.test.js`, `*.spec.ts`, `__tests__/`\n- **Version**: Check `engines.node` in package.json\n\n### Go\n- **Framework**: go test (built-in)\n- **Commands**: `go test ./...`, `go test -v`\n- **Config files**: `go.mod`, `go.sum`\n- **Test patterns**: `*_test.go`\n- **Version**: Check `go` directive in go.mod\n\n## Detection Workflow\n\n1. **Scan for config files**:\n```bash\nfind . -maxdepth 2 -name \"Cargo.toml\" -o -name \"pyproject.toml\" -o -name \"package.json\" -o -name \"go.mod\"\n```\n\n2. **Check test directories**:\n```bash\nfind . -type d -name \"tests\" -o -name \"__tests__\" -o -name \"test\"\n```\n\n3. **Identify test files**:\n```bash\nfind . -not -path \"*/.venv/*\" -not -path \"*/__pycache__/*\" \\\n  -not -path \"*/node_modules/*\" -not -path \"*/.git/*\" \\\n  \\( -name \"*test*\" -o -name \"*spec*\" \\) \\\n  | grep -E '\\.(rs|py|js|ts|go)$'\n```\n\n4. **Version constraints**:\n- Extract MSRV, Python version, Node version\n- Note if constraints affect tooling (e.g., async/await)\n- Document CI/CD version requirements\n\n## Output Format\n\n```markdown\n## Framework Detection\n- **Languages**: Rust, Python\n- **Frameworks**: cargo test, pytest\n- **Versions**:\n  - Rust MSRV: 1.70\n  - Python: >=3.8\n- **Config files**: Cargo.toml, pyproject.toml\n```\n\nFile v1.9.17:modules/remediation-planning.md\n\n---\nparent_skill: pensive:test-review\nname: remediation-planning\ndescription: Test improvement strategies and phased remediation\ncategory: testing\ntags: [remediation, test-improvement, refactoring]\nload_priority: 4\nestimated_tokens: 300\n---\n\n# Remediation Planning\n\nConcrete strategies for test improvement.\n\n## Test Improvement Patterns\n\n### 1. Add Missing Coverage\nTie tests to specific behaviors using Given/When/Then:\n\n```markdown\n### Gap: Authentication edge cases\n**Behavior**: Given missing auth token, When hitting /v1/resource, Then HTTP 401 returned\n**Test**: `tests/test_auth.py::test_missing_token_returns_401`\n**Priority**: High (security boundary)\n```\n\n### 2. Refactor Test Helpers\n\n**Before** (repeated setup):\n```python\ndef test_user_creation():\n    db = setup_database()\n    config = load_test_config()\n    user_data = {\"email\": \"alice@example.com\", \"role\": \"user\"}\n    ...\n\ndef test_user_deletion():\n    db = setup_database()\n    config = load_test_config()\n    user_data = {\"email\": \"bob@example.com\", \"role\": \"admin\"}\n    ...\n```\n\n**After** (fixtures):\n```python\n@pytest.fixture\ndef test_db():\n    db = setup_database()\n    yield db\n    db.teardown()\n\n@pytest.fixture\ndef test_config():\n    return load_test_config()\n\ndef test_user_creation(test_db, test_config):\n    user_data = user_factory(email=\"alice@example.com\")\n    ...\n```\n\n### 3. Data Builders and Factories\n\n**Factory pattern**:\n```python\n# conftest.py\ndef user_factory(**overrides):\n    defaults = {\n        \"email\": \"user@example.com\",\n        \"role\": \"user\",\n        \"verified\": True,\n        \"created_at\": datetime.now()\n    }\n    return User(**{**defaults, **overrides})\n\n# test file\ndef test_admin_access():\n    admin = user_factory(role=\"admin\")\n    assert admin.can_access_dashboard()\n```\n\n**Builder pattern** (Rust):\n```rust\nstruct UserBuilder {\n    email: String,\n    role: Role,\n    verified: bool,\n}\n\nimpl UserBuilder {\n    fn new() -> Self {\n        Self {\n            email: \"user@example.com\".to_string(),\n            role: Role::User,\n            verified: true,\n        }\n    }\n\n    fn with_role(mut self, role: Role) -> Self {\n        self.role = role;\n        self\n    }\n\n    fn build(self) -> User {\n        User { /* ... */ }\n    }\n}\n\n#[test]\nfn test_admin_permissions() {\n    let admin = UserBuilder::new().with_role(Role::Admin).build();\n    assert!(admin.can_delete_users());\n}\n```\n\n### 4. Improve Assertions\n\n**Replace magic values**:\n```python\n# Before\nassert response.status == 200\n\n# After\nfrom http import HTTPStatus\nassert response.status == HTTPStatus.OK\n```\n\n**Add context**:\n```python\n# Before\nassert result\n\n# After\nassert result.success, f\"Expected success, got error: {result.error}\"\n```\n\n### 5. Remove Brittle Patterns\n\n**Dead waits** → **Explicit conditions**:\n```python\n# Before\ntime.sleep(3)\nassert element.visible\n\n# After\nwait_for(element.to_be_visible, timeout=5)\n```\n\n**Mocking internals** → **Mock boundaries**:\n```python\n# Before: mocking private implementation\n@patch('service._internal_helper')\ndef test_service(mock):\n    ...\n\n# After: mock external dependency\n@patch('requests.post')\ndef test_service(mock_requests):\n    ...\n```\n\n## Phased Remediation\n\nFor major test suite rewrites:\n\n### Phase 1: Stabilize (Week 1-2)\n1. Fix flaky tests (eliminate dead waits, order dependencies)\n2. Remove duplicate tests\n3. Add missing critical path tests\n4. **Metric**: Flaky test rate < 1%, critical paths 100%\n\n### Phase 2: Acceptance Specs (Week 3-4)\n1. Add BDD scenarios for user-facing features\n2. Create feature-to-test mapping\n3. Document test strategy per component\n4. **Metric**: All features have acceptance tests\n\n### Phase 3: Enforce Quality (Week 5+)\n1. Set coverage budgets (80% standard, 100% critical)\n2. Add pre-commit hooks for coverage checks\n3. Integrate mutation testing for critical code\n4. **Metric**: Coverage trends upward, no regressions\n\n## Recommendation Template\n\n```markdown\n## Remediation Plan\n\n### Immediate Actions (This Sprint)\n1. **Fix flaky test**: `test_user_login_retries` - Replace sleep with explicit wait\n   - Owner: @alice\n   - Due: 2025-12-10\n\n2. **Add missing coverage**: Password reset flow (currently 0%)\n   - Tests needed: valid token, expired token, invalid token\n   - Owner: @bob\n   - Due: 2025-12-12\n\n### Short-term (Next Sprint)\n3. **Refactor fixtures**: Extract common setup in `tests/test_api.py`\n   - Pattern: Use pytest fixtures for DB, config\n   - Owner: @charlie\n   - Due: 2025-12-20\n\n### Long-term (Next Month)\n4. **BDD acceptance tests**: User registration feature\n   - Tool: Behave/Gherkin\n   - Owner: @diana\n   - Due: 2025-01-15\n```\n\n## Exit Criteria\n\n- [ ] All critical gaps have assigned owners and due dates\n- [ ] Recommendations tied to specific behaviors\n- [ ] Phased approach for large refactorings\n- [ ] Success metrics defined (coverage %, flaky rate, etc.)\n\nFile v1.9.17:modules/scenario-quality.md\n\n---\nparent_skill: pensive:test-review\nname: scenario-quality\ndescription: Test scenario quality assessment with BDD patterns\ncategory: testing\ntags: [bdd, scenario-quality, assertions, anti-patterns]\nload_priority: 3\nestimated_tokens: 350\n---\n\n# Scenario Quality Assessment\n\nEvaluate test quality using BDD principles and assertion patterns.\n\n## Given/When/Then Clarity\n\n### Good Examples\n\n**Rust:**\n```rust\n#[test]\nfn test_authenticated_user_can_access_profile() {\n    // Given: authenticated user\n    let user = create_authenticated_user(\"alice@example.com\");\n    let token = generate_token(&user);\n\n    // When: accessing profile endpoint\n    let response = get(\"/profile\", &token);\n\n    // Then: profile data returned\n    assert_eq!(response.status, 200);\n    assert_eq!(response.body[\"email\"], \"alice@example.com\");\n}\n```\n\n**Python:**\n```python\ndef test_invalid_credentials_rejected():\n    # Given: user with wrong password\n    user = User(email=\"bob@example.com\")\n    wrong_password = \"incorrect\"\n\n    # When: attempting authentication\n    result = authenticate(user.email, wrong_password)\n\n    # Then: authentication fails with 401\n    assert result.status_code == 401\n    assert \"invalid credentials\" in result.error_message\n```\n\n**Gherkin (BDD):**\n```gherkin\nScenario: Registered user logs in successfully\n  Given a registered user with email \"alice@example.com\"\n  When they submit valid credentials\n  Then they receive an authentication token\n  And the token expires in 24 hours\n```\n\n## Assertion Quality\n\n### Bad Assertions (vague, brittle)\n```python\n# Too vague\nassert result\n\n# Multiple unrelated assertions\nassert len(users) > 0 and users[0].active and config.debug\n\n# Magic numbers without context\nassert response.status == 200\n```\n\n### Good Assertions (specific, meaningful)\n```python\n# Specific outcome\nassert result.status_code == 200, \"Expected successful login\"\n\n# Named constants\nassert response.status == HTTP_OK\nassert user.role == UserRole.ADMIN\n\n# Structured assertions\nassert response.json() == {\n    \"user\": {\"email\": expected_email, \"verified\": True},\n    \"token\": {\"expires_at\": ANY_DATETIME}\n}\n```\n\n## Anti-Patterns to Flag\n\n### 1. Dead Waits\n```python\n# BAD: arbitrary sleep\ntime.sleep(5)\nassert element.is_visible()\n\n# GOOD: explicit wait with condition\nwait_until(lambda: element.is_visible(), timeout=5)\n```\n\n### 2. Mocking Internals\n```python\n# BAD: mocking implementation details\n@patch('module.internal._private_helper')\ndef test_feature(mock_helper):\n    ...\n\n# GOOD: mock external dependencies only\n@patch('requests.get')\ndef test_api_call(mock_get):\n    ...\n```\n\n### 3. Repeated Boilerplate\n```python\n# BAD: copy-pasted setup\ndef test_user_creation():\n    db = Database(\"test.db\")\n    db.connect()\n    user = User(\"alice\")\n    ...\n\ndef test_user_deletion():\n    db = Database(\"test.db\")\n    db.connect()\n    user = User(\"bob\")\n    ...\n\n# GOOD: fixture/helper\n@pytest.fixture\ndef db_session():\n    db = Database(\"test.db\")\n    db.connect()\n    yield db\n    db.close()\n```\n\n### 4. Order Dependencies\n```python\n# BAD: tests depend on execution order\ndef test_01_create_user():\n    global user_id\n    user_id = create_user()\n\ndef test_02_delete_user():\n    delete_user(user_id)  # Depends on test_01!\n\n# GOOD: isolated tests\ndef test_delete_user():\n    user_id = create_user()  # Self-contained\n    delete_user(user_id)\n    assert not user_exists(user_id)\n```\n\n### 5. Multiple Assertions Without Context\n```python\n# BAD: unclear which assertion failed\nassert user.active\nassert user.verified\nassert user.role == \"admin\"\n\n# GOOD: grouped with context or separate tests\nassert user.active, \"User should be active\"\nassert user.verified, \"User should be verified\"\nassert user.role == \"admin\", \"User should have admin role\"\n```\n\n## BDD Suite Quality\n\n### Reusable Step Definitions\n```python\n# Good: parameterized, reusable\n@given('a user with email \"{email}\"')\ndef create_user(context, email):\n    context.user = User(email=email)\n\n@when('they submit credentials with password \"{password}\"')\ndef submit_credentials(context, password):\n    context.response = authenticate(context.user.email, password)\n```\n\n### Background Context Sharing\n```gherkin\nFeature: User authentication\n\n  Background:\n    Given a clean database\n    And the authentication service is running\n\n  Scenario: Valid login\n    Given a registered user\n    ...\n```\n\n### Scenario Outlines for Edge Cases\n```gherkin\nScenario Outline: Password validation\n  Given a user registering with password \"<password>\"\n  When they submit the registration form\n  Then they receive response \"<outcome>\"\n\n  Examples:\n    | password    | outcome           |\n    | abc         | too_short         |\n    | password123 | no_special_chars  |\n    | P@ssw0rd!   | success           |\n```\n\n## Quality Scoring\n\nScore each test file 1-5 on:\n- **Clarity**: Given/When/Then structure evident\n- **Assertions**: Specific, meaningful checks\n- **Isolation**: No shared state or order dependencies\n- **Maintainability**: DRY, uses fixtures/helpers\n- **Coverage**: Tests behavior, not implementation\n\n**Overall quality**:\n- 4-5: Excellent, minimal changes needed\n- 3: Good, some improvements recommended\n- 1-2: Poor, significant refactoring required\n\nFile v1.9.17:skill-card.md\n\n## Description: <br>\nEvaluates test suites for coverage gaps, TDD/BDD compliance, and anti-patterns. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[athola](https://clawhub.ai/user/athola) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and engineers use this skill to audit test suites before releases or after failures, identifying framework coverage, scenario quality issues, invariant erosion, and concrete remediation actions. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: Suggested local commands may inspect repository files, run tests, generate coverage artifacts, or install test tooling. <br>\nMitigation: Review each command for the target repository before execution, and approve package installation or coverage tooling explicitly. <br>\nRisk: Broad test-audit triggers can produce recommendations without enough project-specific evidence. <br>\nMitigation: Require the agent to log executed commands, outputs, coverage data, and cited evidence before accepting quality or release recommendations. <br>\n\n\n## Reference(s): <br>\n- [ClawHub skill page](https://clawhub.ai/athola/skills/nm-pensive-test-review) <br>\n- [Project homepage from ClawHub metadata](https://github.com/athola/claude-night-market/tree/master/plugins/pensive) <br>\n- [Publisher profile](https://clawhub.ai/user/athola) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [Analysis, Markdown, Shell commands, Guidance] <br>\n**Output Format:** [Markdown with structured review sections and inline shell commands] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [May include framework detection, coverage findings, quality issues, remediation plans, and approval recommendations.] <br>\n\n## Skill Version(s): <br>\n1.9.17 (source: server release metadata) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nArchive v1.9.16: 8 files, 13557 bytes\n\nFiles: modules/content-assertion-quality.md (2534b), modules/coverage-analysis.md (3330b), modules/framework-detection.md (2315b), modules/remediation-planning.md (4849b), modules/scenario-quality.md (5212b), skill-card.md (1929b), SKILL.md (8033b), _meta.json (142b)\n\nFile v1.9.16:SKILL.md\n\n---\nname: test-review\ndescription: Evaluates test suites for coverage gaps, TDD/BDD compliance, and anti-patterns\nversion: 1.9.8\ntriggers:\n  - testing\n  - tdd\n  - bdd\n  - coverage\n  - quality\n  - fixtures\n  - auditing test quality or before a major release\nmetadata: {\"openclaw\": {\"homepage\": \"https://github.com/athola/claude-night-market/tree/master/plugins/pensive\", \"emoji\": \"\\ud83e\\uddea\", \"requires\": {\"config\": [\"night-market.pensive:shared\", \"night-market.imbue:proof-of-work\"]}}}\nsource: claude-night-market\nsource_plugin: pensive\n---\n\n> **Night Market Skill** — ported from [claude-night-market/pensive](https://github.com/athola/claude-night-market/tree/master/plugins/pensive). For the full experience with agents, hooks, and commands, install the Claude Code plugin.\n\n\n## Table of Contents\n\n- [Quick Start](#quick-start)\n- [When to Use](#when-to-use)\n- [Required TodoWrite Items](#required-todowrite-items)\n- [Progressive Loading](#progressive-loading)\n- [Workflow](#workflow)\n- [Step 1: Detect Languages (`test-review:languages-detected`)](#step-1:-detect-languages-(test-review:languages-detected))\n- [Step 2: Inventory Coverage (`test-review:coverage-inventoried`)](#step-2:-inventory-coverage-(test-review:coverage-inventoried))\n- [Step 3: Assess Scenario Quality (`test-review:scenario-quality`)](#step-3:-assess-scenario-quality-(test-review:scenario-quality))\n- [Step 4: Plan Remediation (`test-review:gap-remediation`)](#step-4:-plan-remediation-(test-review:gap-remediation))\n- [Step 5: Log Evidence (`test-review:evidence-logged`)](#step-5:-log-evidence-(test-review:evidence-logged))\n- [Test Quality Checklist (Condensed)](#test-quality-checklist-(condensed))\n- [Output Format](#output-format)\n- [Summary](#summary)\n- [Framework Detection](#framework-detection)\n- [Coverage Analysis](#coverage-analysis)\n- [Quality Issues](#quality-issues)\n- [Remediation Plan](#remediation-plan)\n- [Recommendation](#recommendation)\n- [Integration Notes](#integration-notes)\n- [Exit Criteria](#exit-criteria)\n\n\n# Test Review Workflow\n\nEvaluate and improve test suites with TDD/BDD rigor.\n\n## Quick Start\n\n```bash\n/test-review\n```\n**Verification:** Run `pytest -v` to verify tests pass.\n\n## When To Use\n\n- Reviewing test suite quality\n- Analyzing coverage gaps\n- Before major releases\n- After test failures\n- Planning test improvements\n\n## When NOT To Use\n\n- Writing new tests - use parseltongue:python-testing\n- Updating existing tests - use sanctum:test-updates\n\n## Required TodoWrite Items\n\n1. `test-review:languages-detected`\n2. `test-review:coverage-inventoried`\n3. `test-review:scenario-quality`\n4. `test-review:invariant-preservation`\n5. `test-review:gap-remediation`\n6. `test-review:evidence-logged`\n\n## Progressive Loading\n\nLoad modules as needed based on review depth:\n\n- **Basic review**: Core workflow (this file)\n- **Framework detection**: Load `modules/framework-detection.md`\n- **Coverage analysis**: Load `modules/coverage-analysis.md`\n- **Quality assessment**: Load `modules/scenario-quality.md`\n- **Remediation planning**: Load `modules/remediation-planning.md`\n\n## Workflow\n\n### Step 1: Detect Languages (`test-review:languages-detected`)\n\nIdentify testing frameworks and version constraints.\n→ **See**: `modules/framework-detection.md`\n\nQuick check:\n```bash\nfind . -maxdepth 2 -name \"Cargo.toml\" -o -name \"pyproject.toml\" -o -name \"package.json\" -o -name \"go.mod\"\n```\n**Verification:** Run the command with `--help` flag to verify availability.\n\n### Step 2: Inventory Coverage (`test-review:coverage-inventoried`)\n\nRun coverage tools and identify gaps.\n→ **See**: `modules/coverage-analysis.md`\n\nQuick check:\n```bash\ngit diff --name-only | rg 'tests|spec|feature'\n```\n**Verification:** Run `pytest -v` to verify tests pass.\n\n### Step 3: Assess Scenario Quality (`test-review:scenario-quality`)\n\nEvaluate test quality using BDD patterns and assertion checks.\n→ **See**: `modules/scenario-quality.md`\n\nFocus on:\n- Given/When/Then clarity\n- Assertion specificity\n- Anti-patterns (dead waits, mocking internals, repeated boilerplate)\n\n### Step 4: Plan Remediation (`test-review:gap-remediation`)\n\nCreate concrete improvement plan with owners and dates.\n→ **See**: `modules/remediation-planning.md`\n\n### Step 5: Log Evidence (`test-review:evidence-logged`)\n\nRecord executed commands, outputs, and recommendations.\n→ **See**: `imbue:proof-of-work`\n\n## Test Quality Checklist (Condensed)\n\n- [ ] Clear test structure (Arrange-Act-Assert)\n- [ ] Critical paths covered (auth, validation, errors)\n- [ ] Specific assertions with context\n- [ ] No flaky tests (dead waits, order dependencies)\n- [ ] Reusable fixtures/factories\n- [ ] Invariant-encoding tests intact (see below)\n\n### Invariant-Encoding Tests\n\nTests do not just verify behavior — they encode design\ninvariants. A test that asserts \"module A never imports\nfrom module B\" encodes a layer boundary. A test that\nasserts \"this function is pure\" encodes a concurrency\nmodel. These tests are load-bearing in ways that\ncoverage metrics cannot capture.\n\n**During review, check:**\n\n1. **Were invariant-encoding tests removed or weakened?**\n   A test that enforced an architectural boundary,\n   data structure constraint, or API contract should\n   not be deleted without naming the invariant being\n   abandoned and escalating to human judgment.\n\n2. **Were test expectations changed to match a broken\n   implementation?** If an assertion value changed, ask:\n   did the *requirement* change, or did the agent change\n   the test to make its code pass? The latter is the\n   single most dangerous form of test tampering.\n\n3. **Are new invariants encoded as tests?** When a design\n   decision is made (choice of data structure, module\n   boundary, error strategy), there should be at least\n   one test whose failure would signal that the\n   invariant was violated.\n\n**Red flag patterns:**\n\n| Pattern | Risk |\n|---------|------|\n| `@pytest.mark.skip` added to a passing test | Invariant being silently dropped |\n| Assertion changed from specific to broad | Constraint being relaxed |\n| Test renamed to describe new behavior | Old invariant erased from history |\n| Test deleted \"because it tested old code\" | Invariant removed without replacement |\n\n**When invariant erosion is detected:**\n\nDo NOT approve. Flag as a BLOCKING quality issue and\npresent the three options to the human:\n\n1. **Preserve**: Revert the test change, fix the\n   implementation to satisfy the invariant\n2. **Layer**: Keep the invariant test, add the new\n   behavior alongside it (accepting inelegance)\n3. **Revise**: The invariant is genuinely wrong — remove\n   the old test AND write a new test encoding the\n   replacement invariant\n\nThis is a judgment call that models get wrong far too\noften. Default to option 1 (preserve) when no human is\navailable.\n\n## Output Format\n\n```markdown\n## Summary\n[Brief assessment]\n\n## Framework Detection\n- Languages: [list] | Frameworks: [list] | Versions: [constraints]\n\n## Coverage Analysis\n- Overall: X% | Critical: X% | Gaps: [list]\n\n## Quality Issues\n[Q1] [Issue] - Location - Fix\n\n## Remediation Plan\n1. [Action] - Owner - Date\n\n## Recommendation\nApprove / Approve with actions / Block\n```\n**Verification:** Run the command with `--help` flag to verify availability.\n\n## Integration Notes\n\n- Use `imbue:proof-of-work` for reproducible evidence capture\n- Reference `imbue:diff-analysis` for risk assessment\n- Format output using `imbue:structured-output` patterns\n\n## Exit Criteria\n\n- Frameworks detected and documented\n- Coverage analyzed and gaps identified\n- Scenario quality assessed\n- Remediation plan created with owners and dates\n- Evidence logged with citations\n## Troubleshooting\n\n### Common Issues\n\n**Tests not discovered**\nEnsure test files match pattern `test_*.py` or `*_test.py`. Run `pytest --collect-only` to verify.\n\n**Import errors**\nCheck that the module being tested is in `PYTHONPATH` or install with `pip install -e .`\n\n**Async tests failing**\nInstall pytest-asyncio and decorate test functions with `@pytest.mark.asyncio`\n\nFile v1.9.16:_meta.json\n\n{\n  \"ownerId\": \"kn7d107jg9jv602h9ytsegydq184a42s\",\n  \"slug\": \"nm-pensive-test-review\",\n  \"version\": \"1.9.16\",\n  \"publishedAt\": 1784058988437\n}\n\nFile v1.9.16:modules/content-assertion-quality.md\n\n# Content Assertion Quality\n\nScoring criteria for evaluating content assertion tests during test review. Extends the scenario quality assessment with a Content Depth dimension.\n\nReference: `leyline:testing-quality-standards/modules/content-assertion-levels.md`\n\n## Content Depth Scoring\n\nRate content assertion depth on a 1-5 scale:\n\n| Score | Level | Description |\n|---|---|---|\n| 1 | None | Tests only file existence or line count |\n| 2 | L1 | Keyword presence checks (`assert \"section\" in content`) |\n| 3 | L2 | Parses embedded examples, validates schema structure |\n| 4 | L3 | Cross-references, anti-patterns, decision framework contracts |\n| 5 | L3+ | Cross-plugin validation (version refs checked against other plugins' docs) |\n\n## When to Flag Missing Content Assertions\n\nDuring test review, flag as a content test gap when:\n\n- A skill has tests but all are L1 (keyword-only) and the skill contains JSON or YAML code blocks\n- A skill has version-gated features but no cross-reference validation\n- A skill defines behavioral guidance (decision trees, strategies) but no anti-pattern or completeness tests\n- A module documents forbidden behaviors but no test asserts their absence\n\n## Content Assertion Anti-Patterns\n\nAvoid these when reviewing content tests:\n\n| Anti-Pattern | Problem | Better Approach |\n|---|---|---|\n| Testing prose style | Brittle to rewording, overlaps with scribe:slop-detector | Test behavioral semantics |\n| Asserting exact wording | Breaks on any edit | Assert concepts (`\"version\" in content.lower()`) |\n| Checking line counts | Not behavioral | Check required sections exist |\n| Testing formatting | Not what Claude interprets | Test parseable structure |\n| Duplicating slop detection | Already handled by scribe | Focus on correctness, not style |\n\n## Review Checklist Addition\n\nAdd this item to the existing Test Quality Checklist when reviewing a plugin that has execution markdown:\n\n```markdown\n- [ ] Content assertion depth matches content complexity\n      (L1 for simple skills, L2+ for code examples, L3 for behavioral guidance)\n```\n\n## Remediation Guidance\n\nWhen content tests are missing or insufficient:\n\n1. **No content tests at all**: Generate L1 scaffolding using `sanctum:test-updates/modules/generation/content-test-templates.md`\n2. **L1 only, has code blocks**: Upgrade to L2 (add JSON/YAML parsing tests)\n3. **L2 only, has version gates**: Upgrade to L3 (add cross-reference validation)\n4. **L2 only, has behavioral guidance**: Upgrade to L3 (add anti-pattern and completeness tests)\n\nFile v1.9.16:modules/coverage-analysis.md\n\n---\nparent_skill: pensive:test-review\nname: coverage-analysis\ndescription: Coverage measurement and gap identification\ncategory: testing\ntags: [coverage, testing, gap-analysis]\nload_priority: 2\nestimated_tokens: 350\n---\n\n# Coverage Analysis\n\nMeasure test coverage and identify gaps.\n\n## Coverage Tools by Language\n\n### Rust\n```bash\n# Using tarpaulin\ncargo install cargo-tarpaulin\ncargo tarpaulin --out Html --output-dir coverage/\n\n# Using llvm-cov\ncargo install cargo-llvm-cov\ncargo llvm-cov --html\n```\n\n### Python\n```bash\n# Using pytest-cov\npytest --cov=src --cov-report=html --cov-report=term-missing\n\n# Using coverage.py\ncoverage run -m pytest\ncoverage html\ncoverage report --show-missing\n```\n\n### JavaScript/TypeScript\n```bash\n# Jest\nnpm test -- --coverage --coverageReporters=html text\n\n# Vitest\nvitest --coverage\n\n# Cypress (code coverage plugin)\ncypress run --env coverage=true\n```\n\n### Go\n```bash\n# Built-in coverage\ngo test -cover ./...\ngo test -coverprofile=coverage.out ./...\ngo tool cover -html=coverage.out\n\n# Detailed coverage\ngo test -covermode=count -coverprofile=coverage.out ./...\n```\n\n## Coverage Thresholds\n\n| Level | Coverage | Use Case |\n|-------|----------|----------|\n| Minimum | 60% | Legacy code, initial cleanup |\n| Standard | 80% | Normal development |\n| High | 90% | Critical systems, libraries |\n| detailed | 95%+ | Safety-critical, financial |\n\n## Gap Identification\n\n### Find impacted test files\n```bash\n# Tests affected by changes\ngit diff --name-only main...HEAD | rg 'tests|spec|feature'\n\n# Find related tests\ngit diff --name-only main...HEAD | while read file; do\n  basename \"$file\" .py | xargs -I {} find . \\\n    -not -path \"*/.venv/*\" -not -path \"*/__pycache__/*\" \\\n    -not -path \"*/node_modules/*\" -not -path \"*/.git/*\" \\\n    -name \"*test*{}*\"\ndone\n```\n\n### Identify uncovered code\n1. Run coverage tool with `--show-missing` flag\n2. Cross-reference with critical paths:\n   - Authentication/authorization\n   - Data validation\n   - Error handling\n   - API endpoints\n   - Database operations\n\n3. Map to requirements:\n   - Feature specifications\n   - User stories\n   - Bug reports\n   - Security requirements\n\n### Coverage Patterns\n\n**Critical paths** (should be 100%):\n- Security boundaries (auth, validation)\n- Data integrity operations\n- Error recovery logic\n- Public API surface\n\n**Lower priority** (can be <80%):\n- Internal helpers\n- Logging/debugging code\n- Trivial getters/setters\n- Deprecated code paths\n\n## Output Format\n\n```markdown\n## Coverage Analysis\n- **Overall**: 78%\n- **Critical paths**: 92%\n- **Changed files**: 85%\n\n### Gaps Identified\n1. **src/auth.py:45-60** - Token validation edge cases\n2. **src/api/routes.py:120-135** - Error handling for 400/500 codes\n3. **src/db/migrations.py** - Rollback scenarios untested\n\n### Test-to-Feature Mapping\n- Feature: User registration → `tests/test_registration.py` (95%)\n- Feature: Password reset → `tests/test_auth.py` (60%) [WARN]\n- Feature: Email validation → Missing tests [FAIL]\n```\n\n## Best Practices\n\n1. **Branch coverage** over line coverage when available\n2. **Mutation testing** for critical code (e.g., `cargo mutants`, `mutmut`)\n3. **Coverage trends**: Track over time, not just absolute values\n4. **Exclude generated code**: Focus on hand-written logic\n5. **Integration coverage**: Don't just unit test in isolation\n\nFile v1.9.16:modules/framework-detection.md\n\n---\nparent_skill: pensive:test-review\nname: framework-detection\ndescription: Language and test framework detection patterns\ncategory: testing\ntags: [testing, framework-detection, language-detection]\nload_priority: 1\nestimated_tokens: 250\n---\n\n# Framework Detection\n\nIdentify testing frameworks and tooling constraints.\n\n## Language Detection Patterns\n\n### Rust\n- **Framework**: cargo test (built-in)\n- **Commands**: `cargo test`, `cargo nextest run`\n- **Config files**: `Cargo.toml`, `Cargo.lock`\n- **Test patterns**: `#[test]`, `#[cfg(test)]`\n- **MSRV**: Check `rust-version` in Cargo.toml\n\n### Python\n- **Frameworks**: pytest, unittest, behave\n- **Commands**: `pytest`, `python -m pytest`, `behave`\n- **Config files**: `pytest.ini`, `pyproject.toml`, `tox.ini`\n- **Test patterns**: `test_*.py`, `*_test.py`, `tests/`\n- **Version**: Check `requires-python` in pyproject.toml\n\n### JavaScript/TypeScript\n- **Frameworks**: Jest, Mocha, Cypress, Vitest\n- **Commands**: `npm test`, `yarn test`, `cypress run`\n- **Config files**: `jest.config.js`, `vitest.config.ts`, `cypress.config.js`\n- **Test patterns**: `*.test.js`, `*.spec.ts`, `__tests__/`\n- **Version**: Check `engines.node` in package.json\n\n### Go\n- **Framework**: go test (built-in)\n- **Commands**: `go test ./...`, `go test -v`\n- **Config files**: `go.mod`, `go.sum`\n- **Test patterns**: `*_test.go`\n- **Version**: Check `go` directive in go.mod\n\n## Detection Workflow\n\n1. **Scan for config files**:\n```bash\nfind . -maxdepth 2 -name \"Cargo.toml\" -o -name \"pyproject.toml\" -o -name \"package.json\" -o -name \"go.mod\"\n```\n\n2. **Check test directories**:\n```bash\nfind . -type d -name \"tests\" -o -name \"__tests__\" -o -name \"test\"\n```\n\n3. **Identify test files**:\n```bash\nfind . -not -path \"*/.venv/*\" -not -path \"*/__pycache__/*\" \\\n  -not -path \"*/node_modules/*\" -not -path \"*/.git/*\" \\\n  \\( -name \"*test*\" -o -name \"*spec*\" \\) \\\n  | grep -E '\\.(rs|py|js|ts|go)$'\n```\n\n4. **Version constraints**:\n- Extract MSRV, Python version, Node version\n- Note if constraints affect tooling (e.g., async/await)\n- Document CI/CD version requirements\n\n## Output Format\n\n```markdown\n## Framework Detection\n- **Languages**: Rust, Python\n- **Frameworks**: cargo test, pytest\n- **Versions**:\n  - Rust MSRV: 1.70\n  - Python: >=3.8\n- **Config files**: Cargo.toml, pyproject.toml\n```\n\nFile v1.9.16:modules/remediation-planning.md\n\n---\nparent_skill: pensive:test-review\nname: remediation-planning\ndescription: Test improvement strategies and phased remediation\ncategory: testing\ntags: [remediation, test-improvement, refactoring]\nload_priority: 4\nestimated_tokens: 300\n---\n\n# Remediation Planning\n\nConcrete strategies for test improvement.\n\n## Test Improvement Patterns\n\n### 1. Add Missing Coverage\nTie tests to specific behaviors using Given/When/Then:\n\n```markdown\n### Gap: Authentication edge cases\n**Behavior**: Given missing auth token, When hitting /v1/resource, Then HTTP 401 returned\n**Test**: `tests/test_auth.py::test_missing_token_returns_401`\n**Priority**: High (security boundary)\n```\n\n### 2. Refactor Test Helpers\n\n**Before** (repeated setup):\n```python\ndef test_user_creation():\n    db = setup_database()\n    config = load_test_config()\n    user_data = {\"email\": \"alice@example.com\", \"role\": \"user\"}\n    ...\n\ndef test_user_deletion():\n    db = setup_database()\n    config = load_test_config()\n    user_data = {\"email\": \"bob@example.com\", \"role\": \"admin\"}\n    ...\n```\n\n**After** (fixtures):\n```python\n@pytest.fixture\ndef test_db():\n    db = setup_database()\n    yield db\n    db.teardown()\n\n@pytest.fixture\ndef test_config():\n    return load_test_config()\n\ndef test_user_creation(test_db, test_config):\n    user_data = user_factory(email=\"alice@example.com\")\n    ...\n```\n\n### 3. Data Builders and Factories\n\n**Factory pattern**:\n```python\n# conftest.py\ndef user_factory(**overrides):\n    defaults = {\n        \"email\": \"user@example.com\",\n        \"role\": \"user\",\n        \"verified\": True,\n        \"created_at\": datetime.now()\n    }\n    return User(**{**defaults, **overrides})\n\n# test file\ndef test_admin_access():\n    admin = user_factory(role=\"admin\")\n    assert admin.can_access_dashboard()\n```\n\n**Builder pattern** (Rust):\n```rust\nstruct UserBuilder {\n    email: String,\n    role: Role,\n    verified: bool,\n}\n\nimpl UserBuilder {\n    fn new() -> Self {\n        Self {\n            email: \"user@example.com\".to_string(),\n            role: Role::User,\n            verified: true,\n        }\n    }\n\n    fn with_role(mut self, role: Role) -> Self {\n        self.role = role;\n        self\n    }\n\n    fn build(self) -> User {\n        User { /* ... */ }\n    }\n}\n\n#[test]\nfn test_admin_permissions() {\n    let admin = UserBuilder::new().with_role(Role::Admin).build();\n    assert!(admin.can_delete_users());\n}\n```\n\n### 4. Improve Assertions\n\n**Replace magic values**:\n```python\n# Before\nassert response.status == 200\n\n# After\nfrom http import HTTPStatus\nassert response.status == HTTPStatus.OK\n```\n\n**Add context**:\n```python\n# Before\nassert result\n\n# After\nassert result.success, f\"Expected success, got error: {result.error}\"\n```\n\n### 5. Remove Brittle Patterns\n\n**Dead waits** → **Explicit conditions**:\n```python\n# Before\ntime.sleep(3)\nassert element.visible\n\n# After\nwait_for(element.to_be_visible, timeout=5)\n```\n\n**Mocking internals** → **Mock boundaries**:\n```python\n# Before: mocking private implementation\n@patch('service._internal_helper')\ndef test_service(mock):\n    ...\n\n# After: mock external dependency\n@patch('requests.post')\ndef test_service(mock_requests):\n    ...\n```\n\n## Phased Remediation\n\nFor major test suite rewrites:\n\n### Phase 1: Stabilize (Week 1-2)\n1. Fix flaky tests (eliminate dead waits, order dependencies)\n2. Remove duplicate tests\n3. Add missing critical path tests\n4. **Metric**: Flaky test rate < 1%, critical paths 100%\n\n### Phase 2: Acceptance Specs (Week 3-4)\n1. Add BDD scenarios for user-facing features\n2. Create feature-to-test mapping\n3. Document test strategy per component\n4. **Metric**: All features have acceptance tests\n\n### Phase 3: Enforce Quality (Week 5+)\n1. Set coverage budgets (80% standard, 100% critical)\n2. Add pre-commit hooks for coverage checks\n3. Integrate mutation testing for critical code\n4. **Metric**: Coverage trends upward, no regressions\n\n## Recommendation Template\n\n```markdown\n## Remediation Plan\n\n### Immediate Actions (This Sprint)\n1. **Fix flaky test**: `test_user_login_retries` - Replace sleep with explicit wait\n   - Owner: @alice\n   - Due: 2025-12-10\n\n2. **Add missing coverage**: Password reset flow (currently 0%)\n   - Tests needed: valid token, expired token, invalid token\n   - Owner: @bob\n   - Due: 2025-12-12\n\n### Short-term (Next Sprint)\n3. **Refactor fixtures**: Extract common setup in `tests/test_api.py`\n   - Pattern: Use pytest fixtures for DB, config\n   - Owner: @charlie\n   - Due: 2025-12-20\n\n### Long-term (Next Month)\n4. **BDD acceptance tests**: User registration feature\n   - Tool: Behave/Gherkin\n   - Owner: @diana\n   - Due: 2025-01-15\n```\n\n## Exit Criteria\n\n- [ ] All critical gaps have assigned owners and due dates\n- [ ] Recommendations tied to specific behaviors\n- [ ] Phased approach for large refactorings\n- [ ] Success metrics defined (coverage %, flaky rate, etc.)\n\nFile v1.9.16:modules/scenario-quality.md\n\n---\nparent_skill: pensive:test-review\nname: scenario-quality\ndescription: Test scenario quality assessment with BDD patterns\ncategory: testing\ntags: [bdd, scenario-quality, assertions, anti-patterns]\nload_priority: 3\nestimated_tokens: 350\n---\n\n# Scenario Quality Assessment\n\nEvaluate test quality using BDD principles and assertion patterns.\n\n## Given/When/Then Clarity\n\n### Good Examples\n\n**Rust:**\n```rust\n#[test]\nfn test_authenticated_user_can_access_profile() {\n    // Given: authenticated user\n    let user = create_authenticated_user(\"alice@example.com\");\n    let token = generate_token(&user);\n\n    // When: accessing profile endpoint\n    let response = get(\"/profile\", &token);\n\n    // Then: profile data returned\n    assert_eq!(response.status, 200);\n    assert_eq!(response.body[\"email\"], \"alice@example.com\");\n}\n```\n\n**Python:**\n```python\ndef test_invalid_credentials_rejected():\n    # Given: user with wrong password\n    user = User(email=\"bob@example.com\")\n    wrong_password = \"incorrect\"\n\n    # When: attempting authentication\n    result = authenticate(user.email, wrong_password)\n\n    # Then: authentication fails with 401\n    assert result.status_code == 401\n    assert \"invalid credentials\" in result.error_message\n```\n\n**Gherkin (BDD):**\n```gherkin\nScenario: Registered user logs in successfully\n  Given a registered user with email \"alice@example.com\"\n  When they submit valid credentials\n  Then they receive an authentication token\n  And the token expires in 24 hours\n```\n\n## Assertion Quality\n\n### Bad Assertions (vague, brittle)\n```python\n# Too vague\nassert result\n\n# Multiple unrelated assertions\nassert len(users) > 0 and users[0].active and config.debug\n\n# Magic numbers without context\nassert response.status == 200\n```\n\n### Good Assertions (specific, meaningful)\n```python\n# Specific outcome\nassert result.status_code == 200, \"Expected successful login\"\n\n# Named constants\nassert response.status == HTTP_OK\nassert user.role == UserRole.ADMIN\n\n# Structured assertions\nassert response.json() == {\n    \"user\": {\"email\": expected_email, \"verified\": True},\n    \"token\": {\"expires_at\": ANY_DATETIME}\n}\n```\n\n## Anti-Patterns to Flag\n\n### 1. Dead Waits\n```python\n# BAD: arbitrary sleep\ntime.sleep(5)\nassert element.is_visible()\n\n# GOOD: explicit wait with condition\nwait_until(lambda: element.is_visible(), timeout=5)\n```\n\n### 2. Mocking Internals\n```python\n# BAD: mocking implementation details\n@patch('module.internal._private_helper')\ndef test_feature(mock_helper):\n    ...\n\n# GOOD: mock external dependencies only\n@patch('requests.get')\ndef test_api_call(mock_get):\n    ...\n```\n\n### 3. Repeated Boilerplate\n```python\n# BAD: copy-pasted setup\ndef test_user_creation():\n    db = Database(\"test.db\")\n    db.connect()\n    user = User(\"alice\")\n    ...\n\ndef test_user_deletion():\n    db = Database(\"test.db\")\n    db.connect()\n    user = User(\"bob\")\n    ...\n\n# GOOD: fixture/helper\n@pytest.fixture\ndef db_session():\n    db = Database(\"test.db\")\n    db.connect()\n    yield db\n    db.close()\n```\n\n### 4. Order Dependencies\n```python\n# BAD: tests depend on execution order\ndef test_01_create_user():\n    global user_id\n    user_id = create_user()\n\ndef test_02_delete_user():\n    delete_user(user_id)  # Depends on test_01!\n\n# GOOD: isolated tests\ndef test_delete_user():\n    user_id = create_user()  # Self-contained\n    delete_user(user_id)\n    assert not user_exists(user_id)\n```\n\n### 5. Multiple Assertions Without Context\n```python\n# BAD: unclear which assertion failed\nassert user.active\nassert user.verified\nassert user.role == \"admin\"\n\n# GOOD: grouped with context or separate tests\nassert user.active, \"User should be active\"\nassert user.verified, \"User should be verified\"\nassert user.role == \"admin\", \"User should have admin role\"\n```\n\n## BDD Suite Quality\n\n### Reusable Step Definitions\n```python\n# Good: parameterized, reusable\n@given('a user with email \"{email}\"')\ndef create_user(context, email):\n    context.user = User(email=email)\n\n@when('they submit credentials with password \"{password}\"')\ndef submit_credentials(context, password):\n    context.response = authenticate(context.user.email, password)\n```\n\n### Background Context Sharing\n```gherkin\nFeature: User authentication\n\n  Background:\n    Given a clean database\n    And the authentication service is running\n\n  Scenario: Valid login\n    Given a registered user\n    ...\n```\n\n### Scenario Outlines for Edge Cases\n```gherkin\nScenario Outline: Password validation\n  Given a user registering with password \"<password>\"\n  When they submit the registration form\n  Then they receive response \"<outcome>\"\n\n  Examples:\n    | password    | outcome           |\n    | abc         | too_short         |\n    | password123 | no_special_chars  |\n    | P@ssw0rd!   | success           |\n```\n\n## Quality Scoring\n\nScore each test file 1-5 on:\n- **Clarity**: Given/When/Then structure evident\n- **Assertions**: Specific, meaningful checks\n- **Isolation**: No shared state or order dependencies\n- **Maintainability**: DRY, uses fixtures/helpers\n- **Coverage**: Tests behavior, not implementation\n\n**Overall quality**:\n- 4-5: Excellent, minimal changes needed\n- 3: Good, some improvements recommended\n- 1-2: Poor, significant refactoring required\n\nFile v1.9.16:skill-card.md\n\n## Description: <br>\nEvaluates test suites for coverage gaps, TDD/BDD compliance, and anti-patterns. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[athola](https://clawhub.ai/user/athola) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and engineers use this skill to review test-suite quality, identify coverage gaps, assess assertion and scenario quality, and plan remediation before releases or after test failures. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: The skill may activate on broad testing-related prompts. <br>\nMitigation: Use it when test-suite review, coverage analysis, or release-readiness assessment is intended. <br>\nRisk: The skill can suggest package installation, test execution, and coverage commands. <br>\nMitigation: Review commands before running them, especially in large or sensitive repositories. <br>\n\n\n## Reference(s): <br>\n- [ClawHub Skill Page](https://clawhub.ai/athola/skills/nm-pensive-test-review) <br>\n- [Pensive Plugin Homepage](https://github.com/athola/claude-night-market/tree/master/plugins/pensive) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [analysis, markdown, shell commands, guidance] <br>\n**Output Format:** [Markdown with structured findings, remediation plans, and inline shell commands] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [May include coverage summaries, quality issues, recommendations, and commands for framework detection or test execution.] <br>\n\n## Skill Version(s): <br>\n1.9.16 (source: server release metadata) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nArchive v1.9.14: 8 files, 13547 bytes\n\nFiles: modules/content-assertion-quality.md (2534b), modules/coverage-analysis.md (3330b), modules/framework-detection.md (2315b), modules/remediation-planning.md (4849b), modules/scenario-quality.md (5212b), skill-card.md (1845b), SKILL.md (8033b), _meta.json (142b)\n\nFile v1.9.14:SKILL.md\n\n---\nname: test-review\ndescription: Evaluates test suites for coverage gaps, TDD/BDD compliance, and anti-patterns\nversion: 1.9.8\ntriggers:\n  - testing\n  - tdd\n  - bdd\n  - coverage\n  - quality\n  - fixtures\n  - auditing test quality or before a major release\nmetadata: {\"openclaw\": {\"homepage\": \"https://github.com/athola/claude-night-market/tree/master/plugins/pensive\", \"emoji\": \"\\ud83e\\uddea\", \"requires\": {\"config\": [\"night-market.pensive:shared\", \"night-market.imbue:proof-of-work\"]}}}\nsource: claude-night-market\nsource_plugin: pensive\n---\n\n> **Night Market Skill** — ported from [claude-night-market/pensive](https://github.com/athola/claude-night-market/tree/master/plugins/pensive). For the full experience with agents, hooks, and commands, install the Claude Code plugin.\n\n\n## Table of Contents\n\n- [Quick Start](#quick-start)\n- [When to Use](#when-to-use)\n- [Required TodoWrite Items](#required-todowrite-items)\n- [Progressive Loading](#progressive-loading)\n- [Workflow](#workflow)\n- [Step 1: Detect Languages (`test-review:languages-detected`)](#step-1:-detect-languages-(test-review:languages-detected))\n- [Step 2: Inventory Coverage (`test-review:coverage-inventoried`)](#step-2:-inventory-coverage-(test-review:coverage-inventoried))\n- [Step 3: Assess Scenario Quality (`test-review:scenario-quality`)](#step-3:-assess-scenario-quality-(test-review:scenario-quality))\n- [Step 4: Plan Remediation (`test-review:gap-remediation`)](#step-4:-plan-remediation-(test-review:gap-remediation))\n- [Step 5: Log Evidence (`test-review:evidence-logged`)](#step-5:-log-evidence-(test-review:evidence-logged))\n- [Test Quality Checklist (Condensed)](#test-quality-checklist-(condensed))\n- [Output Format](#output-format)\n- [Summary](#summary)\n- [Framework Detection](#framework-detection)\n- [Coverage Analysis](#coverage-analysis)\n- [Quality Issues](#quality-issues)\n- [Remediation Plan](#remediation-plan)\n- [Recommendation](#recommendation)\n- [Integration Notes](#integration-notes)\n- [Exit Criteria](#exit-criteria)\n\n\n# Test Review Workflow\n\nEvaluate and improve test suites with TDD/BDD rigor.\n\n## Quick Start\n\n```bash\n/test-review\n```\n**Verification:** Run `pytest -v` to verify tests pass.\n\n## When To Use\n\n- Reviewing test suite quality\n- Analyzing coverage gaps\n- Before major releases\n- After test failures\n- Planning test improvements\n\n## When NOT To Use\n\n- Writing new tests - use parseltongue:python-testing\n- Updating existing tests - use sanctum:test-updates\n\n## Required TodoWrite Items\n\n1. `test-review:languages-detected`\n2. `test-review:coverage-inventoried`\n3. `test-review:scenario-quality`\n4. `test-review:invariant-preservation`\n5. `test-review:gap-remediation`\n6. `test-review:evidence-logged`\n\n## Progressive Loading\n\nLoad modules as needed based on review depth:\n\n- **Basic review**: Core workflow (this file)\n- **Framework detection**: Load `modules/framework-detection.md`\n- **Coverage analysis**: Load `modules/coverage-analysis.md`\n- **Quality assessment**: Load `modules/scenario-quality.md`\n- **Remediation planning**: Load `modules/remediation-planning.md`\n\n## Workflow\n\n### Step 1: Detect Languages (`test-review:languages-detected`)\n\nIdentify testing frameworks and version constraints.\n→ **See**: `modules/framework-detection.md`\n\nQuick check:\n```bash\nfind . -maxdepth 2 -name \"Cargo.toml\" -o -name \"pyproject.toml\" -o -name \"package.json\" -o -name \"go.mod\"\n```\n**Verification:** Run the command with `--help` flag to verify availability.\n\n### Step 2: Inventory Coverage (`test-review:coverage-inventoried`)\n\nRun coverage tools and identify gaps.\n→ **See**: `modules/coverage-analysis.md`\n\nQuick check:\n```bash\ngit diff --name-only | rg 'tests|spec|feature'\n```\n**Verification:** Run `pytest -v` to verify tests pass.\n\n### Step 3: Assess Scenario Quality (`test-review:scenario-quality`)\n\nEvaluate test quality using BDD patterns and assertion checks.\n→ **See**: `modules/scenario-quality.md`\n\nFocus on:\n- Given/When/Then clarity\n- Assertion specificity\n- Anti-patterns (dead waits, mocking internals, repeated boilerplate)\n\n### Step 4: Plan Remediation (`test-review:gap-remediation`)\n\nCreate concrete improvement plan with owners and dates.\n→ **See**: `modules/remediation-planning.md`\n\n### Step 5: Log Evidence (`test-review:evidence-logged`)\n\nRecord executed commands, outputs, and recommendations.\n→ **See**: `imbue:proof-of-work`\n\n## Test Quality Checklist (Condensed)\n\n- [ ] Clear test structure (Arrange-Act-Assert)\n- [ ] Critical paths covered (auth, validation, errors)\n- [ ] Specific assertions with context\n- [ ] No flaky tests (dead waits, order dependencies)\n- [ ] Reusable fixtures/factories\n- [ ] Invariant-encoding tests intact (see below)\n\n### Invariant-Encoding Tests\n\nTests do not just verify behavior — they encode design\ninvariants. A test that asserts \"module A never imports\nfrom module B\" encodes a layer boundary. A test that\nasserts \"this function is pure\" encodes a concurrency\nmodel. These tests are load-bearing in ways that\ncoverage metrics cannot capture.\n\n**During review, check:**\n\n1. **Were invariant-encoding tests removed or weakened?**\n   A test that enforced an architectural boundary,\n   data structure constraint, or API contract should\n   not be deleted without naming the invariant being\n   abandoned and escalating to human judgment.\n\n2. **Were test expectations changed to match a broken\n   implementation?** If an assertion value changed, ask:\n   did the *requirement* change, or did the agent change\n   the test to make its code pass? The latter is the\n   single most dangerous form of test tampering.\n\n3. **Are new invariants encoded as tests?** When a design\n   decision is made (choice of data structure, module\n   boundary, error strategy), there should be at least\n   one test whose failure would signal that the\n   invariant was violated.\n\n**Red flag patterns:**\n\n| Pattern | Risk |\n|---------|------|\n| `@pytest.mark.skip` added to a passing test | Invariant being silently dropped |\n| Assertion changed from specific to broad | Constraint being relaxed |\n| Test renamed to describe new behavior | Old invariant erased from history |\n| Test deleted \"because it tested old code\" | Invariant removed without replacement |\n\n**When invariant erosion is detected:**\n\nDo NOT approve. Flag as a BLOCKING quality issue and\npresent the three options to the human:\n\n1. **Preserve**: Revert the test change, fix the\n   implementation to satisfy the invariant\n2. **Layer**: Keep the invariant test, add the new\n   behavior alongside it (accepting inelegance)\n3. **Revise**: The invariant is genuinely wrong — remove\n   the old test AND write a new test encoding the\n   replacement invariant\n\nThis is a judgment call that models get wrong far too\noften. Default to option 1 (preserve) when no human is\navailable.\n\n## Output Format\n\n```markdown\n## Summary\n[Brief assessment]\n\n## Framework Detection\n- Languages: [list] | Frameworks: [list] | Versions: [constraints]\n\n## Coverage Analysis\n- Overall: X% | Critical: X% | Gaps: [list]\n\n## Quality Issues\n[Q1] [Issue] - Location - Fix\n\n## Remediation Plan\n1. [Action] - Owner - Date\n\n## Recommendation\nApprove / Approve with actions / Block\n```\n**Verification:** Run the command with `--help` flag to verify availability.\n\n## Integration Notes\n\n- Use `imbue:proof-of-work` for reproducible evidence capture\n- Reference `imbue:diff-analysis` for risk assessment\n- Format output using `imbue:structured-output` patterns\n\n## Exit Criteria\n\n- Frameworks detected and documented\n- Coverage analyzed and gaps identified\n- Scenario quality assessed\n- Remediation plan created with owners and dates\n- Evidence logged with citations\n## Troubleshooting\n\n### Common Issues\n\n**Tests not discovered**\nEnsure test files match pattern `test_*.py` or `*_test.py`. Run `pytest --collect-only` to verify.\n\n**Import errors**\nCheck that the module being tested is in `PYTHONPATH` or install with `pip install -e .`\n\n**Async tests failing**\nInstall pytest-asyncio and decorate test functions with `@pytest.mark.asyncio`\n\nFile v1.9.14:_meta.json\n\n{\n  \"ownerId\": \"kn7d107jg9jv602h9ytsegydq184a42s\",\n  \"slug\": \"nm-pensive-test-review\",\n  \"version\": \"1.9.14\",\n  \"publishedAt\": 1782842681848\n}\n\nFile v1.9.14:modules/content-assertion-quality.md\n\n# Content Assertion Quality\n\nScoring criteria for evaluating content assertion tests during test review. Extends the scenario quality assessment with a Content Depth dimension.\n\nReference: `leyline:testing-quality-standards/modules/content-assertion-levels.md`\n\n## Content Depth Scoring\n\nRate content assertion depth on a 1-5 scale:\n\n| Score | Level | Description |\n|---|---|---|\n| 1 | None | Tests only file existence or line count |\n| 2 | L1 | Keyword presence checks (`assert \"section\" in content`) |\n| 3 | L2 | Parses embedded examples, validates schema structure |\n| 4 | L3 | Cross-references, anti-patterns, decision framework contracts |\n| 5 | L3+ | Cross-plugin validation (version refs checked against other plugins' docs) |\n\n## When to Flag Missing Content Assertions\n\nDuring test review, flag as a content test gap when:\n\n- A skill has tests but all are L1 (keyword-only) and the skill contains JSON or YAML code blocks\n- A skill has version-gated features but no cross-reference validation\n- A skill defines behavioral guidance (decision trees, strategies) but no anti-pattern or completeness tests\n- A module documents forbidden behaviors but no test asserts their absence\n\n## Content Assertion Anti-Patterns\n\nAvoid these when reviewing content tests:\n\n| Anti-Pattern | Problem | Better Approach |\n|---|---|---|\n| Testing prose style | Brittle to rewording, overlaps with scribe:slop-detector | Test behavioral semantics |\n| Asserting exact wording | Breaks on any edit | Assert concepts (`\"version\" in content.lower()`) |\n| Checking line counts | Not behavioral | Check required sections exist |\n| Testing formatting | Not what Claude interprets | Test parseable structure |\n| Duplicating slop detection | Already handled by scribe | Focus on correctness, not style |\n\n## Review Checklist Addition\n\nAdd this item to the existing Test Quality Checklist when reviewing a plugin that has execution markdown:\n\n```markdown\n- [ ] Content assertion depth matches content complexity\n      (L1 for simple skills, L2+ for code examples, L3 for behavioral guidance)\n```\n\n## Remediation Guidance\n\nWhen content tests are missing or insufficient:\n\n1. **No content tests at all**: Generate L1 scaffolding using `sanctum:test-updates/modules/generation/content-test-templates.md`\n2. **L1 only, has code blocks**: Upgrade to L2 (add JSON/YAML parsing tests)\n3. **L2 only, has version gates**: Upgrade to L3 (add cross-reference validation)\n4. **L2 only, has behavioral guidance**: Upgrade to L3 (add anti-pattern and completeness tests)\n\nFile v1.9.14:modules/coverage-analysis.md\n\n---\nparent_skill: pensive:test-review\nname: coverage-analysis\ndescription: Coverage measurement and gap identification\ncategory: testing\ntags: [coverage, testing, gap-analysis]\nload_priority: 2\nestimated_tokens: 350\n---\n\n# Coverage Analysis\n\nMeasure test coverage and identify gaps.\n\n## Coverage Tools by Language\n\n### Rust\n```bash\n# Using tarpaulin\ncargo install cargo-tarpaulin\ncargo tarpaulin --out Html --output-dir coverage/\n\n# Using llvm-cov\ncargo install cargo-llvm-cov\ncargo llvm-cov --html\n```\n\n### Python\n```bash\n# Using pytest-cov\npytest --cov=src --cov-report=html --cov-report=term-missing\n\n# Using coverage.py\ncoverage run -m pytest\ncoverage html\ncoverage report --show-missing\n```\n\n### JavaScript/TypeScript\n```bash\n# Jest\nnpm test -- --coverage --coverageReporters=html text\n\n# Vitest\nvitest --coverage\n\n# Cypress (code coverage plugin)\ncypress run --env coverage=true\n```\n\n### Go\n```bash\n# Built-in coverage\ngo test -cover ./...\ngo test -coverprofile=coverage.out ./...\ngo tool cover -html=coverage.out\n\n# Detailed coverage\ngo test -covermode=count -coverprofile=coverage.out ./...\n```\n\n## Coverage Thresholds\n\n| Level | Coverage | Use Case |\n|-------|----------|----------|\n| Minimum | 60% | Legacy code, initial cleanup |\n| Standard | 80% | Normal development |\n| High | 90% | Critical systems, libraries |\n| detailed | 95%+ | Safety-critical, financial |\n\n## Gap Identification\n\n### Find impacted test files\n```bash\n# Tests affected by changes\ngit diff --name-only main...HEAD | rg 'tests|spec|feature'\n\n# Find related tests\ngit diff --name-only main...HEAD | while read file; do\n  basename \"$file\" .py | xargs -I {} find . \\\n    -not -path \"*/.venv/*\" -not -path \"*/__pycache__/*\" \\\n    -not -path \"*/node_modules/*\" -not -path \"*/.git/*\" \\\n    -name \"*test*{}*\"\ndone\n```\n\n### Identify uncovered code\n1. Run coverage tool with `--show-missing` flag\n2. Cross-reference with critical paths:\n   - Authentication/authorization\n   - Data validation\n   - Error handling\n   - API endpoints\n   - Database operations\n\n3. Map to requirements:\n   - Feature specifications\n   - User stories\n   - Bug reports\n   - Security requirements\n\n### Coverage Patterns\n\n**Critical paths** (should be 100%):\n- Security boundaries (auth, validation)\n- Data integrity operations\n- Error recovery logic\n- Public API surface\n\n**Lower priority** (can be <80%):\n- Internal helpers\n- Logging/debugging code\n- Trivial getters/setters\n- Deprecated code paths\n\n## Output Format\n\n```markdown\n## Coverage Analysis\n- **Overall**: 78%\n- **Critical paths**: 92%\n- **Changed files**: 85%\n\n### Gaps Identified\n1. **src/auth.py:45-60** - Token validation edge cases\n2. **src/api/routes.py:120-135** - Error handling for 400/500 codes\n3. **src/db/migrations.py** - Rollback scenarios untested\n\n### Test-to-Feature Mapping\n- Feature: User registration → `tests/test_registration.py` (95%)\n- Feature: Password reset → `tests/test_auth.py` (60%) [WARN]\n- Feature: Email validation → Missing tests [FAIL]\n```\n\n## Best Practices\n\n1. **Branch coverage** over line coverage when available\n2. **Mutation testing** for critical code (e.g., `cargo mutants`, `mutmut`)\n3. **Coverage trends**: Track over time, not just absolute values\n4. **Exclude generated code**: Focus on hand-written logic\n5. **Integration coverage**: Don't just unit test in isolation\n\nFile v1.9.14:modules/framework-detection.md\n\n---\nparent_skill: pensive:test-review\nname: framework-detection\ndescription: Language and test framework detection patterns\ncategory: testing\ntags: [testing, framework-detection, language-detection]\nload_priority: 1\nestimated_tokens: 250\n---\n\n# Framework Detection\n\nIdentify testing frameworks and tooling constraints.\n\n## Language Detection Patterns\n\n### Rust\n- **Framework**: cargo test (built-in)\n- **Commands**: `cargo test`, `cargo nextest run`\n- **Config files**: `Cargo.toml`, `Cargo.lock`\n- **Test patterns**: `#[test]`, `#[cfg(test)]`\n- **MSRV**: Check `rust-version` in Cargo.toml\n\n### Python\n- **Frameworks**: pytest, unittest, behave\n- **Commands**: `pytest`, `python -m pytest`, `behave`\n- **Config files**: `pytest.ini`, `pyproject.toml`, `tox.ini`\n- **Test patterns**: `test_*.py`, `*_test.py`, `tests/`\n- **Version**: Check `requires-python` in pyproject.toml\n\n### JavaScript/TypeScript\n- **Frameworks**: Jest, Mocha, Cypress, Vitest\n- **Commands**: `npm test`, `yarn test`, `cypress run`\n- **Config files**: `jest.config.js`, `vitest.config.ts`, `cypress.config.js`\n- **Test patterns**: `*.test.js`, `*.spec.ts`, `__tests__/`\n- **Version**: Check `engines.node` in package.json\n\n### Go\n- **Framework**: go test (built-in)\n- **Commands**: `go test ./...`, `go test -v`\n- **Config files**: `go.mod`, `go.sum`\n- **Test patterns**: `*_test.go`\n- **Version**: Check `go` directive in go.mod\n\n## Detection Workflow\n\n1. **Scan for config files**:\n```bash\nfind . -maxdepth 2 -name \"Cargo.toml\" -o -name \"pyproject.toml\" -o -name \"package.json\" -o -name \"go.mod\"\n```\n\n2. **Check test directories**:\n```bash\nfind . -type d -name \"tests\" -o -name \"__tests__\" -o -name \"test\"\n```\n\n3. **Identify test files**:\n```bash\nfind . -not -path \"*/.venv/*\" -not -path \"*/__pycache__/*\" \\\n  -not -path \"*/node_modules/*\" -not -path \"*/.git/*\" \\\n  \\( -name \"*test*\" -o -name \"*spec*\" \\) \\\n  | grep -E '\\.(rs|py|js|ts|go)$'\n```\n\n4. **Version constraints**:\n- Extract MSRV, Python version, Node version\n- Note if constraints affect tooling (e.g., async/await)\n- Document CI/CD version requirements\n\n## Output Format\n\n```markdown\n## Framework Detection\n- **Languages**: Rust, Python\n- **Frameworks**: cargo test, pytest\n- **Versions**:\n  - Rust MSRV: 1.70\n  - Python: >=3.8\n- **Config files**: Cargo.toml, pyproject.toml\n```\n\nFile v1.9.14:modules/remediation-planning.md\n\n---\nparent_skill: pensive:test-review\nname: remediation-planning\ndescription: Test improvement strategies and phased remediation\ncategory: testing\ntags: [remediation, test-improvement, refactoring]\nload_priority: 4\nestimated_tokens: 300\n---\n\n# Remediation Planning\n\nConcrete strategies for test improvement.\n\n## Test Improvement Patterns\n\n### 1. Add Missing Coverage\nTie tests to specific behaviors using Given/When/Then:\n\n```markdown\n### Gap: Authentication edge cases\n**Behavior**: Given missing auth token, When hitting /v1/resource, Then HTTP 401 returned\n**Test**: `tests/test_auth.py::test_missing_token_returns_401`\n**Priority**: High (security boundary)\n```\n\n### 2. Refactor Test Helpers\n\n**Before** (repeated setup):\n```python\ndef test_user_creation():\n    db = setup_database()\n    config = load_test_config()\n    user_data = {\"email\": \"alice@example.com\", \"role\": \"user\"}\n    ...\n\ndef test_user_deletion():\n    db = setup_database()\n    config = load_test_config()\n    user_data = {\"email\": \"bob@example.com\", \"role\": \"admin\"}\n    ...\n```\n\n**After** (fixtures):\n```python\n@pytest.fixture\ndef test_db():\n    db = setup_database()\n    yield db\n    db.teardown()\n\n@pytest.fixture\ndef test_config():\n    return load_test_config()\n\ndef test_user_creation(test_db, test_config):\n    user_data = user_factory(email=\"alice@example.com\")\n    ...\n```\n\n### 3. Data Builders and Factories\n\n**Factory pattern**:\n```python\n# conftest.py\ndef user_factory(**overrides):\n    defaults = {\n        \"email\": \"user@example.com\",\n        \"role\": \"user\",\n        \"verified\": True,\n        \"created_at\": datetime.now()\n    }\n    return User(**{**defaults, **overrides})\n\n# test file\ndef test_admin_access():\n    admin = user_factory(role=\"admin\")\n    assert admin.can_access_dashboard()\n```\n\n**Builder pattern** (Rust):\n```rust\nstruct UserBuilder {\n    email: String,\n    role: Role,\n    verified: bool,\n}\n\nimpl UserBuilder {\n    fn new() -> Self {\n        Self {\n            email: \"user@example.com\".to_string(),\n            role: Role::User,\n            verified: true,\n        }\n    }\n\n    fn with_role(mut self, role: Role) -> Self {\n        self.role = role;\n        self\n    }\n\n    fn build(self) -> User {\n        User { /* ... */ }\n    }\n}\n\n#[test]\nfn test_admin_permissions() {\n    let admin = UserBuilder::new().with_role(Role::Admin).build();\n    assert!(admin.can_delete_users());\n}\n```\n\n### 4. Improve Assertions\n\n**Replace magic values**:\n```python\n# Before\nassert response.status == 200\n\n# After\nfrom http import HTTPStatus\nassert response.status == HTTPStatus.OK\n```\n\n**Add context**:\n```python\n# Before\nassert result\n\n# After\nassert result.success, f\"Expected success, got error: {result.error}\"\n```\n\n### 5. Remove Brittle Patterns\n\n**Dead waits** → **Explicit conditions**:\n```python\n# Before\ntime.sleep(3)\nassert element.visible\n\n# After\nwait_for(element.to_be_visible, timeout=5)\n```\n\n**Mocking internals** → **Mock boundaries**:\n```python\n# Before: mocking private implementation\n@patch('service._internal_helper')\ndef test_service(mock):\n    ...\n\n# After: mock external dependency\n@patch('requests.post')\ndef test_service(mock_requests):\n    ...\n```\n\n## Phased Remediation\n\nFor major test suite rewrites:\n\n### Phase 1: Stabilize (Week 1-2)\n1. Fix flaky tests (eliminate dead waits, order dependencies)\n2. Remove duplicate tests\n3. Add missing critical path tests\n4. **Metric**: Flaky test rate < 1%, critical paths 100%\n\n### Phase 2: Acceptance Specs (Week 3-4)\n1. Add BDD scenarios for user-facing features\n2. Create feature-to-test mapping\n3. Document test strategy per component\n4. **Metric**: All features have acceptance tests\n\n### Phase 3: Enforce Quality (Week 5+)\n1. Set coverage budgets (80% standard, 100% critical)\n2. Add pre-commit hooks for coverage checks\n3. Integrate mutation testing for critical code\n4. **Metric**: Coverage trends upward, no regressions\n\n## Recommendation Template\n\n```markdown\n## Remediation Plan\n\n### Immediate Actions (This Sprint)\n1. **Fix flaky test**: `test_user_login_retries` - Replace sleep with explicit wait\n   - Owner: @alice\n   - Due: 2025-12-10\n\n2. **Add missing coverage**: Password reset flow (currently 0%)\n   - Tests needed: valid token, expired token, invalid token\n   - Owner: @bob\n   - Due: 2025-12-12\n\n### Short-term (Next Sprint)\n3. **Refactor fixtures**: Extract common setup in `tests/test_api.py`\n   - Pattern: Use pytest fixtures for DB, config\n   - Owner: @charlie\n   - Due: 2025-12-20\n\n### Long-term (Next Month)\n4. **BDD acceptance tests**: User registration feature\n   - Tool: Behave/Gherkin\n   - Owner: @diana\n   - Due: 2025-01-15\n```\n\n## Exit Criteria\n\n- [ ] All critical gaps have assigned owners and due dates\n- [ ] Recommendations tied to specific behaviors\n- [ ] Phased approach for large refactorings\n- [ ] Success metrics defined (coverage %, flaky rate, etc.)\n\nFile v1.9.14:modules/scenario-quality.md\n\n---\nparent_skill: pensive:test-review\nname: scenario-quality\ndescription: Test scenario quality assessment with BDD patterns\ncategory: testing\ntags: [bdd, scenario-quality, assertions, anti-patterns]\nload_priority: 3\nestimated_tokens: 350\n---\n\n# Scenario Quality Assessment\n\nEvaluate test quality using BDD principles and assertion patterns.\n\n## Given/When/Then Clarity\n\n### Good Examples\n\n**Rust:**\n```rust\n#[test]\nfn test_authenticated_user_can_access_profile() {\n    // Given: authenticated user\n    let user = create_authenticated_user(\"alice@example.com\");\n    let token = generate_token(&user);\n\n    // When: accessing profile endpoint\n    let response = get(\"/profile\", &token);\n\n    // Then: profile data returned\n    assert_eq!(response.status, 200);\n    assert_eq!(response.body[\"email\"], \"alice@example.com\");\n}\n```\n\n**Python:**\n```python\ndef test_invalid_credentials_rejected():\n    # Given: user with wrong password\n    user = User(email=\"bob@example.com\")\n    wrong_password = \"incorrect\"\n\n    # When: attempting authentication\n    result = authenticate(user.email, wrong_password)\n\n    # Then: authentication fails with 401\n    assert result.status_code == 401\n    assert \"invalid credentials\" in result.error_message\n```\n\n**Gherkin (BDD):**\n```gherkin\nScenario: Registered user logs in successfully\n  Given a registered user with email \"alice@example.com\"\n  When they submit valid credentials\n  Then they receive an authentication token\n  And the token expires in 24 hours\n```\n\n## Assertion Quality\n\n### Bad Assertions (vague, brittle)\n```python\n# Too vague\nassert result\n\n# Multiple unrelated assertions\nassert len(users) > 0 and users[0].active and config.debug\n\n# Magic numbers without context\nassert response.status == 200\n```\n\n### Good Assertions (specific, meaningful)\n```python\n# Specific outcome\nassert result.status_code == 200, \"Expected successful login\"\n\n# Named constants\nassert response.status == HTTP_OK\nassert user.role == UserRole.ADMIN\n\n# Structured assertions\nassert response.json() == {\n    \"user\": {\"email\": expected_email, \"verified\": True},\n    \"token\": {\"expires_at\": ANY_DATETIME}\n}\n```\n\n## Anti-Patterns to Flag\n\n### 1. Dead Waits\n```python\n# BAD: arbitrary sleep\ntime.sleep(5)\nassert element.is_visible()\n\n# GOOD: explicit wait with condition\nwait_until(lambda: element.is_visible(), timeout=5)\n```\n\n### 2. Mocking Internals\n```python\n# BAD: mocking implementation details\n@patch('module.internal._private_helper')\ndef test_feature(mock_helper):\n    ...\n\n# GOOD: mock external dependencies only\n@patch('requests.get')\ndef test_api_call(mock_get):\n    ...\n```\n\n### 3. Repeated Boilerplate\n```python\n# BAD: copy-pasted setup\ndef test_user_creation():\n    db = Database(\"test.db\")\n    db.connect()\n    user = User(\"alice\")\n    ...\n\ndef test_user_deletion():\n    db = Database(\"test.db\")\n    db.connect()\n    user = User(\"bob\")\n    ...\n\n# GOOD: fixture/helper\n@pytest.fixture\ndef db_session():\n    db = Database(\"test.db\")\n    db.connect()\n    yield db\n    db.close()\n```\n\n### 4. Order Dependencies\n```python\n# BAD: tests depend on execution order\ndef test_01_create_user():\n    global user_id\n    user_id = create_user()\n\ndef test_02_delete_user():\n    delete_user(user_id)  # Depends on test_01!\n\n# GOOD: isolated tests\ndef test_delete_user():\n    user_id = create_user()  # Self-contained\n    delete_user(user_id)\n    assert not user_exists(user_id)\n```\n\n### 5. Multiple Assertions Without Context\n```python\n# BAD: unclear which assertion failed\nassert user.active\nassert user.verified\nassert user.role == \"admin\"\n\n# GOOD: grouped with context or separate tests\nassert user.active, \"User should be active\"\nassert user.verified, \"User should be verified\"\nassert user.role == \"admin\", \"User should have admin role\"\n```\n\n## BDD Suite Quality\n\n### Reusable Step Definitions\n```python\n# Good: parameterized, reusable\n@given('a user with email \"{email}\"')\ndef create_user(context, email):\n    context.user = User(email=email)\n\n@when('they submit credentials with password \"{password}\"')\ndef submit_credentials(context, password):\n    context.response = authenticate(context.user.email, password)\n```\n\n### Background Context Sharing\n```gherkin\nFeature: User authentication\n\n  Background:\n    Given a clean database\n    And the authentication service is running\n\n  Scenario: Valid login\n    Given a registered user\n    ...\n```\n\n### Scenario Outlines for Edge Cases\n```gherkin\nScenario Outline: Password validation\n  Given a user registering with password \"<password>\"\n  When they submit the registration form\n  Then they receive response \"<outcome>\"\n\n  Examples:\n    | password    | outcome           |\n    | abc         | too_short         |\n    | password123 | no_special_chars  |\n    | P@ssw0rd!   | success           |\n```\n\n## Quality Scoring\n\nScore each test file 1-5 on:\n- **Clarity**: Given/When/Then structure evident\n- **Assertions**: Specific, meaningful checks\n- **Isolation**: No shared state or order dependencies\n- **Maintainability**: DRY, uses fixtures/helpers\n- **Coverage**: Tests behavior, not implementation\n\n**Overall quality**:\n- 4-5: Excellent, minimal changes needed\n- 3: Good, some improvements recommended\n- 1-2: Poor, significant refactoring required\n\nFile v1.9.14:skill-card.md\n\n## Description: <br>\nEvaluates test suites for coverage gaps, TDD/BDD compliance, and anti-patterns. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[athola](https://clawhub.ai/user/athola) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and engineering teams use this skill to review existing test suites for framework coverage, BDD/TDD quality, invariant preservation, and remediation planning before releases or after failures. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: Broad activation can make the review workflow run when a narrower invocation would be preferable. <br>\nMitigation: Install and invoke the skill only when an agent-accessible code review helper is needed, and review provider settings before use. <br>\n\n\n## Reference(s): <br>\n- [ClawHub skill page](https://clawhub.ai/athola/skills/nm-pensive-test-review) <br>\n- [Pensive homepage](https://github.com/athola/claude-night-market/tree/master/plugins/pensive) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [text, markdown, shell commands, guidance] <br>\n**Output Format:** [Markdown review report with shell command examples] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Produces framework detection, coverage analysis, quality issues, remediation planning, and an approve, approve-with-actions, or block recommendation.] <br>\n\n## Skill Version(s): <br>\n1.9.14 (source: ClawHub release metadata; artifact frontmatter states 1.9.8) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nArchive v1.9.13: 8 files, 13688 bytes\n\nFiles: modules/content-assertion-quality.md (2534b), modules/coverage-analysis.md (3330b), modules/framework-detection.md (2315b), modules/remediation-planning.md (4849b), modules/scenario-quality.md (5212b), skill-card.md (2277b), SKILL.md (8033b), _meta.json (142b)\n\nFile v1.9.13:SKILL.md\n\n---\nname: test-review\ndescription: Evaluates test suites for coverage gaps, TDD/BDD compliance, and anti-patterns\nversion: 1.9.8\ntriggers:\n  - testing\n  - tdd\n  - bdd\n  - coverage\n  - quality\n  - fixtures\n  - auditing test quality or before a major release\nmetadata: {\"openclaw\": {\"homepage\": \"https://github.com/athola/claude-night-market/tree/master/plugins/pensive\", \"emoji\": \"\\ud83e\\uddea\", \"requires\": {\"config\": [\"night-market.pensive:shared\", \"night-market.imbue:proof-of-work\"]}}}\nsource: claude-night-market\nsource_plugin: pensive\n---\n\n> **Night Market Skill** — ported from [claude-night-market/pensive](https://github.com/athola/claude-night-market/tree/master/plugins/pensive). For the full experience with agents, hooks, and commands, install the Claude Code plugin.\n\n\n## Table of Contents\n\n- [Quick Start](#quick-start)\n- [When to Use](#when-to-use)\n- [Required TodoWrite Items](#required-todowrite-items)\n- [Progressive Loading](#progressive-loading)\n- [Workflow](#workflow)\n- [Step 1: Detect Languages (`test-review:languages-detected`)](#step-1:-detect-languages-(test-review:languages-detected))\n- [Step 2: Inventory Coverage (`test-review:coverage-inventoried`)](#step-2:-inventory-coverage-(test-review:coverage-inventoried))\n- [Step 3: Assess Scenario Quality (`test-review:scenario-quality`)](#step-3:-assess-scenario-quality-(test-review:scenario-quality))\n- [Step 4: Plan Remediation (`test-review:gap-remediation`)](#step-4:-plan-remediation-(test-review:gap-remediation))\n- [Step 5: Log Evidence (`test-review:evidence-logged`)](#step-5:-log-evidence-(test-review:evidence-logged))\n- [Test Quality Checklist (Condensed)](#test-quality-checklist-(condensed))\n- [Output Format](#output-format)\n- [Summary](#summary)\n- [Framework Detection](#framework-detection)\n- [Coverage Analysis](#coverage-analysis)\n- [Quality Issues](#quality-issues)\n- [Remediation Plan](#remediation-plan)\n- [Recommendation](#recommendation)\n- [Integration Notes](#integration-notes)\n- [Exit Criteria](#exit-criteria)\n\n\n# Test Review Workflow\n\nEvaluate and improve test suites with TDD/BDD rigor.\n\n## Quick Start\n\n```bash\n/test-review\n```\n**Verification:** Run `pytest -v` to verify tests pass.\n\n## When To Use\n\n- Reviewing test suite quality\n- Analyzing coverage gaps\n- Before major releases\n- After test failures\n- Planning test improvements\n\n## When NOT To Use\n\n- Writing new tests - use parseltongue:python-testing\n- Updating existing tests - use sanctum:test-updates\n\n## Required TodoWrite Items\n\n1. `test-review:languages-detected`\n2. `test-review:coverage-inventoried`\n3. `test-review:scenario-quality`\n4. `test-review:invariant-preservation`\n5. `test-review:gap-remediation`\n6. `test-review:evidence-logged`\n\n## Progressive Loading\n\nLoad modules as needed based on review depth:\n\n- **Basic review**: Core workflow (this file)\n- **Framework detection**: Load `modules/framework-detection.md`\n- **Coverage analysis**: Load `modules/coverage-analysis.md`\n- **Quality assessment**: Load `modules/scenario-quality.md`\n- **Remediation planning**: Load `modules/remediation-planning.md`\n\n## Workflow\n\n### Step 1: Detect Languages (`test-review:languages-detected`)\n\nIdentify testing frameworks and version constraints.\n→ **See**: `modules/framework-detection.md`\n\nQuick check:\n```bash\nfind . -maxdepth 2 -name \"Cargo.toml\" -o -name \"pyproject.toml\" -o -name \"package.json\" -o -name \"go.mod\"\n```\n**Verification:** Run the command with `--help` flag to verify availability.\n\n### Step 2: Inventory Coverage (`test-review:coverage-inventoried`)\n\nRun coverage tools and identify gaps.\n→ **See**: `modules/coverage-analysis.md`\n\nQuick check:\n```bash\ngit diff --name-only | rg 'tests|spec|feature'\n```\n**Verification:** Run `pytest -v` to verify tests pass.\n\n### Step 3: Assess Scenario Quality (`test-review:scenario-quality`)\n\nEvaluate test quality using BDD patterns and assertion checks.\n→ **See**: `modules/scenario-quality.md`\n\nFocus on:\n- Given/When/Then clarity\n- Assertion specificity\n- Anti-patterns (dead waits, mocking internals, repeated boilerplate)\n\n### Step 4: Plan Remediation (`test-review:gap-remediation`)\n\nCreate concrete improvement plan with owners and dates.\n→ **See**: `modules/remediation-planning.md`\n\n### Step 5: Log Evidence (`test-review:evidence-logged`)\n\nRecord executed commands, outputs, and recommendations.\n→ **See**: `imbue:proof-of-work`\n\n## Test Quality Checklist (Condensed)\n\n- [ ] Clear test structure (Arrange-Act-Assert)\n- [ ] Critical paths covered (auth, validation, errors)\n- [ ] Specific assertions with context\n- [ ] No flaky tests (dead waits, order dependencies)\n- [ ] Reusable fixtures/factories\n- [ ] Invariant-encoding tests intact (see below)\n\n### Invariant-Encoding Tests\n\nTests do not just verify behavior — they encode design\ninvariants. A test that asserts \"module A never imports\nfrom module B\" encodes a layer boundary. A test that\nasserts \"this function is pure\" encodes a concurrency\nmodel. These tests are load-bearing in ways that\ncoverage metrics cannot capture.\n\n**During review, check:**\n\n1. **Were invariant-encoding tests removed or weakened?**\n   A test that enforced an architectural boundary,\n   data structure constraint, or API contract should\n   not be deleted without naming the invariant being\n   abandoned and escalating to human judgment.\n\n2. **Were test expectations changed to match a broken\n   implementation?** If an assertion value changed, ask:\n   did the *requirement* change, or did the agent change\n   the test to make its code pass? The latter is the\n   single most dangerous form of test tampering.\n\n3. **Are new invariants encoded as tests?** When a design\n   decision is made (choice of data structure, module\n   boundary, error strategy), there should be at least\n   one test whose failure would signal that the\n   invariant was violated.\n\n**Red flag patterns:**\n\n| Pattern | Risk |\n|---------|------|\n| `@pytest.mark.skip` added to a passing test | Invariant being silently dropped |\n| Assertion changed from specific to broad | Constraint being relaxed |\n| Test renamed to describe new behavior | Old invariant erased from history |\n| Test deleted \"because it tested old code\" | Invariant removed without replacement |\n\n**When invariant erosion is detected:**\n\nDo NOT approve. Flag as a BLOCKING quality issue and\npresent the three options to the human:\n\n1. **Preserve**: Revert the test change, fix the\n   implementation to satisfy the invariant\n2. **Layer**: Keep the invariant test, add the new\n   behavior alongside it (accepting inelegance)\n3. **Revise**: The invariant is genuinely wrong — remove\n   the old test AND write a new test encoding the\n   replacement invariant\n\nThis is a judgment call that models get wrong far too\noften. Default to option 1 (preserve) when no human is\navailable.\n\n## Output Format\n\n```markdown\n## Summary\n[Brief assessment]\n\n## Framework Detection\n- Languages: [list] | Frameworks: [list] | Versions: [constraints]\n\n## Coverage Analysis\n- Overall: X% | Critical: X% | Gaps: [list]\n\n## Quality Issues\n[Q1] [Issue] - Location - Fix\n\n## Remediation Plan\n1. [Action] - Owner - Date\n\n## Recommendation\nApprove / Approve with actions / Block\n```\n**Verification:** Run the command with `--help` flag to verify availability.\n\n## Integration Notes\n\n- Use `imbue:proof-of-work` for reproducible evidence capture\n- Reference `imbue:diff-analysis` for risk assessment\n- Format output using `imbue:structured-output` patterns\n\n## Exit Criteria\n\n- Frameworks detected and documented\n- Coverage analyzed and gaps identified\n- Scenario quality assessed\n- Remediation plan created with owners and dates\n- Evidence logged with citations\n## Troubleshooting\n\n### Common Issues\n\n**Tests not discovered**\nEnsure test files match pattern `test_*.py` or `*_test.py`. Run `pytest --collect-only` to verify.\n\n**Import errors**\nCheck that the module being tested is in `PYTHONPATH` or install with `pip install -e .`\n\n**Async tests failing**\nInstall pytest-asyncio and decorate test functions with `@pytest.mark.asyncio`\n\nFile v1.9.13:_meta.json\n\n{\n  \"ownerId\": \"kn7d107jg9jv602h9ytsegydq184a42s\",\n  \"slug\": \"nm-pensive-test-review\",\n  \"version\": \"1.9.13\",\n  \"publishedAt\": 1782577357642\n}\n\nFile v1.9.13:modules/content-assertion-quality.md\n\n# Content Assertion Quality\n\nScoring criteria for evaluating content assertion tests during test review. Extends the scenario quality assessment with a Content Depth dimension.\n\nReference: `leyline:testing-quality-standards/modules/content-assertion-levels.md`\n\n## Content Depth Scoring\n\nRate content assertion depth on a 1-5 scale:\n\n| Score | Level | Description |\n|---|---|---|\n| 1 | None | Tests only file existence or line count |\n| 2 | L1 | Keyword presence checks (`assert \"section\" in content`) |\n| 3 | L2 | Parses embedded examples, validates schema structure |\n| 4 | L3 | Cross-references, anti-patterns, decision framework contracts |\n| 5 | L3+ | Cross-plugin validation (version refs checked against other plugins' docs) |\n\n## When to Flag Missing Content Assertions\n\nDuring test review, flag as a content test gap when:\n\n- A skill has tests but all are L1 (keyword-only) and the skill contains JSON or YAML code blocks\n- A skill has version-gated features but no cross-reference validation\n- A skill defines behavioral guidance (decision trees, strategies) but no anti-pattern or completeness tests\n- A module documents forbidden behaviors but no test asserts their absence\n\n## Content Assertion Anti-Patterns\n\nAvoid these when reviewing content tests:\n\n| Anti-Pattern | Problem | Better Approach |\n|---|---|---|\n| Testing prose style | Brittle to rewording, overlaps with scribe:slop-detector | Test behavioral semantics |\n| Asserting exact wording | Breaks on any edit | Assert concepts (`\"version\" in content.lower()`) |\n| Checking line counts | Not behavioral | Check required sections exist |\n| Testing formatting | Not what Claude interprets | Test parseable structure |\n| Duplicating slop detection | Already handled by scribe | Focus on correctness, not style |\n\n## Review Checklist Addition\n\nAdd this item to the existing Test Quality Checklist when reviewing a plugin that has execution markdown:\n\n```markdown\n- [ ] Content assertion depth matches content complexity\n      (L1 for simple skills, L2+ for code examples, L3 for behavioral guidance)\n```\n\n## Remediation Guidance\n\nWhen content tests are missing or insufficient:\n\n1. **No content tests at all**: Generate L1 scaffolding using `sanctum:test-updates/modules/generation/content-test-templates.md`\n2. **L1 only, has code blocks**: Upgrade to L2 (add JSON/YAML parsing tests)\n3. **L2 only, has version gates**: Upgrade to L3 (add cross-reference validation)\n4. **L2 only, has behavioral guidance**: Upgrade to L3 (add anti-pattern and completeness tests)\n\nFile v1.9.13:modules/coverage-analysis.md\n\n---\nparent_skill: pensive:test-review\nname: coverage-analysis\ndescription: Coverage measurement and gap identification\ncategory: testing\ntags: [coverage, testing, gap-analysis]\nload_priority: 2\nestimated_tokens: 350\n---\n\n# Coverage Analysis\n\nMeasure test coverage and identify gaps.\n\n## Coverage Tools by Language\n\n### Rust\n```bash\n# Using tarpaulin\ncargo install cargo-tarpaulin\ncargo tarpaulin --out Html --output-dir coverage/\n\n# Using llvm-cov\ncargo install cargo-llvm-cov\ncargo llvm-cov --html\n```\n\n### Python\n```bash\n# Using pytest-cov\npytest --cov=src --cov-report=html --cov-report=term-missing\n\n# Using coverage.py\ncoverage run -m pytest\ncoverage html\ncoverage report --show-missing\n```\n\n### JavaScript/TypeScript\n```bash\n# Jest\nnpm test -- --coverage --coverageReporters=html text\n\n# Vitest\nvitest --coverage\n\n# Cypress (code coverage plugin)\ncypress run --env coverage=true\n```\n\n### Go\n```bash\n# Built-in coverage\ngo test -cover ./...\ngo test -coverprofile=coverage.out ./...\ngo tool cover -html=coverage.out\n\n# Detailed coverage\ngo test -covermode=count -coverprofile=coverage.out ./...\n```\n\n## Coverage Thresholds\n\n| Level | Coverage | Use Case |\n|-------|----------|----------|\n| Minimum | 60% | Legacy code, initial cleanup |\n| Standard | 80% | Normal development |\n| High | 90% | Critical systems, libraries |\n| detailed | 95%+ | Safety-critical, financial |\n\n## Gap Identification\n\n### Find impacted test files\n```bash\n# Tests affected by changes\ngit diff --name-only main...HEAD | rg 'tests|spec|feature'\n\n# Find related tests\ngit diff --name-only main...HEAD | while read file; do\n  basename \"$file\" .py | xargs -I {} find . \\\n    -not -path \"*/.venv/*\" -not -path \"*/__pycache__/*\" \\\n    -not -path \"*/node_modules/*\" -not -path \"*/.git/*\" \\\n    -name \"*test*{}*\"\ndone\n```\n\n### Identify uncovered code\n1. Run coverage tool with `--show-missing` flag\n2. Cross-reference with critical paths:\n   - Authentication/authorization\n   - Data validation\n   - Error handling\n   - API endpoints\n   - Database operations\n\n3. Map to requirements:\n   - Feature specifications\n   - User stories\n   - Bug reports\n   - Security requirements\n\n### Coverage Patterns\n\n**Critical paths** (should be 100%):\n- Security boundaries (auth, validation)\n- Data integrity operations\n- Error recovery logic\n- Public API surface\n\n**Lower priority** (can be <80%):\n- Internal helpers\n- Logging/debugging code\n- Trivial getters/setters\n- Deprecated code paths\n\n## Output Format\n\n```markdown\n## Coverage Analysis\n- **Overall**: 78%\n- **Critical paths**: 92%\n- **Changed files**: 85%\n\n### Gaps Identified\n1. **src/auth.py:45-60** - Token validation edge cases\n2. **src/api/routes.py:120-135** - Error handling for 400/500 codes\n3. **src/db/migrations.py** - Rollback scenarios untested\n\n### Test-to-Feature Mapping\n- Feature: User registration → `tests/test_registration.py` (95%)\n- Feature: Password reset → `tests/test_auth.py` (60%) [WARN]\n- Feature: Email validation → Missing tests [FAIL]\n```\n\n## Best Practices\n\n1. **Branch coverage** over line coverage when available\n2. **Mutation testing** for critical code (e.g., `cargo mutants`, `mutmut`)\n3. **Coverage trends**: Track over time, not just absolute values\n4. **Exclude generated code**: Focus on hand-written logic\n5. **Integration coverage**: Don't just unit test in isolation\n\nFile v1.9.13:modules/framework-detection.md\n\n---\nparent_skill: pensive:test-review\nname: framework-detection\ndescription: Language and test framework detection patterns\ncategory: testing\ntags: [testing, framework-detection, language-detection]\nload_priority: 1\nestimated_tokens: 250\n---\n\n# Framework Detection\n\nIdentify testing frameworks and tooling constraints.\n\n## Language Detection Patterns\n\n### Rust\n- **Framework**: cargo test (built-in)\n- **Commands**: `cargo test`, `cargo nextest run`\n- **Config files**: `Cargo.toml`, `Cargo.lock`\n- **Test patterns**: `#[test]`, `#[cfg(test)]`\n- **MSRV**: Check `rust-version` in Cargo.toml\n\n### Python\n- **Frameworks**: pytest, unittest, behave\n- **Commands**: `pytest`, `python -m pytest`, `behave`\n- **Config files**: `pytest.ini`, `pyproject.toml`, `tox.ini`\n- **Test patterns**: `test_*.py`, `*_test.py`, `tests/`\n- **Version**: Check `requires-python` in pyproject.toml\n\n### JavaScript/TypeScript\n- **Frameworks**: Jest, Mocha, Cypress, Vitest\n- **Commands**: `npm test`, `yarn test`, `cypress run`\n- **Config files**: `jest.config.js`, `vitest.config.ts`, `cypress.config.js`\n- **Test patterns**: `*.test.js`, `*.spec.ts`, `__tests__/`\n- **Version**: Check `engines.node` in package.json\n\n### Go\n- **Framework**: go test (built-in)\n- **Commands**: `go test ./...`, `go test -v`\n- **Config files**: `go.mod`, `go.sum`\n- **Test patterns**: `*_test.go`\n- **Version**: Check `go` directive in go.mod\n\n## Detection Workflow\n\n1. **Scan for config files**:\n```bash\nfind . -maxdepth 2 -name \"Cargo.toml\" -o -name \"pyproject.toml\" -o -name \"package.json\" -o -name \"go.mod\"\n```\n\n2. **Check test directories**:\n```bash\nfind . -type d -name \"tests\" -o -name \"__tests__\" -o -name \"test\"\n```\n\n3. **Identify test files**:\n```bash\nfind . -not -path \"*/.venv/*\" -not -path \"*/__pycache__/*\" \\\n  -not -path \"*/node_modules/*\" -not -path \"*/.git/*\" \\\n  \\( -name \"*test*\" -o -name \"*spec*\" \\) \\\n  | grep -E '\\.(rs|py|js|ts|go)$'\n```\n\n4. **Version constraints**:\n- Extract MSRV, Python version, Node version\n- Note if constraints affect tooling (e.g., async/await)\n- Document CI/CD version requirements\n\n## Output Format\n\n```markdown\n## Framework Detection\n- **Languages**: Rust, Python\n- **Frameworks**: cargo test, pytest\n- **Versions**:\n  - Rust MSRV: 1.70\n  - Python: >=3.8\n- **Config files**: Cargo.toml, pyproject.toml\n```\n\nFile v1.9.13:modules/remediation-planning.md\n\n---\nparent_skill: pensive:test-review\nname: remediation-planning\ndescription: Test improvement strategies and phased remediation\ncategory: testing\ntags: [remediation, test-improvement, refactoring]\nload_priority: 4\nestimated_tokens: 300\n---\n\n# Remediation Planning\n\nConcrete strategies for test improvement.\n\n## Test Improvement Patterns\n\n### 1. Add Missing Coverage\nTie tests to specific behaviors using Given/When/Then:\n\n```markdown\n### Gap: Authentication edge cases\n**Behavior**: Given missing auth token, When hitting /v1/resource, Then HTTP 401 returned\n**Test**: `tests/test_auth.py::test_missing_token_returns_401`\n**Priority**: High (security boundary)\n```\n\n### 2. Refactor Test Helpers\n\n**Before** (repeated setup):\n```python\ndef test_user_creation():\n    db = setup_database()\n    config = load_test_config()\n    user_data = {\"email\": \"alice@example.com\", \"role\": \"user\"}\n    ...\n\ndef test_user_deletion():\n    db = setup_database()\n    config = load_test_config()\n    user_data = {\"email\": \"bob@example.com\", \"role\": \"admin\"}\n    ...\n```\n\n**After** (fixtures):\n```python\n@pytest.fixture\ndef test_db():\n    db = setup_database()\n    yield db\n    db.teardown()\n\n@pytest.fixture\ndef test_config():\n    return load_test_config()\n\ndef test_user_creation(test_db, test_config):\n    user_data = user_factory(email=\"alice@example.com\")\n    ...\n```\n\n### 3. Data Builders and Factories\n\n**Factory pattern**:\n```python\n# conftest.py\ndef user_factory(**overrides):\n    defaults = {\n        \"email\": \"user@example.com\",\n        \"role\": \"user\",\n        \"verified\": True,\n        \"created_at\": datetime.now()\n    }\n    return User(**{**defaults, **overrides})\n\n# test file\ndef test_admin_access():\n    admin = user_factory(role=\"admin\")\n    assert admin.can_access_dashboard()\n```\n\n**Builder pattern** (Rust):\n```rust\nstruct UserBuilder {\n    email: String,\n    role: Role,\n    verified: bool,\n}\n\nimpl UserBuilder {\n    fn new() -> Self {\n        Self {\n            email: \"user@example.com\".to_string(),\n            role: Role::User,\n            verified: true,\n        }\n    }\n\n    fn with_role(mut self, role: Role) -> Self {\n        self.role = role;\n        self\n    }\n\n    fn build(self) -> User {\n        User { /* ... */ }\n    }\n}\n\n#[test]\nfn test_admin_permissions() {\n    let admin = UserBuilder::new().with_role(Role::Admin).build();\n    assert!(admin.can_delete_users());\n}\n```\n\n### 4. Improve Assertions\n\n**Replace magic values**:\n```python\n# Before\nassert response.status == 200\n\n# After\nfrom http import HTTPStatus\nassert response.status == HTTPStatus.OK\n```\n\n**Add context**:\n```python\n# Before\nassert result\n\n# After\nassert result.success, f\"Expected success, got error: {result.error}\"\n```\n\n### 5. Remove Brittle Patterns\n\n**Dead waits** → **Explicit conditions**:\n```python\n# Before\ntime.sleep(3)\nassert element.visible\n\n# After\nwait_for(element.to_be_visible, timeout=5)\n```\n\n**Mocking internals** → **Mock boundaries**:\n```python\n# Before: mocking private implementation\n@patch('service._internal_helper')\ndef test_service(mock):\n    ...\n\n# After: mock external dependency\n@patch('requests.post')\ndef test_service(mock_requests):\n    ...\n```\n\n## Phased Remediation\n\nFor major test suite rewrites:\n\n### Phase 1: Stabilize (Week 1-2)\n1. Fix flaky tests (eliminate dead waits, order dependencies)\n2. Remove duplicate tests\n3. Add missing critical path tests\n4. **Metric**: Flaky test rate < 1%, critical paths 100%\n\n### Phase 2: Acceptance Specs (Week 3-4)\n1. Add BDD scenarios for user-facing features\n2. Create feature-to-test mapping\n3. Document test strategy per component\n4. **Metric**: All features have acceptance tests\n\n### Phase 3: Enforce Quality (Week 5+)\n1. Set coverage budgets (80% standard, 100% critical)\n2. Add pre-commit hooks for coverage checks\n3. Integrate mutation testing for critical code\n4. **Metric**: Coverage trends upward, no regressions\n\n## Recommendation Template\n\n```markdown\n## Remediation Plan\n\n### Immediate Actions (This Sprint)\n1. **Fix flaky test**: `test_user_login_retries` - Replace sleep with explicit wait\n   - Owner: @alice\n   - Due: 2025-12-10\n\n2. **Add missing coverage**: Password reset flow (currently 0%)\n   - Tests needed: valid token, expired token, invalid token\n   - Owner: @bob\n   - Due: 2025-12-12\n\n### Short-term (Next Sprint)\n3. **Refactor fixtures**: Extract common setup in `tests/test_api.py`\n   - Pattern: Use pytest fixtures for DB, config\n   - Owner: @charlie\n   - Due: 2025-12-20\n\n### Long-term (Next Month)\n4. **BDD acceptance tests**: User registration feature\n   - Tool: Behave/Gherkin\n   - Owner: @diana\n   - Due: 2025-01-15\n```\n\n## Exit Criteria\n\n- [ ] All critical gaps have assigned owners and due dates\n- [ ] Recommendations tied to specific behaviors\n- [ ] Phased approach for large refactorings\n- [ ] Success metrics defined (coverage %, flaky rate, etc.)\n\nFile v1.9.13:modules/scenario-quality.md\n\n---\nparent_skill: pensive:test-review\nname: scenario-quality\ndescription: Test scenario quality assessment with BDD patterns\ncategory: testing\ntags: [bdd, scenario-quality, assertions, anti-patterns]\nload_priority: 3\nestimated_tokens: 350\n---\n\n# Scenario Quality Assessment\n\nEvaluate test quality using BDD principles and assertion patterns.\n\n## Given/When/Then Clarity\n\n### Good Examples\n\n**Rust:**\n```rust\n#[test]\nfn test_authenticated_user_can_access_profile() {\n    // Given: authenticated user\n    let user = create_authenticated_user(\"alice@example.com\");\n    let token = generate_token(&user);\n\n    // When: accessing profile endpoint\n    let response = get(\"/profile\", &token);\n\n    // Then: profile data returned\n    assert_eq!(response.status, 200);\n    assert_eq!(response.body[\"email\"], \"alice@example.com\");\n}\n```\n\n**Python:**\n```python\ndef test_invalid_credentials_rejected():\n    # Given: user with wrong password\n    user = User(email=\"bob@example.com\")\n    wrong_password = \"incorrect\"\n\n    # When: attempting authentication\n    result = authenticate(user.email, wrong_password)\n\n    # Then: authentication fails with 401\n    assert result.status_code == 401\n    assert \"invalid credentials\" in result.error_message\n```\n\n**Gherkin (BDD):**\n```gherkin\nScenario: Registered user logs in successfully\n  Given a registered user with email \"alice@example.com\"\n  When they submit valid credentials\n  Then they receive an authentication token\n  And the token expires in 24 hours\n```\n\n## Assertion Quality\n\n### Bad Assertions (vague, brittle)\n```python\n# Too vague\nassert result\n\n# Multiple unrelated assertions\nassert len(users) > 0 and users[0].active and config.debug\n\n# Magic numbers without context\nassert response.status == 200\n```\n\n### Good Assertions (specific, meaningful)\n```python\n# Specific outcome\nassert result.status_code == 200, \"Expected successful login\"\n\n# Named constants\nassert response.status == HTTP_OK\nassert user.role == UserRole.ADMIN\n\n# Structured assertions\nassert response.json() == {\n    \"user\": {\"email\": expected_email, \"verified\": True},\n    \"token\": {\"expires_at\": ANY_DATETIME}\n}\n```\n\n## Anti-Patterns to Flag\n\n### 1. Dead Waits\n```python\n# BAD: arbitrary sleep\ntime.sleep(5)\nassert element.is_visible()\n\n# GOOD: explicit wait with condition\nwait_until(lambda: element.is_visible(), timeout=5)\n```\n\n### 2. Mocking Internals\n```python\n# BAD: mocking implementation details\n@patch('module.internal._private_helper')\ndef test_feature(mock_helper):\n    ...\n\n# GOOD: mock external dependencies only\n@patch('requests.get')\ndef test_api_call(mock_get):\n    ...\n```\n\n### 3. Repeated Boilerplate\n```python\n# BAD: copy-pasted setup\ndef test_user_creation():\n    db = Database(\"test.db\")\n    db.connect()\n    user = User(\"alice\")\n    ...\n\ndef test_user_deletion():\n    db = Database(\"test.db\")\n    db.connect()\n    user = User(\"bob\")\n    ...\n\n# GOOD: fixture/helper\n@pytest.fixture\ndef db_session():\n    db = Database(\"test.db\")\n    db.connect()\n    yield db\n    db.close()\n```\n\n### 4. Order Dependencies\n```python\n# BAD: tests depend on execution order\ndef test_01_create_user():\n    global user_id\n    user_id = create_user()\n\ndef test_02_delete_user():\n    delete_user(user_id)  # Depends on test_01!\n\n# GOOD: isolated tests\ndef test_delete_user():\n    user_id = create_user()  # Self-contained\n    delete_user(user_id)\n    assert not user_exists(user_id)\n```\n\n### 5. Multiple Assertions Without Context\n```python\n# BAD: unclear which assertion failed\nassert user.active\nassert user.verified\nassert user.role == \"admin\"\n\n# GOOD: grouped with context or separate tests\nassert user.active, \"User should be active\"\nassert user.verified, \"User should be verified\"\nassert user.role == \"admin\", \"User should have admin role\"\n```\n\n## BDD Suite Quality\n\n### Reusable Step Definitions\n```python\n# Good: parameterized, reusable\n@given('a user with email \"{email}\"')\ndef create_user(context, email):\n    context.user = User(email=email)\n\n@when('they submit credentials with password \"{password}\"')\ndef submit_credentials(context, password):\n    context.response = authenticate(context.user.email, password)\n```\n\n### Background Context Sharing\n```gherkin\nFeature: User authentication\n\n  Background:\n    Given a clean database\n    And the authentication service is running\n\n  Scenario: Valid login\n    Given a registered user\n    ...\n```\n\n### Scenario Outlines for Edge Cases\n```gherkin\nScenario Outline: Password validation\n  Given a user registering with password \"<password>\"\n  When they submit the registration form\n  Then they receive response \"<outcome>\"\n\n  Examples:\n    | password    | outcome           |\n    | abc         | too_short         |\n    | password123 | no_special_chars  |\n    | P@ssw0rd!   | success           |\n```\n\n## Quality Scoring\n\nScore each test file 1-5 on:\n- **Clarity**: Given/When/Then structure evident\n- **Assertions**: Specific, meaningful checks\n- **Isolation**: No shared state or order dependencies\n- **Maintainability**: DRY, uses fixtures/helpers\n- **Coverage**: Tests behavior, not implementation\n\n**Overall quality**:\n- 4-5: Excellent, minimal changes needed\n- 3: Good, some improvements recommended\n- 1-2: Poor, significant refactoring required\n\nFile v1.9.13:skill-card.md\n\n## Description: <br>\nEvaluates test suites for coverage gaps, TDD/BDD compliance, and anti-patterns. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[athola](https://clawhub.ai/user/ath\n\nArchive v1.9.12: 8 files, 13626 bytes\n\nFiles: modules/content-assertion-quality.md (2534b), modules/coverage-analysis.md (3330b), modules/framework-detection.md (2315b), modules/remediation-planning.md (4849b), modules/scenario-quality.md (5212b), skill-card.md (2111b), SKILL.md (8033b), _meta.json (142b)\n\nArchive v1.0.3: 8 files, 13748 bytes\n\nFiles: modules/content-assertion-quality.md (2534b), modules/coverage-analysis.md (3330b), modules/framework-detection.md (2315b), modules/remediation-planning.md (4849b), modules/scenario-quality.md (5212b), skill-card.md (2479b), SKILL.md (8033b), _meta.json (141b)\n\nArchive v1.0.2: 8 files, 12571 bytes\n\nFiles: modules/content-assertion-quality.md (2534b), modules/coverage-analysis.md (3330b), modules/framework-detection.md (2315b), modules/remediation-planning.md (4849b), modules/scenario-quality.md (5212b), skill-card.md (2084b), SKILL.md (5790b), _meta.json (141b)\n\nArchive v1.0.1: 7 files, 11459 bytes\n\nFiles: modules/content-assertion-quality.md (2534b), modules/coverage-analysis.md (3330b), modules/framework-detection.md (2315b), modules/remediation-planning.md (4849b), modules/scenario-quality.md (5212b), SKILL.md (5790b), _meta.json (141b)\n\nArchive v1.0.0: 7 files, 11460 bytes\n\nFiles: modules/content-assertion-quality.md (2534b), modules/coverage-analysis.md (3330b), modules/framework-detection.md (2315b), modules/remediation-planning.md (4849b), modules/scenario-quality.md (5212b), SKILL.md (5790b), _meta.json (141b)","readmeExcerpt":"Skill: test-review Owner: athola Summary: Evaluates test suites for coverage gaps, TDD/BDD compliance, and anti-patterns Tags: latest:1.9.19 Version history: v1.9.19 | 2026-08-26T13:19:33.885Z | user Release v1.9.19 v1.9.17 | 2026-07-30T05:39:43.732Z | user Release v1.9.17 v1.9.16 | 2026-07-14T19:56:28.437Z | user Release v1.9.16 v1.9.14 | 2026-06-30T18:04:41.848Z | user Release v1.9.14 v1.9.13 | 2026-06-27T16:22:37.","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"/test-review"},{"language":"bash","snippet":"find . -maxdepth 2 -name \"Cargo.toml\" -o -name \"pyproject.toml\" -o -name \"package.json\" -o -name \"go.mod\""},{"language":"bash","snippet":"git diff --name-only | rg 'tests|spec|feature'"},{"language":"markdown","snippet":"## Summary\n[Brief assessment]\n\n## Framework Detection\n- Languages: [list] | Frameworks: [list] | Versions: [constraints]\n\n## Coverage Analysis\n- Overall: X% | Critical: X% | Gaps: [list]\n\n## Quality Issues\n[Q1] [Issue] - Location - Fix\n\n## Remediation Plan\n1. [Action] - Owner - Date\n\n## Recommendation\nApprove / Approve with actions / Block"},{"language":"markdown","snippet":"- [ ] Content assertion depth matches content complexity\n      (L1 for simple skills, L2+ for code examples, L3 for behavioral guidance)"},{"language":"bash","snippet":"# Using tarpaulin\ncargo install cargo-tarpaulin\ncargo tarpaulin --out Html --output-dir coverage/\n\n# Using llvm-cov\ncargo install cargo-llvm-cov\ncargo llvm-cov --html"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: test-review\ndescription: Evaluates test suites for coverage gaps, TDD/BDD compliance, and anti-patterns\nversion: 1.9.8\ntriggers:\n  - testing\n  - tdd\n  - bdd\n  - coverage\n  - quality\n  - fixtures\n  - auditing test quality or before a major release\nmetadata: {\"openclaw\": {\"homepage\": \"https://github.com/athola/claude-night-market/tree/master/plugins/pensive\", \"emoji\": \"\\ud83e\\uddea\", \"requires\": {\"config\": [\"night-market.pensive:shared\", \"night-market.imbue:proof-of-work\"]}}}\nsource: claude-night-market\nsource_plugin: pensive\n---\n\n> **Night Market Skill** — ported from [claude-night-market/pensive](https://github.com/athola/claude-night-market/tree/master/plugins/pensive). For the full experience with agents, hooks, and commands, install the Claude Code plugin.\n\n\n## Table of Contents\n\n- [Quick Start](#quick-start)\n- [When to Use](#when-to-use)\n- [Required TodoWrite Items](#required-todowrite-items)\n- [Progressive Loading](#progressive-loading)\n- [Workflow](#workflow)\n- [Step 1: Detect Languages (`test-review:languages-detected`)](#step-1:-detect-languages-(test-review:languages-detected))\n- [Step 2: Inventory Coverage (`test-review:coverage-inventoried`)](#step-2:-inventory-coverage-(test-review:coverage-inventoried))\n- [Step 3: Assess Scenario Quality (`test-review:scenario-quality`)](#step-3:-assess-scenario-quality-(test-review:scenario-quality))\n- [Step 4: Plan Remediation (`test-review:gap-remediation`)](#step-4:-plan-remediation-(test-review:gap-remediation))\n- [Step 5: Log Evidence (`test-review:evidence-logged`)](#step-5:-log-evidence-(test-review:evidence-logged))\n- [Test Quality Checklist (Condensed)](#test-quality-checklist-(condensed))\n- [Output Format](#output-format)\n- [Summary](#summary)\n- [Framework Detection](#framework-detection)\n- [Coverage Analysis](#coverage-analysis)\n- [Quality Issues](#quality-issues)\n- [Remediation Plan](#remediation-plan)\n- [Recommendation](#recommendation)\n- [Integration Notes](#integration-notes)\n- [Exit Criteria](#exit-criteria)\n\n\n# Test Review Workflow\n\nEvaluate and improve test suites with TDD/BDD rigor.\n\n## Quick Start\n\n```bash\n/test-review\n```\n**Verification:** Run `pytest -v` to verify tests pass.\n\n## When To Use\n\n- Reviewing test suite quality\n- Analyzing coverage gaps\n- Before major releases\n- After test failures\n- Planning test improvements\n\n## When NOT To Use\n\n- Writing new tests - use parseltongue:python-testing\n- Updating existing tests - use sanctum:test-updates\n\n## Required TodoWrite Items\n\n1. `test-review:languages-detected`\n2. `test-review:coverage-inventoried`\n3. `test-review:scenario-quality`\n4. `test-review:invariant-preservation`\n5. `test-review:gap-remediation`\n6. `test-review:evidence-logged`\n\n## Progressive Loading\n\nLoad modules as needed based on review depth:\n\n- **Basic review**: Core workflow (this file)\n- **Framework detection**: Load `modules/framework-detection.md`\n- **Coverage analysis**: Load `modules/coverage-analysis.md`\n- **Quality assessment**: Load `modules/sc"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn7d107jg9jv602h9ytsegydq184a42s\",\n  \"slug\": \"nm-pensive-test-review\",\n  \"version\": \"1.9.19\",\n  \"publishedAt\": 1787750373885\n}"},{"path":"modules/content-assertion-quality.md","content":"# Content Assertion Quality\n\nScoring criteria for evaluating content assertion tests during test review. Extends the scenario quality assessment with a Content Depth dimension.\n\nReference: `leyline:testing-quality-standards/modules/content-assertion-levels.md`\n\n## Content Depth Scoring\n\nRate content assertion depth on a 1-5 scale:\n\n| Score | Level | Description |\n|---|---|---|\n| 1 | None | Tests only file existence or line count |\n| 2 | L1 | Keyword presence checks (`assert \"section\" in content`) |\n| 3 | L2 | Parses embedded examples, validates schema structure |\n| 4 | L3 | Cross-references, anti-patterns, decision framework contracts |\n| 5 | L3+ | Cross-plugin validation (version refs checked against other plugins' docs) |\n\n## When to Flag Missing Content Assertions\n\nDuring test review, flag as a content test gap when:\n\n- A skill has tests but all are L1 (keyword-only) and the skill contains JSON or YAML code blocks\n- A skill has version-gated features but no cross-reference validation\n- A skill defines behavioral guidance (decision trees, strategies) but no anti-pattern or completeness tests\n- A module documents forbidden behaviors but no test asserts their absence\n\n## Content Assertion Anti-Patterns\n\nAvoid these when reviewing content tests:\n\n| Anti-Pattern | Problem | Better Approach |\n|---|---|---|\n| Testing prose style | Brittle to rewording, overlaps with scribe:slop-detector | Test behavioral semantics |\n| Asserting exact wording | Breaks on any edit | Assert concepts (`\"version\" in content.lower()`) |\n| Checking line counts | Not behavioral | Check required sections exist |\n| Testing formatting | Not what Claude interprets | Test parseable structure |\n| Duplicating slop detection | Already handled by scribe | Focus on correctness, not style |\n\n## Review Checklist Addition\n\nAdd this item to the existing Test Quality Checklist when reviewing a plugin that has execution markdown:\n\n```markdown\n- [ ] Content assertion depth matches content complexity\n      (L1 for simple skills, L2+ for code examples, L3 for behavioral guidance)\n```\n\n## Remediation Guidance\n\nWhen content tests are missing or insufficient:\n\n1. **No content tests at all**: Generate L1 scaffolding using `sanctum:test-updates/modules/generation/content-test-templates.md`\n2. **L1 only, has code blocks**: Upgrade to L2 (add JSON/YAML parsing tests)\n3. **L2 only, has version gates**: Upgrade to L3 (add cross-reference validation)\n4. **L2 only, has behavioral guidance**: Upgrade to L3 (add anti-pattern and completeness tests)"},{"path":"modules/coverage-analysis.md","content":"---\nparent_skill: pensive:test-review\nname: coverage-analysis\ndescription: Coverage measurement and gap identification\ncategory: testing\ntags: [coverage, testing, gap-analysis]\nload_priority: 2\nestimated_tokens: 350\n---\n\n# Coverage Analysis\n\nMeasure test coverage and identify gaps.\n\n## Coverage Tools by Language\n\n### Rust\n```bash\n# Using tarpaulin\ncargo install cargo-tarpaulin\ncargo tarpaulin --out Html --output-dir coverage/\n\n# Using llvm-cov\ncargo install cargo-llvm-cov\ncargo llvm-cov --html\n```\n\n### Python\n```bash\n# Using pytest-cov\npytest --cov=src --cov-report=html --cov-report=term-missing\n\n# Using coverage.py\ncoverage run -m pytest\ncoverage html\ncoverage report --show-missing\n```\n\n### JavaScript/TypeScript\n```bash\n# Jest\nnpm test -- --coverage --coverageReporters=html text\n\n# Vitest\nvitest --coverage\n\n# Cypress (code coverage plugin)\ncypress run --env coverage=true\n```\n\n### Go\n```bash\n# Built-in coverage\ngo test -cover ./...\ngo test -coverprofile=coverage.out ./...\ngo tool cover -html=coverage.out\n\n# Detailed coverage\ngo test -covermode=count -coverprofile=coverage.out ./...\n```\n\n## Coverage Thresholds\n\n| Level | Coverage | Use Case |\n|-------|----------|----------|\n| Minimum | 60% | Legacy code, initial cleanup |\n| Standard | 80% | Normal development |\n| High | 90% | Critical systems, libraries |\n| detailed | 95%+ | Safety-critical, financial |\n\n## Gap Identification\n\n### Find impacted test files\n```bash\n# Tests affected by changes\ngit diff --name-only main...HEAD | rg 'tests|spec|feature'\n\n# Find related tests\ngit diff --name-only main...HEAD | while read file; do\n  basename \"$file\" .py | xargs -I {} find . \\\n    -not -path \"*/.venv/*\" -not -path \"*/__pycache__/*\" \\\n    -not -path \"*/node_modules/*\" -not -path \"*/.git/*\" \\\n    -name \"*test*{}*\"\ndone\n```\n\n### Identify uncovered code\n1. Run coverage tool with `--show-missing` flag\n2. Cross-reference with critical paths:\n   - Authentication/authorization\n   - Data validation\n   - Error handling\n   - API endpoints\n   - Database operations\n\n3. Map to requirements:\n   - Feature specifications\n   - User stories\n   - Bug reports\n   - Security requirements\n\n### Coverage Patterns\n\n**Critical paths** (should be 100%):\n- Security boundaries (auth, validation)\n- Data integrity operations\n- Error recovery logic\n- Public API surface\n\n**Lower priority** (can be <80%):\n- Internal helpers\n- Logging/debugging code\n- Trivial getters/setters\n- Deprecated code paths\n\n## Output Format\n\n```markdown\n## Coverage Analysis\n- **Overall**: 78%\n- **Critical paths**: 92%\n- **Changed files**: 85%\n\n### Gaps Identified\n1. **src/auth.py:45-60** - Token validation edge cases\n2. **src/api/routes.py:120-135** - Error handling for 400/500 codes\n3. **src/db/migrations.py** - Rollback scenarios untested\n\n### Test-to-Feature Mapping\n- Feature: User registration → `tests/test_registration.py` (95%)\n- Feature: Password reset → `tests/test_auth.py` (60%) [WARN]\n- Feature: Email validation → Missing tests [FAIL]\n```\n\n## Best Practice"},{"path":"modules/framework-detection.md","content":"---\nparent_skill: pensive:test-review\nname: framework-detection\ndescription: Language and test framework detection patterns\ncategory: testing\ntags: [testing, framework-detection, language-detection]\nload_priority: 1\nestimated_tokens: 250\n---\n\n# Framework Detection\n\nIdentify testing frameworks and tooling constraints.\n\n## Language Detection Patterns\n\n### Rust\n- **Framework**: cargo test (built-in)\n- **Commands**: `cargo test`, `cargo nextest run`\n- **Config files**: `Cargo.toml`, `Cargo.lock`\n- **Test patterns**: `#[test]`, `#[cfg(test)]`\n- **MSRV**: Check `rust-version` in Cargo.toml\n\n### Python\n- **Frameworks**: pytest, unittest, behave\n- **Commands**: `pytest`, `python -m pytest`, `behave`\n- **Config files**: `pytest.ini`, `pyproject.toml`, `tox.ini`\n- **Test patterns**: `test_*.py`, `*_test.py`, `tests/`\n- **Version**: Check `requires-python` in pyproject.toml\n\n### JavaScript/TypeScript\n- **Frameworks**: Jest, Mocha, Cypress, Vitest\n- **Commands**: `npm test`, `yarn test`, `cypress run`\n- **Config files**: `jest.config.js`, `vitest.config.ts`, `cypress.config.js`\n- **Test patterns**: `*.test.js`, `*.spec.ts`, `__tests__/`\n- **Version**: Check `engines.node` in package.json\n\n### Go\n- **Framework**: go test (built-in)\n- **Commands**: `go test ./...`, `go test -v`\n- **Config files**: `go.mod`, `go.sum`\n- **Test patterns**: `*_test.go`\n- **Version**: Check `go` directive in go.mod\n\n## Detection Workflow\n\n1. **Scan for config files**:\n```bash\nfind . -maxdepth 2 -name \"Cargo.toml\" -o -name \"pyproject.toml\" -o -name \"package.json\" -o -name \"go.mod\"\n```\n\n2. **Check test directories**:\n```bash\nfind . -type d -name \"tests\" -o -name \"__tests__\" -o -name \"test\"\n```\n\n3. **Identify test files**:\n```bash\nfind . -not -path \"*/.venv/*\" -not -path \"*/__pycache__/*\" \\\n  -not -path \"*/node_modules/*\" -not -path \"*/.git/*\" \\\n  \\( -name \"*test*\" -o -name \"*spec*\" \\) \\\n  | grep -E '\\.(rs|py|js|ts|go)$'\n```\n\n4. **Version constraints**:\n- Extract MSRV, Python version, Node version\n- Note if constraints affect tooling (e.g., async/await)\n- Document CI/CD version requirements\n\n## Output Format\n\n```markdown\n## Framework Detection\n- **Languages**: Rust, Python\n- **Frameworks**: cargo test, pytest\n- **Versions**:\n  - Rust MSRV: 1.70\n  - Python: >=3.8\n- **Config files**: Cargo.toml, pyproject.toml\n```"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Evaluates test suites for coverage gaps, TDD/BDD compliance, and anti-patterns Skill: test-review Owner: athola Summary: Evaluates test suites for coverage gaps, TDD/BDD compliance, and anti-patterns Tags: latest:1.9.19 Version history: v1.9.19 | 2026-08-26T13:19:33.885Z | user Release v1.9.19 v1.9.17 | 2026-07-30T05:39:43.732Z | user Release v1.9.17 v1.9.16 | 2026-07-14T19:56:28.437Z | user Release v1.9.16 v1.9.14 | 2026-06-30T18:04:41.848Z | user Release v1.9.14 v1.9.13 | 2026-06-27T16:22:37.","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1385,"uniquenessScore":50,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-10T07:58:53.992Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-10T07:58:53.992Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T10:49:59.034Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}