{"id":"1b420aab-f0c0-49c3-8b22-77aec7458ee1","entityType":"agent","slug":"clawhub-athola-nm-imbue-feature-review","name":"feature-review","canonicalUrl":"https://www.xpersona.co/agent/clawhub-athola-nm-imbue-feature-review","canonicalPath":"/agent/clawhub-athola-nm-imbue-feature-review","generatedAt":"2026-10-10T08:12:50.985Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-10T07:51:01.745Z","emptyReason":null},"description":"Scores backlog items with RICE/WSJF/Kano and files GitHub issues for top candidates Skill: feature-review Owner: athola Summary: Scores backlog items with RICE/WSJF/Kano and files GitHub issues for top candidates Tags: latest:1.9.19 Version history: v1.9.19 | 2026-08-26T13:12:29.298Z | user Release v1.9.19 v1.9.17 | 2026-07-30T05:34:08.720Z | user Release v1.9.17 v1.9.16 | 2026-07-14T19:50:52.775Z | user Release v1.9.16 v1.9.14 | 2026-06-30T17:59:58.487Z | user Release v1.9.14 v1.9.13 | 2026-06-27T1","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.6K downloads reported by the source. Last updated 10/10/2026.","installCommand":"clawhub skill install s17emme0e2m3cpf7k2jvp3a84984b8z9:nm-imbue-feature-review","sourceUrl":"https://clawhub.ai/athola/nm-imbue-feature-review","homepage":"https://clawhub.ai/athola/skills/nm-imbue-feature-review","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/athola/nm-imbue-feature-review","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/athola/skills/nm-imbue-feature-review","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":40,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Scores backlog items with RICE/WSJF/Kano and files GitHub issues for top candidates Skill: feature-review Owner: athola Summary: Scores backlog items with RICE/"},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-10T07:51:01.745Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T07:51:01.745Z","emptyReason":null},"stars":null,"forks":null,"downloads":1579,"packageName":null,"latestVersion":"1.9.19","tractionLabel":"1.6K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T07:51:01.744Z","emptyReason":null},"lastUpdatedAt":"2026-10-10T07:51:01.745Z","lastCrawledAt":"2026-10-10T07:51:01.744Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-11T07:51:01.744Z","lastVerifiedAt":null,"highlights":[{"version":"1.9.19","createdAt":"2026-08-26T13:12:29.298Z","changelog":"Release v1.9.19","fileCount":9,"zipByteSize":28301},{"version":"1.9.17","createdAt":"2026-07-30T05:34:08.720Z","changelog":"Release v1.9.17","fileCount":9,"zipByteSize":28336},{"version":"1.9.16","createdAt":"2026-07-14T19:50:52.775Z","changelog":"Release v1.9.16","fileCount":9,"zipByteSize":28261},{"version":"1.9.14","createdAt":"2026-06-30T17:59:58.487Z","changelog":"Release v1.9.14","fileCount":9,"zipByteSize":28507},{"version":"1.9.13","createdAt":"2026-06-27T16:18:44.099Z","changelog":"Release v1.9.13","fileCount":9,"zipByteSize":28395},{"version":"1.9.12","createdAt":"2026-06-19T03:12:51.321Z","changelog":"Release v1.9.12","fileCount":9,"zipByteSize":28294},{"version":"1.0.3","createdAt":"2026-06-18T14:07:59.616Z","changelog":"Release v1.9.12","fileCount":9,"zipByteSize":28341},{"version":"1.0.2","createdAt":"2026-05-09T02:17:18.325Z","changelog":"Release v1.9.5","fileCount":8,"zipByteSize":26933}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s17emme0e2m3cpf7k2jvp3a84984b8z9:nm-imbue-feature-review","setupComplexity":"low","setupSteps":["Setup complexity is classified as HIGH. You must provision dedicated cloud infrastructure or an isolated VM. Do not run this directly on your local workstation.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-athola-nm-imbue-feature-review/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-athola-nm-imbue-feature-review/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-athola-nm-imbue-feature-review/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-athola-nm-imbue-feature-review/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-athola-nm-imbue-feature-review/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-athola-nm-imbue-feature-review/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-10T08:12:50.982Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-athola-nm-imbue-feature-review/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-athola-nm-imbue-feature-review/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-athola-nm-imbue-feature-review/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-athola-nm-imbue-feature-review/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-10T07:51:01.745Z","emptyReason":null},"readme":"Skill: feature-review\n\nOwner: athola\n\nSummary: Scores backlog items with RICE/WSJF/Kano and files GitHub issues for top candidates\n\nTags: latest:1.9.19\n\nVersion history:\n\nv1.9.19 | 2026-08-26T13:12:29.298Z | user\n\nRelease v1.9.19\n\nv1.9.17 | 2026-07-30T05:34:08.720Z | user\n\nRelease v1.9.17\n\nv1.9.16 | 2026-07-14T19:50:52.775Z | user\n\nRelease v1.9.16\n\nv1.9.14 | 2026-06-30T17:59:58.487Z | user\n\nRelease v1.9.14\n\nv1.9.13 | 2026-06-27T16:18:44.099Z | user\n\nRelease v1.9.13\n\nv1.9.12 | 2026-06-19T03:12:51.321Z | user\n\nRelease v1.9.12\n\nv1.0.3 | 2026-06-18T14:07:59.616Z | user\n\nRelease v1.9.12\n\nv1.0.2 | 2026-05-09T02:17:18.325Z | user\n\nRelease v1.9.5\n\nv1.0.1 | 2026-05-06T14:17:58.929Z | user\n\nRelease v1.9.4\n\nv1.0.0 | 2026-04-12T12:01:39.591Z | auto\n\n- Initial release of the feature review skill, supporting evidence-based feature prioritization.\n- Enables reviewing and scoring features using RICE, WSJF, and Kano frameworks.\n- Supports multi-phase workflow: inventory, classification, scoring, tradeoff analysis, gap analysis, and GitHub issue creation for suggestions.\n- Offers integration with external research (optional) via the tome plugin to enrich scoring.\n- Can generate actionable GitHub issues for prioritized feature suggestions.\n- Includes clear usage guidance, configuration, and guardrails for safe application.\n\nArchive index:\n\nArchive v1.9.19: 9 files, 28301 bytes\n\nFiles: modules/classification-system.md (8999b), modules/configuration.md (6593b), modules/multi-metric-evaluation-methodology.md (8759b), modules/research-enrichment.md (5903b), modules/scoring-framework.md (10049b), modules/tradeoff-dimensions.md (10886b), skill-card.md (2554b), SKILL.md (12170b), _meta.json (143b)\n\nFile v1.9.19:SKILL.md\n\n---\nname: feature-review\ndescription: |\n  Scores backlog items with RICE/WSJF/Kano and files GitHub issues for top candidates\nversion: 1.9.8\ntriggers:\n  - feature-prioritization\n  - backlog-triage\n  - RICE\n  - WSJF\n  - Kano\n  - roadmap\n  - triaging a roadmap or prioritizing features for a sprint\nmetadata: {\"openclaw\": {\"homepage\": \"https://github.com/athola/claude-night-market/tree/master/plugins/imbue\", \"emoji\": \"\\ud83e\\udd9e\", \"requires\": {\"config\": [\"night-market.imbue:scope-guard\"]}}}\nsource: claude-night-market\nsource_plugin: imbue\n---\n\n> **Night Market Skill** — ported from [claude-night-market/imbue](https://github.com/athola/claude-night-market/tree/master/plugins/imbue). For the full experience with agents, hooks, and commands, install the Claude Code plugin.\n\n\n## Table of Contents\n\n- [Philosophy](#philosophy)\n- [When to Use](#when-to-use)\n- [When NOT to Use](#when-not-to-use)\n- [Quick Start](#quick-start)\n- [1. Inventory Current Features](#1-inventory-current-features)\n- [2. Score and Classify](#2-score-and-classify)\n- [3. Generate Suggestions](#3-generate-suggestions)\n\n## Verification\n\nRun `make test-feature-review` to verify scoring logic after changes.\n- [4. Upload to GitHub](#4-upload-to-github)\n- [Workflow](#workflow)\n- [Phase 1: Feature Discovery (`feature-review:inventory-complete`)](#phase-1:-feature-discovery-(feature-review:inventory-complete))\n- [Phase 2: Classification (`feature-review:classified`)](#phase-2:-classification-(feature-review:classified))\n- [Phase 3: Scoring (`feature-review:scored`)](#phase-3:-scoring-(feature-review:scored))\n- [Phase 4: Tradeoff Analysis (`feature-review:tradeoffs-analyzed`)](#phase-4:-tradeoff-analysis-(feature-review:tradeoffs-analyzed))\n- [Phase 5: Gap Analysis & Suggestions (`feature-review:suggestions-generated`)](#phase-5:-gap-analysis-&-suggestions-(feature-review:suggestions-generated))\n- [Phase 6: GitHub Integration (`feature-review:issues-created`)](#phase-6:-github-integration-(feature-review:issues-created))\n- [Configuration](#configuration)\n- [Configuration File](#configuration-file)\n- [Guardrails](#guardrails)\n- [Required TodoWrite Items](#required-todowrite-items)\n- [Integration Points](#integration-points)\n- [Output Format](#output-format)\n- [Feature Inventory Table](#feature-inventory-table)\n- [Suggestion Report](#suggestion-report)\n- [Feature Suggestions](#feature-suggestions)\n- [High Priority (Score > 2.5)](#high-priority-(score->-25))\n- [Related Skills](#related-skills)\n- [Reference](#reference)\n\n\n# Feature Review\n\nReview implemented features and suggest new ones using evidence-based prioritization. Create GitHub issues for accepted suggestions.\n\n## Philosophy\n\nFeature decisions rely on data. Every feature involves tradeoffs that require evaluation. This skill uses hybrid RICE+WSJF scoring with Kano classification to prioritize work and generates actionable GitHub issues for accepted suggestions.\n\n## When To Use\n\n- Roadmap reviews (sprint planning, quarterly reviews).\n- Retrospective evaluations.\n- Planning new development cycles.\n\n## When NOT To Use\n\n- Emergency bug fixes.\n- Simple documentation updates.\n- Active implementation (use `scope-guard`).\n\n## Quick Start\n\n### 1. Inventory Current Features\n\nDiscover and categorize existing features:\n```bash\n/feature-review --inventory\n```\n\n### 2. Score and Classify\n\nEvaluate features against the prioritization framework:\n```bash\n/feature-review\n```\n\n### 3. Generate Suggestions\n\nReview gaps and suggest new features:\n```bash\n/feature-review --suggest\n```\n\n### 4. Research-Enriched Scoring\n\nUse tome plugin to adjust scores with external evidence:\n```bash\n/feature-review --research\n```\n\n### 5. Upload to GitHub\n\nCreate issues for accepted suggestions:\n```bash\n/feature-review --suggest --create-issues\n```\n\n## Workflow\n\n### Phase 1: Feature Discovery (`feature-review:inventory-complete`)\n\nIdentify features by analyzing:\n\n1. **Code artifacts**: Entry points, public APIs, and configuration surfaces.\n2. **Documentation**: README lists, CHANGELOG entries, and user docs.\n3. **Git history**: Recent feature commits and branches.\n\n**Output:** Feature inventory table.\n\n### Phase 2: Classification (`feature-review:classified`)\n\nClassify each feature along two axes:\n\n**Axis 1: Proactive vs Reactive**\n\n| Type | Definition | Examples |\n|------|------------|----------|\n| **Proactive** | Anticipates user needs. | Suggestions, prefetching. |\n| **Reactive** | Responds to explicit input. | Form handling, click actions. |\n\n**Axis 2: Static vs Dynamic**\n\n| Type | Update Pattern | Storage Model |\n|------|---------------|---------------|\n| **Static** | Incremental, versioned. | File-based, cached. |\n| **Dynamic** | Continuous, streaming. | Database, real-time. |\n\nSee [classification-system.md](modules/classification-system.md) for details.\n\n### Phase 3: Scoring (`feature-review:scored`)\n\nApply hybrid RICE+WSJF scoring:\n\n```\nFeature Score = Value Score / Cost Score\n\nValue Score = (Reach + Impact + Business Value + Time Criticality) / 4\nCost Score = (Effort + Risk + Complexity) / 3\n\nAdjusted Score = Feature Score * Confidence\n```\n\n**Scoring Scale:** Fibonacci (1, 2, 3, 5, 8, 13).\n\n**Thresholds:**\n- **> 2.5**: High priority.\n- **1.5 - 2.5**: Medium priority.\n- **< 1.5**: Low priority.\n\nSee [scoring-framework.md](modules/scoring-framework.md) for the framework.\nSee [multi-metric-evaluation-methodology.md](modules/multi-metric-evaluation-methodology.md)\nwhen one model is not enough: it covers how to combine\nRICE, WSJF, and Kano, where each model fits, and how to\nreconcile conflicting signals.\n\n### Phase 4: Tradeoff Analysis (`feature-review:tradeoffs-analyzed`)\n\nEvaluate each feature across quality dimensions:\n\n| Dimension | Question | Scale |\n|-----------|----------|-------|\n| **Quality** | Does it deliver correct results? | 1-5 |\n| **Latency** | Does it meet timing requirements? | 1-5 |\n| **Token Usage** | Is it context-efficient? | 1-5 |\n| **Resource Usage** | Is CPU/memory reasonable? | 1-5 |\n| **Redundancy** | Does it handle failures gracefully? | 1-5 |\n| **Readability** | Can others understand it? | 1-5 |\n| **Scalability** | Will it handle 10x load? | 1-5 |\n| **Integration** | Does it play well with others? | 1-5 |\n| **API Surface** | Is it backward compatible? | 1-5 |\n\nSee [tradeoff-dimensions.md](modules/tradeoff-dimensions.md) for criteria.\n\n### Phase 4.5: Research Enrichment (`feature-review:research-enriched`)\n\n**Triggered by:** `--research` flag. Requires tome plugin.\n\nUse tome's multi-source research to adjust scoring factors\nwith external evidence. This phase runs between tradeoff\nanalysis and gap analysis.\n\n1. **Dispatch research**: For each feature, construct\n   research topics and dispatch tome channels (code-search,\n   discourse, papers, triz) in parallel.\n2. **Synthesize findings**: Merge results across channels\n   using `tome:synthesize`.\n3. **Calculate deltas**: Map findings to scoring factor\n   adjustments using channel-to-factor mapping.\n4. **Apply deltas**: Adjust initial scores by research\n   deltas, clamp to Fibonacci scale, respect max_delta.\n5. **Present evidence**: Show adjustment table with\n   evidence sources and rationale.\n\nSee [research-enrichment.md](modules/research-enrichment.md)\nfor the full enrichment protocol, delta calculation, and\ngraceful degradation behavior.\n\n**Graceful degradation**: If tome is not installed, prints\na warning and proceeds with initial scores unchanged.\n\n### Phase 5: Gap Analysis & Suggestions (`feature-review:suggestions-generated`)\n\n1. **Identify gaps**: Missing Kano basics.\n2. **Surface opportunities**: High-value, low-effort features.\n3. **Flag technical debt**: Features with declining scores.\n4. **Recommend actions**: Build, improve, deprecate, or maintain.\n\n### Phase 6: GitHub Integration (`feature-review:issues-created`)\n\n1. Generate issue title and body from suggestions.\n2. Apply labels (feature, enhancement, priority/*).\n3. Link to related issues.\n4. Confirm with user before creation.\n\n**Deferred capture for high-scoring suggestions:**\nAfter the user confirms which suggestions to act on, any\nhigh-scoring suggestion (score > 2.5) that is not acted on\nshould be preserved as a deferred item.\nRun once per skipped high-scoring suggestion:\n\n```bash\npython3 scripts/deferred_capture.py \\\n  --title \"<suggestion title>\" \\\n  --source feature-review \\\n  --context \"RICE score: <score>. <description>\"\n```\n\nThis runs automatically without prompting the user.\nSuggestions with scores of 2.5 or below do not need\nto be captured.\n\n## Configuration\n\nFeature-review uses opinionated defaults but allows customization.\n\n### Configuration File\n\nCreate `.feature-review.yaml` in project root:\n\n```yaml\n# .feature-review.yaml\nversion: 1.9.3\n\n# Scoring weights (must sum to 1.0)\nweights:\n  value:\n    reach: 0.25\n    impact: 0.30\n    business_value: 0.25\n    time_criticality: 0.20\n  cost:\n    effort: 0.40\n    risk: 0.30\n    complexity: 0.30\n\n# Score thresholds\nthresholds:\n  high_priority: 2.5\n  medium_priority: 1.5\n\n# Tradeoff dimension weights (0.0 to disable)\ntradeoffs:\n  quality: 1.0\n  latency: 1.0\n  token_usage: 1.0\n  resource_usage: 0.8\n  redundancy: 0.5\n  readability: 1.0\n  scalability: 0.8\n  integration: 1.0\n  api_surface: 1.0\n```\n\nSee [configuration.md](modules/configuration.md) for options.\n\n### Guardrails\n\nThese rules apply to all configurations:\n\n1. **Minimum dimensions**: Evaluate at least 5 tradeoff dimensions.\n2. **Confidence requirement**: Review scores below 50% confidence.\n3. **Breaking change warning**: Require acknowledgment for API surface changes.\n4. **Backlog limit**: Limit suggestion queue to 25 items.\n\n## Required TodoWrite Items\n\n1. `feature-review:inventory-complete`\n2. `feature-review:classified`\n3. `feature-review:scored`\n4. `feature-review:tradeoffs-analyzed`\n5. `feature-review:research-enriched` (if `--research`)\n6. `feature-review:suggestions-generated`\n7. `feature-review:issues-created` (if requested)\n\n## Integration Points\n\n- **`imbue:scope-guard`**: Provides Worthiness Scores for suggestions.\n- **`sanctum:do-issue`**: Prioritizes issues with high scores.\n- **`superpowers:brainstorming`**: Evaluates new ideas against existing features.\n- **`tome:research`**: Multi-source research for score enrichment (optional, `--research`).\n\n## Output Format\n\n### Feature Inventory Table\n\n```markdown\n| Feature | Type | Data | Score | Priority | Status |\n|---------|------|------|-------|----------|--------|\n| Auth middleware | Reactive | Dynamic | 2.8 | High | Stable |\n| Skill loader | Reactive | Static | 2.3 | Medium | Needs improvement |\n```\n\n### Research-Enriched Table (with `--research`)\n\n```markdown\n| Feature | Type | Score | Adj. | Priority | Evidence |\n|---------|------|-------|------|----------|----------|\n| Auth    | R/D  | 2.8   | 3.1  | High     | 3 sources |\n| Loader  | R/S  | 2.3   | 2.3  | Medium   | none      |\n\n## Research Evidence\n\n### Code Search (GitHub)\n- 12 implementations, avg 340 stars\n- **Reach**: +1 (broad adoption)\n\n### Discourse (HN/Reddit)\n- 47 mentions, 78% positive\n- **Impact**: +1 (strong demand)\n```\n\n### Suggestion Report\n\n```markdown\n## Feature Suggestions\n\n### High Priority (Score > 2.5)\n\n1. **[Feature Name]** (Score: 2.7)\n   - Classification: Proactive/Dynamic\n   - Value: High reach\n   - Cost: Moderate effort\n   - Recommendation: Build in next sprint\n```\n\n## Related Skills\n\n- `imbue:scope-guard`: Prevent overengineering.\n- `sanctum:pr-review`: Code-level review (different scope: this\n  skill prioritizes feature ideas, pr-review reviews diffs).\n\n## Reference\n\n- **[scoring-framework.md](modules/scoring-framework.md)**: RICE+WSJF hybrid.\n- **[classification-system.md](modules/classification-system.md)**: Axes definition.\n- **[tradeoff-dimensions.md](modules/tradeoff-dimensions.md)**: Quality attributes.\n- **[research-enrichment.md](modules/research-enrichment.md)**: tome-driven score deltas, channel-to-factor mapping, graceful degradation.\n- **[multi-metric-evaluation-methodology.md](modules/multi-metric-evaluation-methodology.md)**: Combining RICE, WSJF, and Kano when no single model suffices.\n- **[configuration.md](modules/configuration.md)**: Customization options.\n\nFile v1.9.19:_meta.json\n\n{\n  \"ownerId\": \"kn7d107jg9jv602h9ytsegydq184a42s\",\n  \"slug\": \"nm-imbue-feature-review\",\n  \"version\": \"1.9.19\",\n  \"publishedAt\": 1787749949298\n}\n\nFile v1.9.19:modules/classification-system.md\n\n# Classification System\n\nFeatures are classified along two orthogonal axes that determine architectural and UX implications.\n\n## Axis 1: Proactive vs Reactive\n\nThis axis describes **when** the feature acts relative to user intent.\n\n### Proactive Features\n\n**Definition:** Anticipates user needs and acts before explicit request.\n\n**Characteristics:**\n- Runs in background or ahead of user action\n- Requires prediction/inference of user intent\n- May consume resources speculatively\n- Higher latency tolerance (users don't wait)\n\n**Latency Tolerance:**\n- Background processing acceptable (seconds to minutes)\n- User doesn't perceive delay directly\n- Can be batched or deferred\n\n**Examples:**\n| Feature | How It's Proactive |\n|---------|-------------------|\n| Auto-save | Saves before user requests |\n| Prefetching | Loads data before navigation |\n| Suggestions | Offers options before user types |\n| Health checks | Monitors before problems occur |\n| Cache warming | Prepares data before access |\n\n**Tradeoffs:**\n| Pro | Con |\n|-----|-----|\n| Reduces user effort | May waste resources |\n| Feels \"smart\" | Can be wrong/intrusive |\n| Prevents problems | Requires more data |\n| Smoother UX | Higher complexity |\n\n**Architecture Patterns:**\n- Event-driven / pub-sub\n- Background workers\n- Predictive models\n- Eventual consistency acceptable\n\n### Reactive Features\n\n**Definition:** Responds to explicit user input or system events.\n\n**Characteristics:**\n- Triggered by user action\n- Must feel immediate\n- Resources used on-demand\n- Correctness over speculation\n\n**Latency Tolerance:**\n- Sub-100ms for UI feedback\n- Sub-1s for completion\n- User actively waiting\n\n**Examples:**\n| Feature | How It's Reactive |\n|---------|------------------|\n| Form submission | User clicks submit |\n| Search | User types query |\n| Navigation | User clicks link |\n| Validation | User enters input |\n| Commands | User invokes action |\n\n**Tradeoffs:**\n| Pro | Con |\n|-----|-----|\n| User in control | User must initiate |\n| Predictable behavior | No anticipation |\n| Lower resource waste | Perceived latency |\n| Simpler to implement | Less \"magical\" UX |\n\n**Architecture Patterns:**\n- Request/response\n- Synchronous processing\n- Strong consistency\n- Direct invocation\n\n### Classification Decision Tree\n\n```\nIs the feature triggered by explicit user action?\n├── Yes → Is immediate response critical?\n│   ├── Yes → REACTIVE\n│   └── No → Could be either (consider UX goals)\n└── No → Does it require user data/context?\n    ├── Yes → PROACTIVE (with data)\n    └── No → PROACTIVE (autonomous)\n```\n\n## Axis 2: Static vs Dynamic\n\nThis axis describes **how** feature data changes over time.\n\n### Static Features\n\n**Definition:** Data changes incrementally through explicit updates.\n\n**Characteristics:**\n- Version-controlled or release-based updates\n- Can be cached aggressively\n- Deterministic lookups\n- Stale data possible but predictable\n\n**Update Pattern:**\n- Deploy-time updates\n- Batch processing\n- Periodic refresh\n- Manual triggers\n\n**Storage Models:**\n| Model | Use Case |\n|-------|----------|\n| Files | Configuration, templates |\n| Embedded | Constants, schemas |\n| CDN | Assets, documentation |\n| Read replicas | Reference data |\n\n**Lookup Cost:** O(1) or O(log n), highly cacheable\n\n**Examples:**\n| Feature | Why It's Static |\n|---------|----------------|\n| Skill definitions | Updated via deploy |\n| Documentation | Published versions |\n| Configuration | Changed by admin |\n| Templates | Version-controlled |\n| Schema definitions | Release-based |\n\n**Tradeoffs:**\n| Pro | Con |\n|-----|-----|\n| Fast lookups | Can be stale |\n| Simple architecture | Update lag |\n| Highly cacheable | Deployment required |\n| Predictable performance | Less responsive |\n\n### Dynamic Features\n\n**Definition:** Data changes continuously through ongoing operations.\n\n**Characteristics:**\n- Real-time or near-real-time updates\n- Limited caching opportunity\n- Query-based lookups\n- Consistency challenges\n\n**Update Pattern:**\n- Event-driven updates\n- Streaming ingestion\n- Live queries\n- Continuous sync\n\n**Storage Models:**\n| Model | Use Case |\n|-------|----------|\n| Database | Transactional data |\n| Cache layer | Hot data |\n| Stream | Events, logs |\n| Search index | Queryable content |\n\n**Lookup Cost:** O(log n) to O(n), cache hit-rate varies\n\n**Examples:**\n| Feature | Why It's Dynamic |\n|---------|-----------------|\n| User sessions | Real-time state |\n| Search results | Live queries |\n| Notifications | Streaming events |\n| Analytics | Continuous ingestion |\n| Collaboration | Multi-user sync |\n\n**Tradeoffs:**\n| Pro | Con |\n|-----|-----|\n| Always fresh | Higher latency |\n| Responsive to change | Complex architecture |\n| Real-time capable | Consistency challenges |\n| User-specific | Harder to cache |\n\n### Classification Decision Tree\n\n```\nDoes the data change based on user actions in real-time?\n├── Yes → DYNAMIC\n└── No → Is freshness critical (< 1 hour)?\n    ├── Yes → DYNAMIC\n    └── No → Could the data be served from cache/CDN?\n        ├── Yes → STATIC\n        └── No → Consider hybrid (static + refresh)\n```\n\n## The 2x2 Matrix\n\nCombining both axes creates four feature archetypes:\n\n```\n                    STATIC                 DYNAMIC\n              ┌─────────────────────┬─────────────────────┐\n              │                     │                     │\n   PROACTIVE  │   Predictive Cache  │   Smart Assistant   │\n              │   (prefetch static) │   (live suggestions)│\n              │                     │                     │\n              │   Latency: Low      │   Latency: Medium   │\n              │   Complexity: Low   │   Complexity: High  │\n              │                     │                     │\n              ├─────────────────────┼─────────────────────┤\n              │                     │                     │\n   REACTIVE   │   Reference Lookup  │   Interactive Query │\n              │   (docs, configs)   │   (search, forms)   │\n              │                     │                     │\n              │   Latency: Very Low │   Latency: Low      │\n              │   Complexity: Low   │   Complexity: Medium│\n              │                     │                     │\n              └─────────────────────┴─────────────────────┘\n```\n\n### Archetype Details\n\n#### Predictive Cache (Proactive and Static)\n\n- **Example:** Prefetching documentation pages\n- **Pattern:** Background worker loads static assets\n- **Complexity:** Low - just scheduling and caching\n- **Risk:** Wasted bandwidth if prediction wrong\n\n#### Smart Assistant (Proactive and Dynamic)\n\n- **Example:** AI-powered suggestions based on context\n- **Pattern:** Real-time inference on streaming data\n- **Complexity:** High - ML models, data pipelines\n- **Risk:** Expensive, can be wrong, privacy concerns\n\n#### Reference Lookup (Reactive and Static)\n\n- **Example:** Loading skill definitions\n- **Pattern:** Cache-first, fallback to file\n- **Complexity:** Low - simple read operations\n- **Risk:** Stale data if cache not invalidated\n\n#### Interactive Query (Reactive and Dynamic)\n\n- **Example:** Search across current repository\n- **Pattern:** Query on demand, may use indexes\n- **Complexity:** Medium - query optimization, indexing\n- **Risk:** Variable latency, consistency windows\n\n## Classification for Common Features\n\n| Feature Type | Typical Classification | Notes |\n|--------------|----------------------|-------|\n| CLI Commands | Reactive and Static | User-invoked, defined behavior |\n| Auto-complete | Proactive and Dynamic | Predicts input from context |\n| Configuration | Reactive and Static | Loaded on demand, versioned |\n| Session state | Reactive and Dynamic | User-driven, real-time |\n| Caching layer | Proactive and Static | Anticipates access patterns |\n| Notifications | Proactive and Dynamic | Pushed based on events |\n| Validation | Reactive and Static | Rules are static, input is dynamic |\n| Analytics | Proactive and Dynamic | Background collection |\n\n## Using Classification in Review\n\nWhen reviewing features:\n\n1. **Identify current classification** - What is it today?\n2. **Evaluate fit** - Does classification match use case?\n3. **Consider migration** - Would different classification improve UX?\n4. **Note tradeoffs** - What would change with different classification?\n\n**Red Flags:**\n- Reactive feature with high latency → Consider proactive alternative\n- Dynamic feature rarely changing → Could be static for performance\n- Proactive feature often wrong → Consider reactive fallback\n- Static feature causing staleness issues → Consider dynamic refresh\n\nFile v1.9.19:modules/configuration.md\n\n# Configuration\n\nFeature-review uses opinionated defaults but allows\nproject-specific customization through a YAML\nconfiguration file.\n\n## Configuration File Location\n\nCreate `.feature-review.yaml` in your project root:\n\n```\nproject/\n├── .feature-review.yaml    # Configuration file\n├── src/\n└── ...\n```\n\n## Full Configuration Schema\n\n```yaml\n# .feature-review.yaml\n# All values shown are defaults - only specify what you want to change\n\nversion: 1  # Schema version (required if file exists)\n\nweights:\n  value:\n    reach: 0.25              # How many users affected\n    impact: 0.30             # How much improvement per user\n    business_value: 0.25     # OKR/strategic alignment\n    time_criticality: 0.20   # Cost of delay\n  cost:\n    effort: 0.40             # Development time\n    risk: 0.30               # Uncertainty/unknowns\n    complexity: 0.30         # Technical difficulty\n\nthresholds:\n  high_priority: 2.5         # Score > 2.5 = implement soon\n  medium_priority: 1.5       # Score > 1.5 = roadmap candidate\n  confidence_warning: 0.5    # Scores below this get flagged\n\nclassification:\n  default_type: reactive     # proactive | reactive\n  default_data: static       # static | dynamic\n  patterns:\n    proactive_patterns: [\"*auto*\", \"*suggest*\", \"*predict*\", \"*prefetch*\"]\n    dynamic_patterns: [\"*session*\", \"*realtime*\", \"*live*\", \"*stream*\"]\n\ntradeoffs:\n  quality: 1.0               # Correctness of results\n  latency: 1.0               # Response time\n  token_usage: 1.0           # Context efficiency (LLM-specific)\n  resource_usage: 0.8        # CPU/memory consumption\n  redundancy: 0.5            # Fault tolerance\n  readability: 1.0           # Code maintainability\n  scalability: 0.8           # Growth handling\n  integration: 1.0           # Ecosystem fit\n  api_surface: 1.0           # Contract stability\n\ngithub:\n  enabled: true\n  auto_label: true\n  label_prefix: \"priority/\"\n  default_labels: [enhancement, feature-review]\n  priority_labels:\n    high: \"priority/high\"\n    medium: \"priority/medium\"\n    low: \"priority/low\"\n\ninventory:\n  scan_paths: [\"commands/\", \"skills/\", \"agents/\", \"src/\"]\n  exclude_patterns: [\"**/test*\", \"**/mock*\", \"**/__pycache__/**\"]\n\noutput:\n  format: markdown           # markdown | json | yaml\n  include_rationale: true\n  include_tradeoffs: true\n  max_suggestions: 10\n\nbacklog:\n  max_items: 25              # Maximum items (guardrail, cannot exceed 25)\n  stale_days: 30\n  auto_archive: false\n  file: \"docs/backlog/feature-queue.md\"\n```\n\n## Minimal Configuration Examples\n\n### Startup (Move Fast)\n\n```yaml\nversion: 1\nthresholds:\n  high_priority: 2.0\n  medium_priority: 1.0\ntradeoffs:\n  redundancy: 0.3\n  scalability: 0.5\nbacklog:\n  max_items: 15\n```\n\n### Enterprise (Stability First)\n\n```yaml\nversion: 1\nthresholds:\n  high_priority: 3.0\n  confidence_warning: 0.7\ntradeoffs:\n  api_surface: 1.5\n  redundancy: 1.2\n  readability: 1.2\n```\n\n## Project-Type Templates\n\nAdjust tradeoff weights based on project type:\n\n| Project Type | Key Weight Adjustments |\n|-------------|----------------------|\n| LLM/AI Plugin | `token_usage: 1.4`, `integration: 1.3`, `api_surface: 1.3` |\n| SaaS Product | `quality: 1.3`, `redundancy: 1.2`, `scalability: 1.3` |\n| Internal Tool | `latency: 1.3`, `integration: 1.3`, `redundancy: 0.5` |\n| Mobile App | `latency: 1.4`, `resource_usage: 1.3`, `quality: 1.3` |\n\n## Guardrails (Always Enforced)\n\nThese rules apply regardless of configuration:\n\n| Guardrail | Rule |\n|-----------|------|\n| Minimum dimensions | At least 5 tradeoff dimensions must have non-zero weight |\n| Weight sum | Weights within each category must sum to 1.0 (within 0.01) |\n| Confidence | Features below `confidence_warning` are always flagged |\n| Breaking changes | API surface changes require explicit acknowledgment |\n| Backlog limit | Maximum 25 items (forces prioritization decisions) |\n\n## Environment Variable Overrides\n\nPattern: `FEATURE_REVIEW_` and uppercase path with\nunderscores.\n\n```bash\nFEATURE_REVIEW_HIGH_PRIORITY=3.0\nFEATURE_REVIEW_GITHUB_ENABLED=false\nFEATURE_REVIEW_OUTPUT_FORMAT=json\n```\n\n## Configuration Validation\n\n```bash\n/feature-review --validate-config\n```\n\n## Inheritance and Overrides\n\n### Directory-Level Config\n\nChild configs inherit from parent and override specific\nvalues:\n\n```\nproject/\n├── .feature-review.yaml           # Project defaults\n├── plugins/\n│   └── .feature-review.yaml       # Plugin-specific overrides\n└── experimental/\n    └── .feature-review.yaml       # Experimental area config\n```\n\n### Command-Line Overrides\n\n```bash\n/feature-review --threshold.high_priority=3.0\n/feature-review --weights.value.impact=0.4\n/feature-review --github.enabled=false\n```\n\n## Migration Guide\n\n### From No Configuration\n\n1. Run `/feature-review` with defaults\n2. Review output for misaligned priorities\n3. Create minimal `.feature-review.yaml` with only\n   changed values\n\n### From Other Frameworks\n\n| Framework | Mapping Strategy |\n|-----------|-----------------|\n| RICE | Set `reach: 0.35`, `impact: 0.35`, `effort: 0.70` (cost) |\n| MoSCoW | Map to thresholds: Must (3.0), Should (2.0), Could (1.0-2.0) |\n\n## Research Enrichment\n\nConfigure external research via the tome plugin.\nWhen enabled, research findings adjust scoring factors\nwith evidence-backed deltas.\n\n```yaml\nresearch:\n  enabled: true\n  channels:\n    code_search: true          # GitHub code search\n    discourse: true            # HN, Reddit, Lobsters\n    papers: true               # arXiv, Semantic Scholar\n    triz: true                 # Cross-domain analogical reasoning\n  evidence_threshold: 0.3      # Minimum evidence to apply delta\n  max_delta: 2                 # Max Fibonacci steps adjustment\n  timeout_seconds: 120\n```\n\n| Channel | Speed | Best For |\n|---------|-------|----------|\n| code_search | Fast | Measuring ecosystem adoption |\n| discourse | Medium | Gauging community demand |\n| papers | Slow | Academic validation |\n| triz | Slow | Cross-domain innovation |\n\nWhen the tome plugin is not installed, `--research` prints\na warning and proceeds with initial scores unchanged.\n\n## Advanced Patterns\n\n### Custom Scoring Dimensions\n\n```yaml\ncustom_dimensions:\n  regulatory_compliance:\n    weight: 1.5\n    description: \"Meets GDPR/SOC2/HIPAA requirements\"\n    scoring: {5: \"Fully compliant\", 3: \"Minor gaps\", 1: \"Concerns\"}\n```\n\n### Conditional Configuration\n\nOverride weights based on feature classification:\n\n```yaml\nconditional:\n  proactive:\n    tradeoffs:\n      latency: 0.6\n      resource_usage: 1.2\n  dynamic:\n    tradeoffs:\n      redundancy: 1.2\n      scalability: 1.2\n```\n\nFile v1.9.19:modules/multi-metric-evaluation-methodology.md\n\n# Multi-Metric Evaluation Methodology\n\nHow to combine RICE, WSJF, Kano, and related models when\nprioritizing a feature backlog. Each model encodes a\ndifferent assumption about what makes a feature worth\nbuilding. This module shows the formulas, where each model\nfits, and how to combine them when no single model is\nenough on its own.\n\n## The Models at a Glance\n\n| Model | Origin | Output | Captures |\n|-------|--------|--------|----------|\n| RICE | Intercom (Sean McBride, 2017) | Number | Reach * Impact * Confidence / Effort |\n| WSJF | SAFe (Scaled Agile) | Number | (Value, Time, and Risk) / Effort |\n| Kano | Noriaki Kano (1984) | Category | Basic, Performance, Delighter, Indifferent, Reverse |\n| MoSCoW | DSDM Consortium (1994) | Bucket | Must, Should, Could, Won't |\n| Cost-of-Delay | Don Reinertsen (2009) | $ / week | Value lost per week of delay |\n\nSingle-model use is rare in practice. Most teams converge\non a hybrid: RICE for a base score, WSJF to raise time-\ncritical items, Kano to gate basics. The rest of this\nmodule explains why and how.\n\n## Model 1: RICE\n\n```\nRICE = (Reach * Impact * Confidence) / Effort\n```\n\n| Factor | Unit | Typical scale |\n|--------|------|---------------|\n| Reach | users / period | absolute count |\n| Impact | satisfaction delta | 0.25, 0.5, 1, 2, 3 |\n| Confidence | probability | 0.5, 0.8, 1.0 |\n| Effort | person-months | 0.5, 1, 2, 5, 10 |\n\n**Best for**: large user-facing roadmaps where reach is\nmeasurable and a single team can absorb most items.\n\n**Worst for**: backlogs dominated by infrastructure or\ncompliance work where \"reach\" is meaningless or every item\nshares similar reach.\n\n**Worked example**:\n\n```\nFeature: Auto-save drafts\n  Reach:      8,000 users / quarter\n  Impact:     1.0 (significant satisfaction)\n  Confidence: 0.8\n  Effort:     2 person-months\n\nRICE = (8000 * 1.0 * 0.8) / 2 = 3200\n```\n\n## Model 2: WSJF\n\nWeighted Shortest Job First. From SAFe; treats\nprioritization as a cost-of-delay optimization.\n\n```\nWSJF = Cost_of_Delay / Job_Size\n\nCost_of_Delay = User_Value + Time_Criticality + Risk_Reduction\nJob_Size      = Effort estimate\n```\n\nEach input uses a Fibonacci scale: 1, 2, 3, 5, 8, 13, 20.\n\n| Factor | Question |\n|--------|----------|\n| User_Value | How much does the user/business gain? |\n| Time_Criticality | What does delay cost? Does the value decay? |\n| Risk_Reduction | Does this open future options or de-risk? |\n| Job_Size | How much work? |\n\n**Best for**: backlogs with strong time pressure and many\nitems where deferral has measurable cost. Common in\nSAFe-aligned organizations.\n\n**Worst for**: small teams without explicit\ncost-of-delay numbers; reduces to \"gut feel times Fibonacci\".\n\n**Worked example**:\n\n```\nFeature: GDPR consent banner\n  User_Value:        5\n  Time_Criticality:  20  (regulatory deadline)\n  Risk_Reduction:    13\n  Job_Size:          3\n\nWSJF = (5 + 20 + 13) / 3 = 12.67\n```\n\nCompare with the auto-save example: WSJF would put\nauto-save at roughly (8 + 3 + 2) / 5 = 2.6, far below the\nGDPR item, even though RICE might rank them similarly.\nWSJF surfaces the deadline.\n\n## Model 3: Kano\n\nKano classifies features by user reaction, not score.\n\n| Class | If present | If absent |\n|-------|-----------|-----------|\n| Basic | Expected; no joy | Strong dissatisfaction |\n| Performance | Linear satisfaction | Linear dissatisfaction |\n| Delighter | Joy | No reaction |\n| Indifferent | No reaction | No reaction |\n| Reverse | Dissatisfaction | Satisfaction |\n\n**Source**: Noriaki Kano et al., \"Attractive Quality and\nMust-Be Quality\" (1984).\n\n**Best for**: avoiding the most common backlog mistake:\nshipping a Delighter while a Basic is still missing.\n\n**Worst for**: numeric ranking. Kano gives categories, not\nscores. Pair it with RICE or WSJF for the actual ordering.\n\n**How to classify**: present users with two questions per\nfeature:\n\n```\nFunctional:    \"How would you feel if X were present?\"\nDysfunctional: \"How would you feel if X were absent?\"\n```\n\nEach answered on a 5-point scale from \"I like it\" to \"I\ndislike it\". The answer pair maps to a Kano category via a\nfixed table (see Berger et al. 1993).\n\n## When Each Model Fits\n\n```\nBacklog has clear users and reach measurable?\n  Yes -> RICE base\n  No  -> skip RICE\n\nItems have time-critical deadlines or value decay?\n  Yes -> WSJF overlay\n  No  -> skip WSJF\n\nBacklog mixes table-stakes and aspirational features?\n  Yes -> Kano gate\n  No  -> skip Kano\n\nStakeholders need narrative buckets, not numbers?\n  Yes -> MoSCoW translation layer\n  No  -> skip MoSCoW\n```\n\n| Backlog shape | First model | Second |\n|---------------|-------------|--------|\n| Consumer product, many features | RICE | Kano gate |\n| Enterprise SaaS with deadlines | WSJF | RICE |\n| New product, no users yet | Kano and MoSCoW | RICE later |\n| Regulated domain | WSJF | Cost-of-Delay |\n| Internal tooling | RICE with reach=team_size | Kano |\n\n## The Hybrid Used in This Skill\n\nThe `feature-review` skill combines RICE-like value /\ncost ratios, WSJF time criticality, and Kano gating. The\nformula is documented in `modules/scoring-framework.md`:\n\n```\nValue = weighted_avg(Reach, Impact, Business_Value, Time_Criticality)\nCost  = weighted_avg(Effort, Risk, Complexity)\nScore = (Value / Cost) * Confidence\n```\n\nKano enters as a hard gate before scoring. Any feature\nclassified Basic that is absent today is bumped above the\nranked list. The Score then orders everything else.\n\n```\n1. Classify every feature with Kano.\n2. Pull all missing Basics to the top, ordered by user impact.\n3. Score the rest with the Value/Cost formula.\n4. Rank by Score; apply confidence multiplier.\n5. Sensitivity-check the top 10 with +/- 20% weight perturbation.\n```\n\n## Worked Example: Combining RICE, WSJF, and Kano\n\nA team scores three candidates for the next sprint.\n\n```text\nCandidates:\n  A: Auto-save drafts\n  B: GDPR consent banner\n  C: Dark mode\n\nStep 1 (Kano):\n  A: Performance  (more frequent saves = more value)\n  B: Basic        (legally required; absent today)\n  C: Delighter\n\nStep 2: Pull Basics. B is bumped to top of queue.\n\nStep 3: Score A and C with hybrid:\n  A: Value=4.75, Cost=2.67, Conf=0.8 -> 1.42\n  C: Value=2.25, Cost=2.00, Conf=0.9 -> 1.01\n\nStep 4: WSJF check on B for sizing:\n  WSJF(B) = (5 + 20 + 13) / 3 = 12.67\n  Confirms B is the largest cost-of-delay item.\n\nStep 5: Sprint order:\n  1. B (regulatory Basic)\n  2. A (Score 1.42)\n  3. C (Score 1.01)\n\nSensitivity: vary all weights by +/- 20%. Order is stable\nin 18 of 20 perturbations. Fragile case: if Time\nCriticality weight drops below 0.10, A and C swap. Action:\nkeep weight at the documented 0.20.\n```\n\n## Anti-Patterns\n\n**Single-model orthodoxy.** Picking RICE because the blog\npost said so, then shoehorning every item into a \"reach\"\nestimate that does not exist. If the model does not fit\nthe input, change the model.\n\n**Hidden recalibration.** Reweighting Impact from 1.0 to\n3.0 mid-quarter to make a favored project rank higher.\nTrack weight history in version control; flag mid-cycle\nchanges.\n\n**Confidence rubber-stamping.** Every item scored at\nConfidence 1.0. Confidence 1.0 means \"I would bet the\nquarter on this estimate\". Real backlogs cluster around\n0.5 to 0.8.\n\n**Score inflation by Fibonacci jump.** \"It feels like an 8\"\nwhen the difference between 5 and 8 should be a 60% larger\ninvestment. Force a comparison: \"Is this 60% bigger than\nthe last 5 we shipped?\"\n\n**Aggregating Kano with a number.** Kano is categorical.\nAdding \"Basic = 5, Performance = 3, Delighter = 1\" to a\nscore creates the illusion of math.\n\n**Ignoring the Pareto front.** When two items tie on\nScore but trade off on different axes (one scales reach,\none buys time), report both and let humans pick. Do not\nbreak ties with a third decimal place.\n\n## Pitfalls Specific to AI/Plugin Backlogs\n\n**Reach is a mirage.** Plugin reach is bounded by who\ninstalls the plugin, not by the addressable market. Use\n\"installed teams\" as the reach unit, not \"potential users\".\n\n**Effort underestimates evals.** A new skill is not done\nwhen the prose is written. Add the cost of subagent test\nauthoring to Effort or the score will overpromise.\n\n**Confidence collapses on token-driven features.** New\ncontext-window or prompt features cannot be confidently\nestimated until measured against real workloads. Hold\nConfidence at 0.5 until benchmarks land.\n\n**Kano Basics drift.** A Delighter (auto-completion) can\nbecome a Basic in two release cycles. Re-classify the\nBasic set quarterly.\n\n## Cross-Reference\n\nSee `modules/scoring-framework.md` for the per-factor\nscales used in this skill,\n`modules/tradeoff-dimensions.md` for the quality axes\napplied after prioritization, and\n`plugins/leyline/skills/evaluation-framework/modules/multi-metric-evaluation-methodology.md`\nfor the math behind aggregation rules.\n\nFile v1.9.19:modules/research-enrichment.md\n\n# Research Enrichment\n\nExternal evidence from the tome plugin adjusts feature-review\nscoring factors. Research findings produce deltas applied to\ninitial human assessments, not replacement scores.\n\n## Channel-to-Factor Mapping\n\nEach tome research channel maps to primary and secondary\nscoring factors:\n\n| Channel | Primary Factor | Secondary Factor | Evidence Produced |\n|---------|---------------|-----------------|-------------------|\n| code-search | Reach | Complexity | Competitor count, star counts, implementation prevalence |\n| discourse | Impact | Business Value | Sentiment score, mention volume, request frequency |\n| papers | Impact | Risk | Citation count, novelty assessment, validation level |\n| triz | Business Value | Impact | Cross-domain analogy count, inventive principle match |\n\n## Score Delta Calculation\n\nResearch findings produce an adjustment delta for each factor:\n\n```\nresearch_delta = findings_consensus * evidence_strength\n\nWhere:\n  findings_consensus: -2 to +2 (direction and magnitude)\n  evidence_strength: 0.0 to 1.0 (how reliable the findings are)\n\napplied_delta = research_delta * channel_weight\n\nIf abs(applied_delta) < evidence_threshold:\n    applied_delta = 0  (insufficient evidence, discard)\n```\n\n### Channel Weight\n\nThe channel weight reflects how directly a channel's findings\nmap to its primary factor:\n\n| Channel | Weight | Rationale |\n|---------|--------|-----------|\n| code-search | 0.8 | Star counts approximate adoption well |\n| discourse | 0.7 | Sentiment is noisy but indicative |\n| papers | 0.9 | Peer-reviewed evidence is strong |\n| triz | 0.6 | Analogies are suggestive, not conclusive |\n\n### Evidence Strength Sources\n\n| Source | Strength | When |\n|--------|----------|------|\n| > 10 independent findings | 0.8-1.0 | High-volume channels |\n| 5-10 findings | 0.5-0.8 | Moderate evidence |\n| 1-5 findings | 0.3-0.5 | Sparse evidence |\n| 0 findings | 0.0 | No evidence (discard delta) |\n\n## Fibonacci Clamping\n\nAdjusted scores must remain on the Fibonacci scale used by\nthe scoring framework: [1, 2, 3, 5, 8, 13].\n\n```python\nFIBONACCI = [1, 2, 3, 5, 8, 13]\n\ndef clamp_to_fibonacci(raw_score: float) -> int:\n    \"\"\"Clamp raw score to nearest Fibonacci value.\"\"\"\n    return min(FIBONACCI, key=lambda f: abs(f - raw_score))\n```\n\n### Clamping Rules\n\n1. Calculate `raw_adjusted = initial_score + applied_delta`\n2. Clamp to nearest Fibonacci value\n3. Result must differ from initial by at most `max_delta`\n   Fibonacci steps\n4. If the clamped result exceeds `max_delta` steps from\n   initial, use the value `max_delta` steps away\n\nExample (max_delta = 2 steps):\n- Initial: 5, raw_adjusted: 7 -> clamp: 8 (1 step away) OK\n- Initial: 3, raw_adjusted: 11 -> clamp: 8 (3 steps away)\n  exceeds max_delta -> use 13 (2 steps away from 3)\n  Wait: 13 is 4 steps from 3. So use 8 (2 steps from 3).\n  Correction: count Fibonacci index steps, not arithmetic.\n\nFibonacci indices: 1=0, 2=1, 3=2, 5=3, 8=4, 13=5\n\n```python\ndef max_delta_clamp(initial: int, target: int, max_steps: int = 2) -> int:\n    initial_idx = FIBONACCI.index(initial)\n    target_idx = FIBONACCI.index(target)\n    if abs(target_idx - initial_idx) <= max_steps:\n        return target\n    direction = 1 if target_idx > initial_idx else -1\n    return FIBONACCI[initial_idx + direction * max_steps]\n```\n\n## Graceful Degradation\n\nWhen the tome plugin is not installed or research fails:\n\n1. **Tome not installed**: Print warning, skip Phase 4.5\n   entirely. Initial scores stand unchanged.\n2. **Individual channel fails**: Continue with remaining\n   channels. Only apply deltas from successful channels.\n3. **All channels fail**: Equivalent to tome not installed.\n   Log the failure, proceed with initial scores.\n4. **Timeout exceeded**: Use whatever findings collected so\n   far. Partial results are acceptable.\n\n### Detection Protocol\n\nCheck for tome availability:\n\n1. Look for `plugins/tome/` directory in the project\n2. If not found, check for tome in the global plugin path\n3. If neither found, activate graceful degradation\n\n## Integration with tome Skill Interfaces\n\nPhase 4.5 dispatches research via tome's public skill\ninterfaces:\n\n| Step | Action | tome Skill |\n|------|--------|------------|\n| 1 | Classify the project domain | `tome:research` (domain classifier) |\n| 2 | Dispatch parallel research agents | `tome:research` (agent dispatch) |\n| 3 | Synthesize findings | `tome:synthesize` |\n| 4 | (Optional) Refine high-potential areas | `tome:dig` |\n\nThe feature-review skill invokes these via `Skill()` calls,\nnot direct Python imports. This maintains loose coupling.\n\n### Research Topic Construction\n\nFor each feature under review, construct research topics:\n\n```\ntopic = f\"{feature_name} {feature_category} plugin/tool\"\n```\n\nExample: \"auto-save drafts developer tool\" or \"token\noptimization LLM CLI\"\n\n### Synthesis Integration\n\nAfter tome returns findings, extract deltas:\n\n1. Parse synthesized report for quantitative signals\n   (star counts, mention counts, citation counts)\n2. Map quantitative signals to delta values using the\n   channel-to-factor table\n3. Extract qualitative signals (sentiment, novelty) for\n   secondary factor adjustments\n4. Apply delta calculation formula\n5. Clamp to Fibonacci scale with max_delta constraint\n\n## Output Enhancement\n\nWhen research enrichment runs, add to the feature inventory:\n\n```markdown\n## Research Evidence\n\n### Code Search (GitHub)\n- Found 12 similar implementations, avg 340 stars\n- **Reach adjustment**: +1 (broad ecosystem adoption)\n\n### Discourse (HN/Reddit)\n- 47 mentions in last 90 days, 78% positive sentiment\n- **Impact adjustment**: +1 (strong community demand)\n\n### Score Adjustments\n\n| Feature | Factor | Initial | Delta | Adjusted | Evidence |\n|---------|--------|---------|-------|----------|----------|\n| Auth    | Reach  | 5       | +1    | 8        | 3 sources |\n| Auth    | Impact | 3       | 0     | 3        | Low      |\n```\n\nFile v1.9.19:modules/scoring-framework.md\n\n# Scoring Framework\n\nHybrid prioritization combining RICE (Intercom), WSJF (SAFe), and Kano classification, grounded in Multi-Criteria Decision Analysis (MCDA) principles.\n\n## Mathematical Foundation\n\nThis framework extends standard prioritization models with MCDA best practices:\n\n- **Normalization**: Logarithmic normalization for score scales (handles non-linear value perception)\n- **Weighting**: Customizable weights with validation requirements\n- **Trade-offs**: Explicit handling through Value/Cost ratio\n- **Uncertainty**: Confidence factor adjusts for estimation risk\n- **Sensitivity**: Weight variations tested for robustness\n\n**Documentation**: See [Multi-Metric Evaluation Methodology](https://claude-night-market/plugins/abstract/skills/skills-eval/modules/multi-metric-evaluation-methodology.md) for theoretical foundations.\n\n## The Formula\n\n```\nFeature Score = (Value Score / Cost Score) * Confidence\n\nWhere:\n  Value Score = weighted_avg(Reach, Impact, Business Value, Time Criticality)\n  Cost Score = weighted_avg(Effort, Risk, Complexity)\n  Confidence = 0.0 to 1.0 (how certain are we about estimates?)\n```\n\n### Validation Requirements\n\nBefore using this framework:\n\n```yaml\nvalidation:\n  weights:\n    - Document weight derivation method (AHP, expert judgment, empirical)\n    - Verify weights sum to 1.0 within each category (value, cost)\n    - Test sensitivity to ±20% weight variations\n    - Flag critical weights that significantly change rankings\n\n  normalization:\n    - Method: \"logarithmic\" (handles non-linear perception)\n    - Rationale: \"Diminishing returns on raw scores\"\n    - Scale_invariance: \"Not required (absolute scale used)\"\n\n  uncertainty:\n    - Confidence < 0.5: Require research before commitment\n    - Document basis for confidence assessment\n    - Consider worst-case scenario for low-confidence items\n```\n\n## Value Factors\n\n### Reach (R)\n\n**Question:** How many users/use-cases does this affect?\n\n| Score | Meaning | Example |\n|-------|---------|---------|\n| 1 | Very few (<5%) | Niche admin feature |\n| 2 | Some (5-15%) | Power user feature |\n| 3 | Moderate (15-35%) | Common workflow |\n| 5 | Many (35-60%) | Core user journey |\n| 8 | Most (60-85%) | Essential feature |\n| 13 | Nearly all (>85%) | Universal need |\n\n### Impact (I)\n\n**Question:** How much does this improve the user experience?\n\n| Score | Meaning | Kano Category |\n|-------|---------|---------------|\n| 1 | Minimal improvement | Basic (expected) |\n| 2 | Slight improvement | Basic |\n| 3 | Noticeable improvement | Performance |\n| 5 | Significant improvement | Performance |\n| 8 | Major improvement | Performance |\n| 13 | Transformative | Delighter |\n\n### Business Value (BV)\n\n**Question:** How does this contribute to business goals/OKRs?\n\n| Score | Meaning | OKR Alignment |\n|-------|---------|---------------|\n| 1 | Tangential | No direct OKR connection |\n| 2 | Supporting | Supports an initiative |\n| 3 | Contributing | Contributes to Key Result |\n| 5 | Advancing | Directly advances Key Result |\n| 8 | Critical | Required for Key Result |\n| 13 | Strategic | Core to company Objective |\n\n### Time Criticality (TC)\n\n**Question:** What's the cost of delay?\n\n| Score | Meaning | Urgency |\n|-------|---------|---------|\n| 1 | Can wait indefinitely | Nice to have |\n| 2 | Can wait 6+ months | Low urgency |\n| 3 | Should do this quarter | Moderate urgency |\n| 5 | Should do this month | High urgency |\n| 8 | Should do this sprint | Very high urgency |\n| 13 | Must do immediately | Blocking/critical |\n\n## Cost Factors\n\n### Effort (E)\n\n**Question:** How much work is this?\n\n| Score | Meaning | Time Estimate |\n|-------|---------|---------------|\n| 1 | Trivial | < 1 day |\n| 2 | Small | 1-3 days |\n| 3 | Moderate | 3-5 days |\n| 5 | Large | 1-2 weeks |\n| 8 | Very large | 2-4 weeks |\n| 13 | Huge | > 1 month |\n\n### Risk (Rk)\n\n**Question:** What could go wrong?\n\n| Score | Meaning | Risk Level |\n|-------|---------|------------|\n| 1 | Very low risk | Well-understood, no dependencies |\n| 2 | Low risk | Minor unknowns |\n| 3 | Moderate risk | Some unknowns or dependencies |\n| 5 | High risk | Significant unknowns |\n| 8 | Very high risk | Many unknowns, critical dependencies |\n| 13 | Extreme risk | Uncharted territory |\n\n### Complexity (Cx)\n\n**Question:** How hard is this to build correctly?\n\n| Score | Meaning | Complexity Level |\n|-------|---------|------------------|\n| 1 | Simple | Single component, clear requirements |\n| 2 | Straightforward | Few components, clear interfaces |\n| 3 | Moderate | Multiple components, some edge cases |\n| 5 | Complex | Cross-cutting concerns, many edge cases |\n| 8 | Very complex | Architectural changes, distributed state |\n| 13 | Extremely complex | Novel algorithms, fundamental changes |\n\n## Confidence Scoring\n\nRate your confidence in the estimates:\n\n| Confidence | Meaning | When to Use |\n|------------|---------|-------------|\n| 0.9-1.0 | High | Clear requirements, similar past work |\n| 0.7-0.9 | Moderate | Some unknowns, reasonable estimates |\n| 0.5-0.7 | Low | Many unknowns, rough estimates |\n| 0.3-0.5 | Very low | Mostly guessing |\n| < 0.3 | Speculative | Requires spike/research first |\n\n**Guardrail:** Features with confidence < 0.5 should be flagged for research before commitment.\n\n## Kano Classification\n\nAfter scoring, classify the feature:\n\n### Basic (Must-Have)\n\n- Users expect this; absence causes dissatisfaction\n- Doesn't increase satisfaction when present\n- **Action:** validate these exist before anything else\n\n### Performance (Linear)\n\n- More is better; satisfaction scales with quality\n- Competitive differentiator\n- **Action:** Optimize based on ROI\n\n### Delighters (Wow Factors)\n\n- Unexpected features that create joy\n- Absence doesn't hurt; presence delights\n- **Action:** Build after basics and key performers\n\n### Indifferent\n\n- Users don't care either way\n- **Action:** Deprioritize or cut\n\n### Reverse\n\n- Feature that some users actively dislike\n- **Action:** Make optional or remove\n\n## Calculation Example\n\n```yaml\nFeature: Auto-save drafts\n\n# Value Factors\nReach: 8          # Most users write drafts\nImpact: 5         # Significant UX improvement\nBusiness Value: 3 # Supports retention KR\nTime Criticality: 3 # Should do this quarter\n\nValue Score = (8 + 5 + 3 + 3) / 4 = 4.75\n\n# Cost Factors\nEffort: 3         # 3-5 days\nRisk: 2           # Low risk, understood problem\nComplexity: 3     # Moderate, needs state management\n\nCost Score = (3 + 2 + 3) / 3 = 2.67\n\n# Confidence\nConfidence: 0.8   # Similar features built before\n\n# Final Score\nFeature Score = (4.75 / 2.67) * 0.8 = 1.42\n\n# Classification\nKano: Performance (more saving = better UX)\nPriority: Medium (1.42 is between 1.5-2.5 threshold)\n```\n\n## Interpreting Scores\n\n| Score Range | Priority | Recommendation |\n|-------------|----------|----------------|\n| > 2.5 | High | Schedule for next sprint |\n| 1.5 - 2.5 | Medium | Add to roadmap, plan timing |\n| 1.0 - 1.5 | Low | Backlog, revisit quarterly |\n| < 1.0 | Very Low | Defer indefinitely or reject |\n\n## Custom Weights\n\nDefault weights can be customized in `.feature-review.yaml`:\n\n```yaml\nweights:\n  value:\n    reach: 0.25           # Equal weighting\n    impact: 0.30          # Slightly favor user impact\n    business_value: 0.25  # Equal to reach\n    time_criticality: 0.20 # Slightly less weight\n  cost:\n    effort: 0.40          # Effort matters most\n    risk: 0.30            # Risk is significant\n    complexity: 0.30      # Complexity matters\n\n# REQUIRED: Document weight derivation\nderivation:\n  method: \"expert_judgment\"  # or \"AHP\" or \"empirical\"\n  experts: 3\n  date: \"2025-01-07\"\n  rationale: \"Impact weighted slightly higher based on user feedback\"\n```\n\n**Guardrails**:\n- Weights within each category must sum to 1.0\n- Document how weights were derived (not arbitrary)\n- Run sensitivity analysis before finalizing\n- Flag weights that cause ranking instability\n\n## Comparison with Pure RICE\n\n| Aspect | RICE | Feature Review |\n|--------|------|----------------|\n| Value factors | Reach, Impact | Reach, Impact, BV, TC |\n| Cost factors | Effort only | Effort, Risk, Complexity |\n| Time sensitivity | Not explicit | Time Criticality factor |\n| Business alignment | Not explicit | Business Value factor |\n| Uncertainty | Confidence | Confidence |\n| Classification | None | Kano model |\n| **MCDA Compliance** | Basic | **Full** (normalization, weighting, sensitivity) |\n\nFeature Review extends RICE with WSJF's time criticality and business value, plus Kano classification for strategic context, all grounded in MCDA best practices.\n\n## Sensitivity Analysis\n\nBefore committing to prioritization, test robustness:\n\n```python\ndef priority_sensitivity_analysis(features, weights, variation=0.20):\n    \"\"\"\n    Tests if rankings are stable to weight variations.\n\n    Args:\n        features: List of features with scores\n        weights: Current weight configuration\n        variation: Test ±20% changes\n\n    Returns:\n        Dict with sensitivity metrics\n    \"\"\"\n    base_ranking = rank_features(features, weights)\n    sensitivity = {}\n\n    for category in [\"value\", \"cost\"]:\n        for factor in weights[category].keys():\n            # Test weight increase\n            weights_plus = adjust_weight(weights, factor, +variation)\n            ranking_plus = rank_features(features, weights_plus)\n            correlation_plus = spearman_correlation(base_ranking, ranking_plus)\n\n            # Test weight decrease\n            weights_minus = adjust_weight(weights, factor, -variation)\n            ranking_minus = rank_features(features, weights_minus)\n            correlation_minus = spearman_correlation(base_ranking, ranking_minus)\n\n            sensitivity[factor] = {\n                \"avg_correlation\": (correlation_plus + correlation_minus) / 2,\n                \"sensitive\": correlation_plus < 0.8 or correlation_minus < 0.8\n            }\n\n    return sensitivity\n```\n\n**Interpretation**:\n- Correlation > 0.9: Ranking stable to this weight variation\n- Correlation 0.8-0.9: Moderately sensitive\n- Correlation < 0.8: Highly sensitive, weight is critical\n\nFile v1.9.19:modules/tradeoff-dimensions.md\n\n# Tradeoff Dimensions\n\nQuality attributes for evaluating features. Based on ISO 25010 software quality model, CAP/PACELC theorems, and practical engineering concerns.\n\n## Overview\n\nEvery feature makes tradeoffs. This module provides structured evaluation across nine dimensions, each with specific criteria and scoring guidance.\n\n**Scoring Scale:** 1-5 for each dimension\n\n| Score | Meaning |\n|-------|---------|\n| 1 | Poor - Significant issues |\n| 2 | Below average - Notable gaps |\n| 3 | Adequate - Meets basic needs |\n| 4 | Good - Above expectations |\n| 5 | Excellent - Best in class |\n\n## Dimension 1: Quality of Results\n\n**Question:** Does the feature deliver correct, accurate results?\n\n### Evaluation Criteria\n\n| Score | Criteria |\n|-------|----------|\n| 5 | Always correct, handles all edge cases, validated against ground truth |\n| 4 | Correct in normal cases, handles most edge cases |\n| 3 | Usually correct, some known edge case issues |\n| 2 | Occasionally incorrect, multiple edge case failures |\n| 1 | Frequently incorrect, unreliable outputs |\n\n### Considerations\n\n- **Correctness:** Does it produce right answers?\n- **Completeness:** Does it handle all expected inputs?\n- **Precision:** How accurate are numerical/search results?\n- **Recall:** Does it find everything it should?\n\n### Tradeoff Partners\n\n- Quality often trades against **Latency** (more checks = slower)\n- Quality often trades against **Resource Usage** (validation costs)\n\n---\n\n## Dimension 2: Latency\n\n**Question:** Does the feature meet timing requirements for its classification?\n\n### Evaluation Criteria by Type\n\n**For Reactive Features:**\n\n| Score | Criteria |\n|-------|----------|\n| 5 | < 50ms response, feels instant |\n| 4 | 50-100ms response, very responsive |\n| 3 | 100-300ms response, acceptable |\n| 2 | 300ms-1s response, noticeable delay |\n| 1 | > 1s response, frustrating delay |\n\n**For Proactive Features:**\n\n| Score | Criteria |\n|-------|----------|\n| 5 | Completes before needed, no user awareness |\n| 4 | Usually ready when needed |\n| 3 | Sometimes user waits briefly |\n| 2 | Often not ready, visible loading |\n| 1 | Rarely ready, defeats purpose |\n\n### Considerations\n\n- **P50 latency:** Typical case\n- **P99 latency:** Worst case (matters for reliability)\n- **Cold start:** First invocation time\n- **Warm path:** Subsequent invocations\n\n### Tradeoff Partners\n\n- Latency trades against **Quality** (PACELC theorem)\n- Latency trades against **Consistency** (eventual vs strong)\n\n---\n\n## Dimension 3: Token Usage\n\n**Question:** Is the feature context-efficient for LLM interactions?\n\n### Evaluation Criteria\n\n| Score | Criteria |\n|-------|----------|\n| 5 | Minimal tokens, highly compressed, efficient prompts |\n| 4 | Reasonable tokens, well-structured |\n| 3 | Average tokens, some verbosity |\n| 2 | High token usage, could be optimized |\n| 1 | Excessive tokens, bloated context |\n\n### Considerations\n\n- **Input tokens:** How much context needed?\n- **Output tokens:** How verbose are results?\n- **Caching potential:** Can results be reused?\n- **Streaming:** Can partial results reduce perception?\n\n### Measurement\n\n```\nToken Efficiency = Useful Output / Total Tokens\n```\n\nTarget: > 0.5 for most features\n\n### Tradeoff Partners\n\n- Token usage trades against **Quality** (more context = better results)\n- Token usage trades against **Readability** (compression reduces clarity)\n\n---\n\n## Dimension 4: Resource Usage (CPU/Memory)\n\n**Question:** Is CPU and memory consumption reasonable?\n\n### Evaluation Criteria\n\n| Score | Criteria |\n|-------|----------|\n| 5 | Minimal footprint, efficient algorithms |\n| 4 | Low resource usage, well-optimized |\n| 3 | Moderate usage, acceptable overhead |\n| 2 | High usage, performance impact on system |\n| 1 | Excessive usage, causes degradation |\n\n### Considerations\n\n- **Peak memory:** Maximum allocation\n- **Sustained memory:** Ongoing consumption\n- **CPU intensity:** Processing load\n- **I/O patterns:** Disk/network usage\n\n### Measurement Guidance\n\n| Resource | Good | Acceptable | Poor |\n|----------|------|------------|------|\n| Memory delta | < 10MB | 10-50MB | > 50MB |\n| CPU spike | < 100ms | 100-500ms | > 500ms |\n| Sustained CPU | < 5% | 5-20% | > 20% |\n\n### Tradeoff Partners\n\n- Resources trade against **Latency** (caching uses memory)\n- Resources trade against **Scalability** (per-user costs)\n\n---\n\n## Dimension 5: Redundancy (Fault Tolerance)\n\n**Question:** Does the feature handle failures gracefully?\n\n### Evaluation Criteria\n\n| Score | Criteria |\n|-------|----------|\n| 5 | Full redundancy, automatic failover, no data loss |\n| 4 | Good failover, minimal disruption |\n| 3 | Basic error handling, recoverable failures |\n| 2 | Some failure handling, may require retry |\n| 1 | No redundancy, failures cause data loss or crashes |\n\n### Considerations\n\n- **Graceful degradation:** Does it fail partially vs completely?\n- **Retry logic:** Does it handle transient failures?\n- **Data durability:** Is data protected from loss?\n- **Recovery time:** How fast to recover?\n\n### CAP Theorem Implications\n\nFor distributed features:\n- **CP systems:** May sacrifice availability for consistency\n- **AP systems:** May sacrifice consistency for availability\n\n### Tradeoff Partners\n\n- Redundancy trades against **Complexity** (more failure modes)\n- Redundancy trades against **Latency** (replication delays)\n\n---\n\n## Dimension 6: Readability (Maintainability)\n\n**Question:** Can others understand and modify this feature?\n\n### Evaluation Criteria\n\n| Score | Criteria |\n|-------|----------|\n| 5 | Self-documenting, clear abstractions, easy to extend |\n| 4 | Well-structured, good comments, learnable |\n| 3 | Understandable with effort, some complexity |\n| 2 | Hard to follow, requires tribal knowledge |\n| 1 | Opaque, only original author understands |\n\n### Considerations\n\n- **Code clarity:** Is logic obvious?\n- **Documentation:** Are complex parts explained?\n- **Naming:** Are variables/functions descriptive?\n- **Structure:** Is code well-organized?\n\n### Measurement Proxies\n\n- Cyclomatic complexity < 10\n- Functions < 50 lines\n- Clear separation of concerns\n- Test coverage > 80%\n\n### Tradeoff Partners\n\n- Readability trades against **Token Usage** (verbose = clearer)\n- Readability trades against **Resource Usage** (abstractions have cost)\n\n---\n\n## Dimension 7: Scalability\n\n**Question:** Will the feature handle 10x load?\n\n### Evaluation Criteria\n\n| Score | Criteria |\n|-------|----------|\n| 5 | Linear or sub-linear scaling, handles massive load |\n| 4 | Good scaling, handles significant growth |\n| 3 | Adequate scaling, may need attention at 5x |\n| 2 | Poor scaling, issues at 2-3x load |\n| 1 | Doesn't scale, breaks under modest increase |\n\n### Considerations\n\n- **Horizontal scaling:** Can add more instances?\n- **Vertical scaling:** Can add more resources?\n- **Bottlenecks:** Where does it fail first?\n- **State management:** How is state distributed?\n\n### Scaling Patterns\n\n| Pattern | Scalability | Complexity |\n|---------|-------------|------------|\n| Stateless | Excellent | Low |\n| Cached | Very good | Medium |\n| Sharded | Good | High |\n| Stateful single | Poor | Low |\n\n### Tradeoff Partners\n\n- Scalability trades against **Complexity** (distributed systems are hard)\n- Scalability trades against **Redundancy** (more nodes = more failure modes)\n\n---\n\n## Dimension 8: Integration (Interoperability)\n\n**Question:** Does the feature play well with existing systems?\n\n### Evaluation Criteria\n\n| Score | Criteria |\n|-------|----------|\n| 5 | smooth integration, follows all conventions, composable |\n| 4 | Good integration, minor adaptations needed |\n| 3 | Integrates with effort, some friction |\n| 2 | Difficult integration, significant workarounds |\n| 1 | Isolated, doesn't integrate without major changes |\n\n### Considerations\n\n- **API consistency:** Matches existing patterns?\n- **Data formats:** Uses standard formats?\n- **Dependencies:** Minimal coupling?\n- **Extension points:** Can be extended/wrapped?\n\n### Integration Checklist\n\n- [ ] Uses project's standard data formats\n- [ ] Follows naming conventions\n- [ ] Integrates with existing logging/metrics\n- [ ] Works with existing auth/permissions\n- [ ] Compatible with existing tooling\n\n### Tradeoff Partners\n\n- Integration trades against **Innovation** (conventions limit novelty)\n- Integration trades against **Optimization** (generic > specialized)\n\n---\n\n## Dimension 9: API Surface\n\n**Question:** Is the API backward compatible and well-designed?\n\n### Evaluation Criteria\n\n| Score | Criteria |\n|-------|----------|\n| 5 | Stable API, versioned, excellent documentation, no breaking changes |\n| 4 | Good API, rare breaking changes with migration path |\n| 3 | Adequate API, occasional breaking changes |\n| 2 | Unstable API, frequent breaking changes |\n| 1 | No stable API, constant churn |\n\n### Considerations\n\n- **Breaking changes:** How often do consumers break?\n- **Deprecation policy:** Are changes communicated?\n- **Versioning:** Is there a versioning strategy?\n- **Documentation:** Is the contract clear?\n\n### API Design Checklist\n\n- [ ] Additive changes only (no removal)\n- [ ] Optional new fields with defaults\n- [ ] Deprecation warnings before removal\n- [ ] Semantic versioning\n- [ ] Clear error contracts\n\n### Tradeoff Partners\n\n- API stability trades against **Innovation** (can't change freely)\n- API stability trades against **Quality** (may keep suboptimal designs)\n\n---\n\n## Composite Scoring\n\nCalculate overall tradeoff score:\n\n```\nTradeoff Score = Σ(dimension_score * dimension_weight) / Σ(weights)\n```\n\n### Default Weights\n\n| Dimension | Default Weight | Rationale |\n|-----------|---------------|-----------|\n| Quality | 1.0 | Core requirement |\n| Latency | 1.0 | User experience |\n| Token Usage | 1.0 | LLM efficiency |\n| Resource Usage | 0.8 | Important but secondary |\n| Redundancy | 0.5 | Context-dependent |\n| Readability | 1.0 | Maintainability |\n| Scalability | 0.8 | Future-proofing |\n| Integration | 1.0 | Ecosystem fit |\n| API Surface | 1.0 | Contract stability |\n\n### Guardrail\n\n**Minimum 5 dimensions must be evaluated.** Cannot skip all tradeoff analysis.\n\n---\n\n## Using Tradeoffs in Review\n\n### For Existing Features\n\n1. Score each dimension\n2. Identify dimensions below 3\n3. Determine if improvement is feasible\n4. Prioritize based on impact\n\n### For Proposed Features\n\n1. Estimate scores for each dimension\n2. Compare against existing features\n3. Identify which tradeoffs are acceptable\n4. Document accepted tradeoffs explicitly\n\n### Red Flag Combinations\n\n| Pattern | Concern | Action |\n|---------|---------|--------|\n| High Quality and High Latency | May frustrate users | Optimize or classify as Proactive |\n| Low Readability and Low API Surface | Maintenance nightmare | Refactor before extending |\n| High Token and Low Quality | Wasteful | Optimize prompts |\n| Low Redundancy and High Integration | Cascading failures | Add fault tolerance |\n\nFile v1.9.19:skill-card.md\n\n## Description:\n\nScores backlog items with RICE/WSJF/Kano and files GitHub issues for top candidates.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[athola](https://clawhub.ai/user/athola)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and product teams use this skill to inventory existing features, prioritize backlog candidates with hybrid RICE/WSJF/Kano scoring, and turn accepted high-priority suggestions into GitHub issues.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The deferred-capture step can automatically run a local Python script for skipped high-scoring suggestions.\n\nMitigation: Require explicit user confirmation before running deferred capture, use a trusted bundled helper, and pass arguments safely without shell interpolation.\n\nRisk: GitHub issue creation and research enrichment can contact external services when requested.\n\nMitigation: Review requested integrations before use and confirm credentials, network access, and data-sharing expectations.\n\nRisk: Backlog scores can be misleading if confidence is low or scoring weights are not validated.\n\nMitigation: Flag low-confidence items, verify weight sums, and sensitivity-check top candidates before committing roadmap decisions.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/athola/skills/nm-imbue-feature-review)\n- [Clawdis homepage](https://github.com/athola/claude-night-market/tree/master/plugins/imbue)\n- [Scoring framework](modules/scoring-framework.md)\n- [Classification system](modules/classification-system.md)\n- [Tradeoff dimensions](modules/tradeoff-dimensions.md)\n- [Research enrichment](modules/research-enrichment.md)\n- [Configuration](modules/configuration.md)\n- [Multi-metric evaluation methodology](modules/multi-metric-evaluation-methodology.md)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, shell commands, configuration, guidance]\n\n**Output Format:** [Markdown reports, tables, GitHub issue drafts, and optional shell commands]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Can include feature inventories, score tables, research evidence summaries, suggestion reports, and issue creation steps.]\n\n## Skill Version(s):\n\n1.9.19 (source: server release evidence)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nArchive v1.9.17: 9 files, 28336 bytes\n\nFiles: modules/classification-system.md (8999b), modules/configuration.md (6593b), modules/multi-metric-evaluation-methodology.md (8759b), modules/research-enrichment.md (5903b), modules/scoring-framework.md (10049b), modules/tradeoff-dimensions.md (10886b), skill-card.md (2744b), SKILL.md (12170b), _meta.json (143b)\n\nFile v1.9.17:SKILL.md\n\n---\nname: feature-review\ndescription: |\n  Scores backlog items with RICE/WSJF/Kano and files GitHub issues for top candidates\nversion: 1.9.8\ntriggers:\n  - feature-prioritization\n  - backlog-triage\n  - RICE\n  - WSJF\n  - Kano\n  - roadmap\n  - triaging a roadmap or prioritizing features for a sprint\nmetadata: {\"openclaw\": {\"homepage\": \"https://github.com/athola/claude-night-market/tree/master/plugins/imbue\", \"emoji\": \"\\ud83e\\udd9e\", \"requires\": {\"config\": [\"night-market.imbue:scope-guard\"]}}}\nsource: claude-night-market\nsource_plugin: imbue\n---\n\n> **Night Market Skill** — ported from [claude-night-market/imbue](https://github.com/athola/claude-night-market/tree/master/plugins/imbue). For the full experience with agents, hooks, and commands, install the Claude Code plugin.\n\n\n## Table of Contents\n\n- [Philosophy](#philosophy)\n- [When to Use](#when-to-use)\n- [When NOT to Use](#when-not-to-use)\n- [Quick Start](#quick-start)\n- [1. Inventory Current Features](#1-inventory-current-features)\n- [2. Score and Classify](#2-score-and-classify)\n- [3. Generate Suggestions](#3-generate-suggestions)\n\n## Verification\n\nRun `make test-feature-review` to verify scoring logic after changes.\n- [4. Upload to GitHub](#4-upload-to-github)\n- [Workflow](#workflow)\n- [Phase 1: Feature Discovery (`feature-review:inventory-complete`)](#phase-1:-feature-discovery-(feature-review:inventory-complete))\n- [Phase 2: Classification (`feature-review:classified`)](#phase-2:-classification-(feature-review:classified))\n- [Phase 3: Scoring (`feature-review:scored`)](#phase-3:-scoring-(feature-review:scored))\n- [Phase 4: Tradeoff Analysis (`feature-review:tradeoffs-analyzed`)](#phase-4:-tradeoff-analysis-(feature-review:tradeoffs-analyzed))\n- [Phase 5: Gap Analysis & Suggestions (`feature-review:suggestions-generated`)](#phase-5:-gap-analysis-&-suggestions-(feature-review:suggestions-generated))\n- [Phase 6: GitHub Integration (`feature-review:issues-created`)](#phase-6:-github-integration-(feature-review:issues-created))\n- [Configuration](#configuration)\n- [Configuration File](#configuration-file)\n- [Guardrails](#guardrails)\n- [Required TodoWrite Items](#required-todowrite-items)\n- [Integration Points](#integration-points)\n- [Output Format](#output-format)\n- [Feature Inventory Table](#feature-inventory-table)\n- [Suggestion Report](#suggestion-report)\n- [Feature Suggestions](#feature-suggestions)\n- [High Priority (Score > 2.5)](#high-priority-(score->-25))\n- [Related Skills](#related-skills)\n- [Reference](#reference)\n\n\n# Feature Review\n\nReview implemented features and suggest new ones using evidence-based prioritization. Create GitHub issues for accepted suggestions.\n\n## Philosophy\n\nFeature decisions rely on data. Every feature involves tradeoffs that require evaluation. This skill uses hybrid RICE+WSJF scoring with Kano classification to prioritize work and generates actionable GitHub issues for accepted suggestions.\n\n## When To Use\n\n- Roadmap reviews (sprint planning, quarterly reviews).\n- Retrospective evaluations.\n- Planning new development cycles.\n\n## When NOT To Use\n\n- Emergency bug fixes.\n- Simple documentation updates.\n- Active implementation (use `scope-guard`).\n\n## Quick Start\n\n### 1. Inventory Current Features\n\nDiscover and categorize existing features:\n```bash\n/feature-review --inventory\n```\n\n### 2. Score and Classify\n\nEvaluate features against the prioritization framework:\n```bash\n/feature-review\n```\n\n### 3. Generate Suggestions\n\nReview gaps and suggest new features:\n```bash\n/feature-review --suggest\n```\n\n### 4. Research-Enriched Scoring\n\nUse tome plugin to adjust scores with external evidence:\n```bash\n/feature-review --research\n```\n\n### 5. Upload to GitHub\n\nCreate issues for accepted suggestions:\n```bash\n/feature-review --suggest --create-issues\n```\n\n## Workflow\n\n### Phase 1: Feature Discovery (`feature-review:inventory-complete`)\n\nIdentify features by analyzing:\n\n1. **Code artifacts**: Entry points, public APIs, and configuration surfaces.\n2. **Documentation**: README lists, CHANGELOG entries, and user docs.\n3. **Git history**: Recent feature commits and branches.\n\n**Output:** Feature inventory table.\n\n### Phase 2: Classification (`feature-review:classified`)\n\nClassify each feature along two axes:\n\n**Axis 1: Proactive vs Reactive**\n\n| Type | Definition | Examples |\n|------|------------|----------|\n| **Proactive** | Anticipates user needs. | Suggestions, prefetching. |\n| **Reactive** | Responds to explicit input. | Form handling, click actions. |\n\n**Axis 2: Static vs Dynamic**\n\n| Type | Update Pattern | Storage Model |\n|------|---------------|---------------|\n| **Static** | Incremental, versioned. | File-based, cached. |\n| **Dynamic** | Continuous, streaming. | Database, real-time. |\n\nSee [classification-system.md](modules/classification-system.md) for details.\n\n### Phase 3: Scoring (`feature-review:scored`)\n\nApply hybrid RICE+WSJF scoring:\n\n```\nFeature Score = Value Score / Cost Score\n\nValue Score = (Reach + Impact + Business Value + Time Criticality) / 4\nCost Score = (Effort + Risk + Complexity) / 3\n\nAdjusted Score = Feature Score * Confidence\n```\n\n**Scoring Scale:** Fibonacci (1, 2, 3, 5, 8, 13).\n\n**Thresholds:**\n- **> 2.5**: High priority.\n- **1.5 - 2.5**: Medium priority.\n- **< 1.5**: Low priority.\n\nSee [scoring-framework.md](modules/scoring-framework.md) for the framework.\nSee [multi-metric-evaluation-methodology.md](modules/multi-metric-evaluation-methodology.md)\nwhen one model is not enough: it covers how to combine\nRICE, WSJF, and Kano, where each model fits, and how to\nreconcile conflicting signals.\n\n### Phase 4: Tradeoff Analysis (`feature-review:tradeoffs-analyzed`)\n\nEvaluate each feature across quality dimensions:\n\n| Dimension | Question | Scale |\n|-----------|----------|-------|\n| **Quality** | Does it deliver correct results? | 1-5 |\n| **Latency** | Does it meet timing requirements? | 1-5 |\n| **Token Usage** | Is it context-efficient? | 1-5 |\n| **Resource Usage** | Is CPU/memory reasonable? | 1-5 |\n| **Redundancy** | Does it handle failures gracefully? | 1-5 |\n| **Readability** | Can others understand it? | 1-5 |\n| **Scalability** | Will it handle 10x load? | 1-5 |\n| **Integration** | Does it play well with others? | 1-5 |\n| **API Surface** | Is it backward compatible? | 1-5 |\n\nSee [tradeoff-dimensions.md](modules/tradeoff-dimensions.md) for criteria.\n\n### Phase 4.5: Research Enrichment (`feature-review:research-enriched`)\n\n**Triggered by:** `--research` flag. Requires tome plugin.\n\nUse tome's multi-source research to adjust scoring factors\nwith external evidence. This phase runs between tradeoff\nanalysis and gap analysis.\n\n1. **Dispatch research**: For each feature, construct\n   research topics and dispatch tome channels (code-search,\n   discourse, papers, triz) in parallel.\n2. **Synthesize findings**: Merge results across channels\n   using `tome:synthesize`.\n3. **Calculate deltas**: Map findings to scoring factor\n   adjustments using channel-to-factor mapping.\n4. **Apply deltas**: Adjust initial scores by research\n   deltas, clamp to Fibonacci scale, respect max_delta.\n5. **Present evidence**: Show adjustment table with\n   evidence sources and rationale.\n\nSee [research-enrichment.md](modules/research-enrichment.md)\nfor the full enrichment protocol, delta calculation, and\ngraceful degradation behavior.\n\n**Graceful degradation**: If tome is not installed, prints\na warning and proceeds with initial scores unchanged.\n\n### Phase 5: Gap Analysis & Suggestions (`feature-review:suggestions-generated`)\n\n1. **Identify gaps**: Missing Kano basics.\n2. **Surface opportunities**: High-value, low-effort features.\n3. **Flag technical debt**: Features with declining scores.\n4. **Recommend actions**: Build, improve, deprecate, or maintain.\n\n### Phase 6: GitHub Integration (`feature-review:issues-created`)\n\n1. Generate issue title and body from suggestions.\n2. Apply labels (feature, enhancement, priority/*).\n3. Link to related issues.\n4. Confirm with user before creation.\n\n**Deferred capture for high-scoring suggestions:**\nAfter the user confirms which suggestions to act on, any\nhigh-scoring suggestion (score > 2.5) that is not acted on\nshould be preserved as a deferred item.\nRun once per skipped high-scoring suggestion:\n\n```bash\npython3 scripts/deferred_capture.py \\\n  --title \"<suggestion title>\" \\\n  --source feature-review \\\n  --context \"RICE score: <score>. <description>\"\n```\n\nThis runs automatically without prompting the user.\nSuggestions with scores of 2.5 or below do not need\nto be captured.\n\n## Configuration\n\nFeature-review uses opinionated defaults but allows customization.\n\n### Configuration File\n\nCreate `.feature-review.yaml` in project root:\n\n```yaml\n# .feature-review.yaml\nversion: 1.9.3\n\n# Scoring weights (must sum to 1.0)\nweights:\n  value:\n    reach: 0.25\n    impact: 0.30\n    business_value: 0.25\n    time_criticality: 0.20\n  cost:\n    effort: 0.40\n    risk: 0.30\n    complexity: 0.30\n\n# Score thresholds\nthresholds:\n  high_priority: 2.5\n  medium_priority: 1.5\n\n# Tradeoff dimension weights (0.0 to disable)\ntradeoffs:\n  quality: 1.0\n  latency: 1.0\n  token_usage: 1.0\n  resource_usage: 0.8\n  redundancy: 0.5\n  readability: 1.0\n  scalability: 0.8\n  integration: 1.0\n  api_surface: 1.0\n```\n\nSee [configuration.md](modules/configuration.md) for options.\n\n### Guardrails\n\nThese rules apply to all configurations:\n\n1. **Minimum dimensions**: Evaluate at least 5 tradeoff dimensions.\n2. **Confidence requirement**: Review scores below 50% confidence.\n3. **Breaking change warning**: Require acknowledgment for API surface changes.\n4. **Backlog limit**: Limit suggestion queue to 25 items.\n\n## Required TodoWrite Items\n\n1. `feature-review:inventory-complete`\n2. `feature-review:classified`\n3. `feature-review:scored`\n4. `feature-review:tradeoffs-analyzed`\n5. `feature-review:research-enriched` (if `--research`)\n6. `feature-review:suggestions-generated`\n7. `feature-review:issues-created` (if requested)\n\n## Integration Points\n\n- **`imbue:scope-guard`**: Provides Worthiness Scores for suggestions.\n- **`sanctum:do-issue`**: Prioritizes issues with high scores.\n- **`superpowers:brainstorming`**: Evaluates new ideas against existing features.\n- **`tome:research`**: Multi-source research for score enrichment (optional, `--research`).\n\n## Output Format\n\n### Feature Inventory Table\n\n```markdown\n| Feature | Type | Data | Score | Priority | Status |\n|---------|------|------|-------|----------|--------|\n| Auth middleware | Reactive | Dynamic | 2.8 | High | Stable |\n| Skill loader | Reactive | Static | 2.3 | Medium | Needs improvement |\n```\n\n### Research-Enriched Table (with `--research`)\n\n```markdown\n| Feature | Type | Score | Adj. | Priority | Evidence |\n|---------|------|-------|------|----------|----------|\n| Auth    | R/D  | 2.8   | 3.1  | High     | 3 sources |\n| Loader  | R/S  | 2.3   | 2.3  | Medium   | none      |\n\n## Research Evidence\n\n### Code Search (GitHub)\n- 12 implementations, avg 340 stars\n- **Reach**: +1 (broad adoption)\n\n### Discourse (HN/Reddit)\n- 47 mentions, 78% positive\n- **Impact**: +1 (strong demand)\n```\n\n### Suggestion Report\n\n```markdown\n## Feature Suggestions\n\n### High Priority (Score > 2.5)\n\n1. **[Feature Name]** (Score: 2.7)\n   - Classification: Proactive/Dynamic\n   - Value: High reach\n   - Cost: Moderate effort\n   - Recommendation: Build in next sprint\n```\n\n## Related Skills\n\n- `imbue:scope-guard`: Prevent overengineering.\n- `sanctum:pr-review`: Code-level review (different scope: this\n  skill prioritizes feature ideas, pr-review reviews diffs).\n\n## Reference\n\n- **[scoring-framework.md](modules/scoring-framework.md)**: RICE+WSJF hybrid.\n- **[classification-system.md](modules/classification-system.md)**: Axes definition.\n- **[tradeoff-dimensions.md](modules/tradeoff-dimensions.md)**: Quality attributes.\n- **[research-enrichment.md](modules/research-enrichment.md)**: tome-driven score deltas, channel-to-factor mapping, graceful degradation.\n- **[multi-metric-evaluation-methodology.md](modules/multi-metric-evaluation-methodology.md)**: Combining RICE, WSJF, and Kano when no single model suffices.\n- **[configuration.md](modules/configuration.md)**: Customization options.\n\nFile v1.9.17:_meta.json\n\n{\n  \"ownerId\": \"kn7d107jg9jv602h9ytsegydq184a42s\",\n  \"slug\": \"nm-imbue-feature-review\",\n  \"version\": \"1.9.17\",\n  \"publishedAt\": 1785389648720\n}\n\nFile v1.9.17:modules/classification-system.md\n\n# Classification System\n\nFeatures are classified along two orthogonal axes that determine architectural and UX implications.\n\n## Axis 1: Proactive vs Reactive\n\nThis axis describes **when** the feature acts relative to user intent.\n\n### Proactive Features\n\n**Definition:** Anticipates user needs and acts before explicit request.\n\n**Characteristics:**\n- Runs in background or ahead of user action\n- Requires prediction/inference of user intent\n- May consume resources speculatively\n- Higher latency tolerance (users don't wait)\n\n**Latency Tolerance:**\n- Background processing acceptable (seconds to minutes)\n- User doesn't perceive delay directly\n- Can be batched or deferred\n\n**Examples:**\n| Feature | How It's Proactive |\n|---------|-------------------|\n| Auto-save | Saves before user requests |\n| Prefetching | Loads data before navigation |\n| Suggestions | Offers options before user types |\n| Health checks | Monitors before problems occur |\n| Cache warming | Prepares data before access |\n\n**Tradeoffs:**\n| Pro | Con |\n|-----|-----|\n| Reduces user effort | May waste resources |\n| Feels \"smart\" | Can be wrong/intrusive |\n| Prevents problems | Requires more data |\n| Smoother UX | Higher complexity |\n\n**Architecture Patterns:**\n- Event-driven / pub-sub\n- Background workers\n- Predictive models\n- Eventual consistency acceptable\n\n### Reactive Features\n\n**Definition:** Responds to explicit user input or system events.\n\n**Characteristics:**\n- Triggered by user action\n- Must feel immediate\n- Resources used on-demand\n- Correctness over speculation\n\n**Latency Tolerance:**\n- Sub-100ms for UI feedback\n- Sub-1s for completion\n- User actively waiting\n\n**Examples:**\n| Feature | How It's Reactive |\n|---------|------------------|\n| Form submission | User clicks submit |\n| Search | User types query |\n| Navigation | User clicks link |\n| Validation | User enters input |\n| Commands | User invokes action |\n\n**Tradeoffs:**\n| Pro | Con |\n|-----|-----|\n| User in control | User must initiate |\n| Predictable behavior | No anticipation |\n| Lower resource waste | Perceived latency |\n| Simpler to implement | Less \"magical\" UX |\n\n**Architecture Patterns:**\n- Request/response\n- Synchronous processing\n- Strong consistency\n- Direct invocation\n\n### Classification Decision Tree\n\n```\nIs the feature triggered by explicit user action?\n├── Yes → Is immediate response critical?\n│   ├── Yes → REACTIVE\n│   └── No → Could be either (consider UX goals)\n└── No → Does it require user data/context?\n    ├── Yes → PROACTIVE (with data)\n    └── No → PROACTIVE (autonomous)\n```\n\n## Axis 2: Static vs Dynamic\n\nThis axis describes **how** feature data changes over time.\n\n### Static Features\n\n**Definition:** Data changes incrementally through explicit updates.\n\n**Characteristics:**\n- Version-controlled or release-based updates\n- Can be cached aggressively\n- Deterministic lookups\n- Stale data possible but predictable\n\n**Update Pattern:**\n- Deploy-time updates\n- Batch processing\n- Periodic refresh\n- Manual triggers\n\n**Storage Models:**\n| Model | Use Case |\n|-------|----------|\n| Files | Configuration, templates |\n| Embedded | Constants, schemas |\n| CDN | Assets, documentation |\n| Read replicas | Reference data |\n\n**Lookup Cost:** O(1) or O(log n), highly cacheable\n\n**Examples:**\n| Feature | Why It's Static |\n|---------|----------------|\n| Skill definitions | Updated via deploy |\n| Documentation | Published versions |\n| Configuration | Changed by admin |\n| Templates | Version-controlled |\n| Schema definitions | Release-based |\n\n**Tradeoffs:**\n| Pro | Con |\n|-----|-----|\n| Fast lookups | Can be stale |\n| Simple architecture | Update lag |\n| Highly cacheable | Deployment required |\n| Predictable performance | Less responsive |\n\n### Dynamic Features\n\n**Definition:** Data changes continuously through ongoing operations.\n\n**Characteristics:**\n- Real-time or near-real-time updates\n- Limited caching opportunity\n- Query-based lookups\n- Consistency challenges\n\n**Update Pattern:**\n- Event-driven updates\n- Streaming ingestion\n- Live queries\n- Continuous sync\n\n**Storage Models:**\n| Model | Use Case |\n|-------|----------|\n| Database | Transactional data |\n| Cache layer | Hot data |\n| Stream | Events, logs |\n| Search index | Queryable content |\n\n**Lookup Cost:** O(log n) to O(n), cache hit-rate varies\n\n**Examples:**\n| Feature | Why It's Dynamic |\n|---------|-----------------|\n| User sessions | Real-time state |\n| Search results | Live queries |\n| Notifications | Streaming events |\n| Analytics | Continuous ingestion |\n| Collaboration | Multi-user sync |\n\n**Tradeoffs:**\n| Pro | Con |\n|-----|-----|\n| Always fresh | Higher latency |\n| Responsive to change | Complex architecture |\n| Real-time capable | Consistency challenges |\n| User-specific | Harder to cache |\n\n### Classification Decision Tree\n\n```\nDoes the data change based on user actions in real-time?\n├── Yes → DYNAMIC\n└── No → Is freshness critical (< 1 hour)?\n    ├── Yes → DYNAMIC\n    └── No → Could the data be served from cache/CDN?\n        ├── Yes → STATIC\n        └── No → Consider hybrid (static + refresh)\n```\n\n## The 2x2 Matrix\n\nCombining both axes creates four feature archetypes:\n\n```\n                    STATIC                 DYNAMIC\n              ┌─────────────────────┬─────────────────────┐\n              │                     │                     │\n   PROACTIVE  │   Predictive Cache  │   Smart Assistant   │\n              │   (prefetch static) │   (live suggestions)│\n              │                     │                     │\n              │   Latency: Low      │   Latency: Medium   │\n              │   Complexity: Low   │   Complexity: High  │\n              │                     │                     │\n              ├─────────────────────┼─────────────────────┤\n              │                     │                     │\n   REACTIVE   │   Reference Lookup  │   Interactive Query │\n              │   (docs, configs)   │   (search, forms)   │\n              │                     │                     │\n              │   Latency: Very Low │   Latency: Low      │\n              │   Complexity: Low   │   Complexity: Medium│\n              │                     │                     │\n              └─────────────────────┴─────────────────────┘\n```\n\n### Archetype Details\n\n#### Predictive Cache (Proactive and Static)\n\n- **Example:** Prefetching documentation pages\n- **Pattern:** Background worker loads static assets\n- **Complexity:** Low - just scheduling and caching\n- **Risk:** Wasted bandwidth if prediction wrong\n\n#### Smart Assistant (Proactive and Dynamic)\n\n- **Example:** AI-powered suggestions based on context\n- **Pattern:** Real-time inference on streaming data\n- **Complexity:** High - ML models, data pipelines\n- **Risk:** Expensive, can be wrong, privacy concerns\n\n#### Reference Lookup (Reactive and Static)\n\n- **Example:** Loading skill definitions\n- **Pattern:** Cache-first, fallback to file\n- **Complexity:** Low - simple read operations\n- **Risk:** Stale data if cache not invalidated\n\n#### Interactive Query (Reactive and Dynamic)\n\n- **Example:** Search across current repository\n- **Pattern:** Query on demand, may use indexes\n- **Complexity:** Medium - query optimization, indexing\n- **Risk:** Variable latency, consistency windows\n\n## Classification for Common Features\n\n| Feature Type | Typical Classification | Notes |\n|--------------|----------------------|-------|\n| CLI Commands | Reactive and Static | User-invoked, defined behavior |\n| Auto-complete | Proactive and Dynamic | Predicts input from context |\n| Configuration | Reactive and Static | Loaded on demand, versioned |\n| Session state | Reactive and Dynamic | User-driven, real-time |\n| Caching layer | Proactive and Static | Anticipates access patterns |\n| Notifications | Proactive and Dynamic | Pushed based on events |\n| Validation | Reactive and Static | Rules are static, input is dynamic |\n| Analytics | Proactive and Dynamic | Background collection |\n\n## Using Classification in Review\n\nWhen reviewing features:\n\n1. **Identify current classification** - What is it today?\n2. **Evaluate fit** - Does classification match use case?\n3. **Consider migration** - Would different classification improve UX?\n4. **Note tradeoffs** - What would change with different classification?\n\n**Red Flags:**\n- Reactive feature with high latency → Consider proactive alternative\n- Dynamic feature rarely changing → Could be static for performance\n- Proactive feature often wrong → Consider reactive fallback\n- Static feature causing staleness issues → Consider dynamic refresh\n\nFile v1.9.17:modules/configuration.md\n\n# Configuration\n\nFeature-review uses opinionated defaults but allows\nproject-specific customization through a YAML\nconfiguration file.\n\n## Configuration File Location\n\nCreate `.feature-review.yaml` in your project root:\n\n```\nproject/\n├── .feature-review.yaml    # Configuration file\n├── src/\n└── ...\n```\n\n## Full Configuration Schema\n\n```yaml\n# .feature-review.yaml\n# All values shown are defaults - only specify what you want to change\n\nversion: 1  # Schema version (required if file exists)\n\nweights:\n  value:\n    reach: 0.25              # How many users affected\n    impact: 0.30             # How much improvement per user\n    business_value: 0.25     # OKR/strategic alignment\n    time_criticality: 0.20   # Cost of delay\n  cost:\n    effort: 0.40             # Development time\n    risk: 0.30               # Uncertainty/unknowns\n    complexity: 0.30         # Technical difficulty\n\nthresholds:\n  high_priority: 2.5         # Score > 2.5 = implement soon\n  medium_priority: 1.5       # Score > 1.5 = roadmap candidate\n  confidence_warning: 0.5    # Scores below this get flagged\n\nclassification:\n  default_type: reactive     # proactive | reactive\n  default_data: static       # static | dynamic\n  patterns:\n    proactive_patterns: [\"*auto*\", \"*suggest*\", \"*predict*\", \"*prefetch*\"]\n    dynamic_patterns: [\"*session*\", \"*realtime*\", \"*live*\", \"*stream*\"]\n\ntradeoffs:\n  quality: 1.0               # Correctness of results\n  latency: 1.0               # Response time\n  token_usage: 1.0           # Context efficiency (LLM-specific)\n  resource_usage: 0.8        # CPU/memory consumption\n  redundancy: 0.5            # Fault tolerance\n  readability: 1.0           # Code maintainability\n  scalability: 0.8           # Growth handling\n  integration: 1.0           # Ecosystem fit\n  api_surface: 1.0           # Contract stability\n\ngithub:\n  enabled: true\n  auto_label: true\n  label_prefix: \"priority/\"\n  default_labels: [enhancement, feature-review]\n  priority_labels:\n    high: \"priority/high\"\n    medium: \"priority/medium\"\n    low: \"priority/low\"\n\ninventory:\n  scan_paths: [\"commands/\", \"skills/\", \"agents/\", \"src/\"]\n  exclude_patterns: [\"**/test*\", \"**/mock*\", \"**/__pycache__/**\"]\n\noutput:\n  format: markdown           # markdown | json | yaml\n  include_rationale: true\n  include_tradeoffs: true\n  max_suggestions: 10\n\nbacklog:\n  max_items: 25              # Maximum items (guardrail, cannot exceed 25)\n  stale_days: 30\n  auto_archive: false\n  file: \"docs/backlog/feature-queue.md\"\n```\n\n## Minimal Configuration Examples\n\n### Startup (Move Fast)\n\n```yaml\nversion: 1\nthresholds:\n  high_priority: 2.0\n  medium_priority: 1.0\ntradeoffs:\n  redundancy: 0.3\n  scalability: 0.5\nbacklog:\n  max_items: 15\n```\n\n### Enterprise (Stability First)\n\n```yaml\nversion: 1\nthresholds:\n  high_priority: 3.0\n  confidence_warning: 0.7\ntradeoffs:\n  api_surface: 1.5\n  redundancy: 1.2\n  readability: 1.2\n```\n\n## Project-Type Templates\n\nAdjust tradeoff weights based on project type:\n\n| Project Type | Key Weight Adjustments |\n|-------------|----------------------|\n| LLM/AI Plugin | `token_usage: 1.4`, `integration: 1.3`, `api_surface: 1.3` |\n| SaaS Product | `quality: 1.3`, `redundancy: 1.2`, `scalability: 1.3` |\n| Internal Tool | `latency: 1.3`, `integration: 1.3`, `redundancy: 0.5` |\n| Mobile App | `latency: 1.4`, `resource_usage: 1.3`, `quality: 1.3` |\n\n## Guardrails (Always Enforced)\n\nThese rules apply regardless of configuration:\n\n| Guardrail | Rule |\n|-----------|------|\n| Minimum dimensions | At least 5 tradeoff dimensions must have non-zero weight |\n| Weight sum | Weights within each category must sum to 1.0 (within 0.01) |\n| Confidence | Features below `confidence_warning` are always flagged |\n| Breaking changes | API surface changes require explicit acknowledgment |\n| Backlog limit | Maximum 25 items (forces prioritization decisions) |\n\n## Environment Variable Overrides\n\nPattern: `FEATURE_REVIEW_` and uppercase path with\nunderscores.\n\n```bash\nFEATURE_REVIEW_HIGH_PRIORITY=3.0\nFEATURE_REVIEW_GITHUB_ENABLED=false\nFEATURE_REVIEW_OUTPUT_FORMAT=json\n```\n\n## Configuration Validation\n\n```bash\n/feature-review --validate-config\n```\n\n## Inheritance and Overrides\n\n### Directory-Level Config\n\nChild configs inherit from parent and override specific\nvalues:\n\n```\nproject/\n├── .feature-review.yaml           # Project defaults\n├── plugins/\n│   └── .feature-review.yaml       # Plugin-specific overrides\n└── experimental/\n    └── .feature-review.yaml       # Experimental area config\n```\n\n### Command-Line Overrides\n\n```bash\n/feature-review --threshold.high_priority=3.0\n/feature-review --weights.value.impact=0.4\n/feature-review --github.enabled=false\n```\n\n## Migration Guide\n\n### From No Configuration\n\n1. Run `/feature-review` with defaults\n2. Review output for misaligned priorities\n3. Create minimal `.feature-review.yaml` with only\n   changed values\n\n### From Other Frameworks\n\n| Framework | Mapping Strategy |\n|-----------|-----------------|\n| RICE | Set `reach: 0.35`, `impact: 0.35`, `effort: 0.70` (cost) |\n| MoSCoW | Map to thresholds: Must (3.0), Should (2.0), Could (1.0-2.0) |\n\n## Research Enrichment\n\nConfigure external research via the tome plugin.\nWhen enabled, research findings adjust scoring factors\nwith evidence-backed deltas.\n\n```yaml\nresearch:\n  enabled: true\n  channels:\n    code_search: true          # GitHub code search\n    discourse: true            # HN, Reddit, Lobsters\n    papers: true               # arXiv, Semantic Scholar\n    triz: true                 # Cross-domain analogical reasoning\n  evidence_threshold: 0.3      # Minimum evidence to apply delta\n  max_delta: 2                 # Max Fibonacci steps adjustment\n  timeout_seconds: 120\n```\n\n| Channel | Speed | Best For |\n|---------|-------|----------|\n| code_search | Fast | Measuring ecosystem adoption |\n| discourse | Medium | Gauging community demand |\n| papers | Slow | Academic validation |\n| triz | Slow | Cross-domain innovation |\n\nWhen the tome plugin is not installed, `--research` prints\na warning and proceeds with initial scores unchanged.\n\n## Advanced Patterns\n\n### Custom Scoring Dimensions\n\n```yaml\ncustom_dimensions:\n  regulatory_compliance:\n    weight: 1.5\n    description: \"Meets GDPR/SOC2/HIPAA requirements\"\n    scoring: {5: \"Fully compliant\", 3: \"Minor gaps\", 1: \"Concerns\"}\n```\n\n### Conditional Configuration\n\nOverride weights based on feature classification:\n\n```yaml\nconditional:\n  proactive:\n    tradeoffs:\n      latency: 0.6\n      resource_usage: 1.2\n  dynamic:\n    tradeoffs:\n      redundancy: 1.2\n      scalability: 1.2\n```\n\nFile v1.9.17:modules/multi-metric-evaluation-methodology.md\n\n# Multi-Metric Evaluation Methodology\n\nHow to combine RICE, WSJF, Kano, and related models when\nprioritizing a feature backlog. Each model encodes a\ndifferent assumption about what makes a feature worth\nbuilding. This module shows the formulas, where each model\nfits, and how to combine them when no single model is\nenough on its own.\n\n## The Models at a Glance\n\n| Model | Origin | Output | Captures |\n|-------|--------|--------|----------|\n| RICE | Intercom (Sean McBride, 2017) | Number | Reach * Impact * Confidence / Effort |\n| WSJF | SAFe (Scaled Agile) | Number | (Value, Time, and Risk) / Effort |\n| Kano | Noriaki Kano (1984) | Category | Basic, Performance, Delighter, Indifferent, Reverse |\n| MoSCoW | DSDM Consortium (1994) | Bucket | Must, Should, Could, Won't |\n| Cost-of-Delay | Don Reinertsen (2009) | $ / week | Value lost per week of delay |\n\nSingle-model use is rare in practice. Most teams converge\non a hybrid: RICE for a base score, WSJF to raise time-\ncritical items, Kano to gate basics. The rest of this\nmodule explains why and how.\n\n## Model 1: RICE\n\n```\nRICE = (Reach * Impact * Confidence) / Effort\n```\n\n| Factor | Unit | Typical scale |\n|--------|------|---------------|\n| Reach | users / period | absolute count |\n| Impact | satisfaction delta | 0.25, 0.5, 1, 2, 3 |\n| Confidence | probability | 0.5, 0.8, 1.0 |\n| Effort | person-months | 0.5, 1, 2, 5, 10 |\n\n**Best for**: large user-facing roadmaps where reach is\nmeasurable and a single team can absorb most items.\n\n**Worst for**: backlogs dominated by infrastructure or\ncompliance work where \"reach\" is meaningless or every item\nshares similar reach.\n\n**Worked example**:\n\n```\nFeature: Auto-save drafts\n  Reach:      8,000 users / quarter\n  Impact:     1.0 (significant satisfaction)\n  Confidence: 0.8\n  Effort:     2 person-months\n\nRICE = (8000 * 1.0 * 0.8) / 2 = 3200\n```\n\n## Model 2: WSJF\n\nWeighted Shortest Job First. From SAFe; treats\nprioritization as a cost-of-delay optimization.\n\n```\nWSJF = Cost_of_Delay / Job_Size\n\nCost_of_Delay = User_Value + Time_Criticality + Risk_Reduction\nJob_Size      = Effort estimate\n```\n\nEach input uses a Fibonacci scale: 1, 2, 3, 5, 8, 13, 20.\n\n| Factor | Question |\n|--------|----------|\n| User_Value | How much does the user/business gain? |\n| Time_Criticality | What does delay cost? Does the value decay? |\n| Risk_Reduction | Does this open future options or de-risk? |\n| Job_Size | How much work? |\n\n**Best for**: backlogs with strong time pressure and many\nitems where deferral has measurable cost. Common in\nSAFe-aligned organizations.\n\n**Worst for**: small teams without explicit\ncost-of-delay numbers; reduces to \"gut feel times Fibonacci\".\n\n**Worked example**:\n\n```\nFeature: GDPR consent banner\n  User_Value:        5\n  Time_Criticality:  20  (regulatory deadline)\n  Risk_Reduction:    13\n  Job_Size:          3\n\nWSJF = (5 + 20 + 13) / 3 = 12.67\n```\n\nCompare with the auto-save example: WSJF would put\nauto-save at roughly (8 + 3 + 2) / 5 = 2.6, far below the\nGDPR item, even though RICE might rank them similarly.\nWSJF surfaces the deadline.\n\n## Model 3: Kano\n\nKano classifies features by user reaction, not score.\n\n| Class | If present | If absent |\n|-------|-----------|-----------|\n| Basic | Expected; no joy | Strong dissatisfaction |\n| Performance | Linear satisfaction | Linear dissatisfaction |\n| Delighter | Joy | No reaction |\n| Indifferent | No reaction | No reaction |\n| Reverse | Dissatisfaction | Satisfaction |\n\n**Source**: Noriaki Kano et al., \"Attractive Quality and\nMust-Be Quality\" (1984).\n\n**Best for**: avoiding the most common backlog mistake:\nshipping a Delighter while a Basic is still missing.\n\n**Worst for**: numeric ranking. Kano gives categories, not\nscores. Pair it with RICE or WSJF for the actual ordering.\n\n**How to classify**: present users with two questions per\nfeature:\n\n```\nFunctional:    \"How would you feel if X were present?\"\nDysfunctional: \"How would you feel if X were absent?\"\n```\n\nEach answered on a 5-point scale from \"I like it\" to \"I\ndislike it\". The answer pair maps to a Kano category via a\nfixed table (see Berger et al. 1993).\n\n## When Each Model Fits\n\n```\nBacklog has clear users and reach measurable?\n  Yes -> RICE base\n  No  -> skip RICE\n\nItems have time-critical deadlines or value decay?\n  Yes -> WSJF overlay\n  No  -> skip WSJF\n\nBacklog mixes table-stakes and aspirational features?\n  Yes -> Kano gate\n  No  -> skip Kano\n\nStakeholders need narrative buckets, not numbers?\n  Yes -> MoSCoW translation layer\n  No  -> skip MoSCoW\n```\n\n| Backlog shape | First model | Second |\n|---------------|-------------|--------|\n| Consumer product, many features | RICE | Kano gate |\n| Enterprise SaaS with deadlines | WSJF | RICE |\n| New product, no users yet | Kano and MoSCoW | RICE later |\n| Regulated domain | WSJF | Cost-of-Delay |\n| Internal tooling | RICE with reach=team_size | Kano |\n\n## The Hybrid Used in This Skill\n\nThe `feature-review` skill combines RICE-like value /\ncost ratios, WSJF time criticality, and Kano gating. The\nformula is documented in `modules/scoring-framework.md`:\n\n```\nValue = weighted_avg(Reach, Impact, Business_Value, Time_Criticality)\nCost  = weighted_avg(Effort, Risk, Complexity)\nScore = (Value / Cost) * Confidence\n```\n\nKano enters as a hard gate before scoring. Any feature\nclassified Basic that is absent today is bumped above the\nranked list. The Score then orders everything else.\n\n```\n1. Classify every feature with Kano.\n2. Pull all missing Basics to the top, ordered by user impact.\n3. Score the rest with the Value/Cost formula.\n4. Rank by Score; apply confidence multiplier.\n5. Sensitivity-check the top 10 with +/- 20% weight perturbation.\n```\n\n## Worked Example: Combining RICE, WSJF, and Kano\n\nA team scores three candidates for the next sprint.\n\n```text\nCandidates:\n  A: Auto-save drafts\n  B: GDPR consent banner\n  C: Dark mode\n\nStep 1 (Kano):\n  A: Performance  (more frequent saves = more value)\n  B: Basic        (legally required; absent today)\n  C: Delighter\n\nStep 2: Pull Basics. B is bumped to top of queue.\n\nStep 3: Score A and C with hybrid:\n  A: Value=4.75, Cost=2.67, Conf=0.8 -> 1.42\n  C: Value=2.25, Cost=2.00, Conf=0.9 -> 1.01\n\nStep 4: WSJF check on B for sizing:\n  WSJF(B) = (5 + 20 + 13) / 3 = 12.67\n  Confirms B is the largest cost-of-delay item.\n\nStep 5: Sprint order:\n  1. B (regulatory Basic)\n  2. A (Score 1.42)\n  3. C (Score 1.01)\n\nSensitivity: vary all weights by +/- 20%. Order is stable\nin 18 of 20 perturbations. Fragile case: if Time\nCriticality weight drops below 0.10, A and C swap. Action:\nkeep weight at the documented 0.20.\n```\n\n## Anti-Patterns\n\n**Single-model orthodoxy.** Picking RICE because the blog\npost said so, then shoehorning every item into a \"reach\"\nestimate that does not exist. If the model does not fit\nthe input, change the model.\n\n**Hidden recalibration.** Reweighting Impact from 1.0 to\n3.0 mid-quarter to make a favored project rank higher.\nTrack weight history in version control; flag mid-cycle\nchanges.\n\n**Confidence rubber-stamping.** Every item scored at\nConfidence 1.0. Confidence 1.0 means \"I would bet the\nquarter on this estimate\". Real backlogs cluster around\n0.5 to 0.8.\n\n**Score inflation by Fibonacci jump.** \"It feels like an 8\"\nwhen the difference between 5 and 8 should be a 60% larger\ninvestment. Force a comparison: \"Is this 60% bigger than\nthe last 5 we shipped?\"\n\n**Aggregating Kano with a number.** Kano is categorical.\nAdding \"Basic = 5, Performance = 3, Delighter = 1\" to a\nscore creates the illusion of math.\n\n**Ignoring the Pareto front.** When two items tie on\nScore but trade off on different axes (one scales reach,\none buys time), report both and let humans pick. Do not\nbreak ties with a third decimal place.\n\n## Pitfalls Specific to AI/Plugin Backlogs\n\n**Reach is a mirage.** Plugin reach is bounded by who\ninstalls the plugin, not by the addressable market. Use\n\"installed teams\" as the reach unit, not \"potential users\".\n\n**Effort underestimates evals.** A new skill is not done\nwhen the prose is written. Add the cost of subagent test\nauthoring to Effort or the score will overpromise.\n\n**Confidence collapses on token-driven features.** New\ncontext-window or prompt features cannot be confidently\nestimated until measured against real workloads. Hold\nConfidence at 0.5 until benchmarks land.\n\n**Kano Basics drift.** A Delighter (auto-completion) can\nbecome a Basic in two release cycles. Re-classify the\nBasic set quarterly.\n\n## Cross-Reference\n\nSee `modules/scoring-framework.md` for the per-factor\nscales used in this skill,\n`modules/tradeoff-dimensions.md` for the quality axes\napplied after prioritization, and\n`plugins/leyline/skills/evaluation-framework/modules/multi-metric-evaluation-methodology.md`\nfor the math behind aggregation rules.\n\nFile v1.9.17:modules/research-enrichment.md\n\n# Research Enrichment\n\nExternal evidence from the tome plugin adjusts feature-review\nscoring factors. Research findings produce deltas applied to\ninitial human assessments, not replacement scores.\n\n## Channel-to-Factor Mapping\n\nEach tome research channel maps to primary and secondary\nscoring factors:\n\n| Channel | Primary Factor | Secondary Factor | Evidence Produced |\n|---------|---------------|-----------------|-------------------|\n| code-search | Reach | Complexity | Competitor count, star counts, implementation prevalence |\n| discourse | Impact | Business Value | Sentiment score, mention volume, request frequency |\n| papers | Impact | Risk | Citation count, novelty assessment, validation level |\n| triz | Business Value | Impact | Cross-domain analogy count, inventive principle match |\n\n## Score Delta Calculation\n\nResearch findings produce an adjustment delta for each factor:\n\n```\nresearch_delta = findings_consensus * evidence_strength\n\nWhere:\n  findings_consensus: -2 to +2 (direction and magnitude)\n  evidence_strength: 0.0 to 1.0 (how reliable the findings are)\n\napplied_delta = research_delta * channel_weight\n\nIf abs(applied_delta) < evidence_threshold:\n    applied_delta = 0  (insufficient evidence, discard)\n```\n\n### Channel Weight\n\nThe channel weight reflects how directly a channel's findings\nmap to its primary factor:\n\n| Channel | Weight | Rationale |\n|---------|--------|-----------|\n| code-search | 0.8 | Star counts approximate adoption well |\n| discourse | 0.7 | Sentiment is noisy but indicative |\n| papers | 0.9 | Peer-reviewed evidence is strong |\n| triz | 0.6 | Analogies are suggestive, not conclusive |\n\n### Evidence Strength Sources\n\n| Source | Strength | When |\n|--------|----------|------|\n| > 10 independent findings | 0.8-1.0 | High-volume channels |\n| 5-10 findings | 0.5-0.8 | Moderate evidence |\n| 1-5 findings | 0.3-0.5 | Sparse evidence |\n| 0 findings | 0.0 | No evidence (discard delta) |\n\n## Fibonacci Clamping\n\nAdjusted scores must remain on the Fibonacci scale used by\nthe scoring framework: [1, 2, 3, 5, 8, 13].\n\n```python\nFIBONACCI = [1, 2, 3, 5, 8, 13]\n\ndef clamp_to_fibonacci(raw_score: float) -> int:\n    \"\"\"Clamp raw score to nearest Fibonacci value.\"\"\"\n    return min(FIBONACCI, key=lambda f: abs(f - raw_score))\n```\n\n### Clamping Rules\n\n1. Calculate `raw_adjusted = initial_score + applied_delta`\n2. Clamp to nearest Fibonacci value\n3. Result must differ from initial by at most `max_delta`\n   Fibonacci steps\n4. If the clamped result exceeds `max_delta` steps from\n   initial, use the value `max_delta` steps away\n\nExample (max_delta = 2 steps):\n- Initial: 5, raw_adjusted: 7 -> clamp: 8 (1 step away) OK\n- Initial: 3, raw_adjusted: 11 -> clamp: 8 (3 steps away)\n  exceeds max_delta -> use 13 (2 steps away from 3)\n  Wait: 13 is 4 steps from 3. So use 8 (2 steps from 3).\n  Correction: count Fibonacci index steps, not arithmetic.\n\nFibonacci indices: 1=0, 2=1, 3=2, 5=3, 8=4, 13=5\n\n```python\ndef max_delta_clamp(initial: int, target: int, max_steps: int = 2) -> int:\n    initial_idx = FIBONACCI.index(initial)\n    target_idx = FIBONACCI.index(target)\n    if abs(target_idx - initial_idx) <= max_steps:\n        return target\n    direction = 1 if target_idx > initial_idx else -1\n    return FIBONACCI[initial_idx + direction * max_steps]\n```\n\n## Graceful Degradation\n\nWhen the tome plugin is not installed or research fails:\n\n1. **Tome not installed**: Print warning, skip Phase 4.5\n   entirely. Initial scores stand unchanged.\n2. **Individual channel fails**: Continue with remaining\n   channels. Only apply deltas from successful channels.\n3. **All channels fail**: Equivalent to tome not installed.\n   Log the failure, proceed with initial scores.\n4. **Timeout exceeded**: Use whatever findings collected so\n   far. Partial results are acceptable.\n\n### Detection Protocol\n\nCheck for tome availability:\n\n1. Look for `plugins/tome/` directory in the project\n2. If not found, check for tome in the global plugin path\n3. If neither found, activate graceful degradation\n\n## Integration with tome Skill Interfaces\n\nPhase 4.5 dispatches research via tome's public skill\ninterfaces:\n\n| Step | Action | tome Skill |\n|------|--------|------------|\n| 1 | Classify the project domain | `tome:research` (domain classifier) |\n| 2 | Dispatch parallel research agents | `tome:research` (agent dispatch) |\n| 3 | Synthesize findings | `tome:synthesize` |\n| 4 | (Optional) Refine high-potential areas | `tome:dig` |\n\nThe feature-review skill invokes these via `Skill()` calls,\nnot direct Python imports. This maintains loose coupling.\n\n### Research Topic Construction\n\nFor each feature under review, construct research topics:\n\n```\ntopic = f\"{feature_name} {feature_category} plugin/tool\"\n```\n\nExample: \"auto-save drafts developer tool\" or \"token\noptimization LLM CLI\"\n\n### Synthesis Integration\n\nAfter tome returns findings, extract deltas:\n\n1. Parse synthesized report for quantitative signals\n   (star counts, mention counts, citation counts)\n2. Map quantitative signals to delta values using the\n   channel-to-factor table\n3. Extract qualitative signals (sentiment, novelty) for\n   secondary factor adjustments\n4. Apply delta calculation formula\n5. Clamp to Fibonacci scale with max_delta constraint\n\n## Output Enhancement\n\nWhen research enrichment runs, add to the feature inventory:\n\n```markdown\n## Research Evidence\n\n### Code Search (GitHub)\n- Found 12 similar implementations, avg 340 stars\n- **Reach adjustment**: +1 (broad ecosystem adoption)\n\n### Discourse (HN/Reddit)\n- 47 mentions in last 90 days, 78% positive sentiment\n- **Impact adjustment**: +1 (strong community demand)\n\n### Score Adjustments\n\n| Feature | Factor | Initial | Delta | Adjusted | Evidence |\n|---------|--------|---------|-------|----------|----------|\n| Auth    | Reach  | 5       | +1    | 8        | 3 sources |\n| Auth    | Impact | 3       | 0     | 3        | Low      |\n```\n\nFile v1.9.17:modules/scoring-framework.md\n\n# Scoring Framework\n\nHybrid prioritization combining RICE (Intercom), WSJF (SAFe), and Kano classification, grounded in Multi-Criteria Decision Analysis (MCDA) principles.\n\n## Mathematical Foundation\n\nThis framework extends standard prioritization models with MCDA best practices:\n\n- **Normalization**: Logarithmic normalization for score scales (handles non-linear value perception)\n- **Weighting**: Customizable weights with validation requirements\n- **Trade-offs**: Explicit handling through Value/Cost ratio\n- **Uncertainty**: Confidence factor adjusts for estimation risk\n- **Sensitivity**: Weight variations tested for robustness\n\n**Documentation**: See [Multi-Metric Evaluation Methodology](https://claude-night-market/plugins/abstract/skills/skills-eval/modules/multi-metric-evaluation-methodology.md) for theoretical foundations.\n\n## The Formula\n\n```\nFeature Score = (Value Score / Cost Score) * Confidence\n\nWhere:\n  Value Score = weighted_avg(Reach, Impact, Business Value, Time Criticality)\n  Cost Score = weighted_avg(Effort, Risk, Complexity)\n  Confidence = 0.0 to 1.0 (how certain are we about estimates?)\n```\n\n### Validation Requirements\n\nBefore using this framework:\n\n```yaml\nvalidation:\n  weights:\n    - Document weight derivation method (AHP, expert judgment, empirical)\n    - Verify weights sum to 1.0 within each category (value, cost)\n    - Test sensitivity to ±20% weight variations\n    - Flag critical weights that significantly change rankings\n\n  normalization:\n    - Method: \"logarithmic\" (handles non-linear perception)\n    - Rationale: \"Diminishing returns on raw scores\"\n    - Scale_invariance: \"Not required (absolute scale used)\"\n\n  uncertainty:\n    - Confidence < 0.5: Require research before commitment\n    - Document basis for confidence assessment\n    - Consider worst-case scenario for low-confidence items\n```\n\n## Value Factors\n\n### Reach (R)\n\n**Question:** How many users/use-cases does this affect?\n\n| Score | Meaning | Example |\n|-------|---------|---------|\n| 1 | Very few (<5%) | Niche admin feature |\n| 2 | Some (5-15%) | Power user feature |\n| 3 | Moderate (15-35%) | Common workflow |\n| 5 | Many (35-60%) | Core user journey |\n| 8 | Most (60-85%) | Essential feature |\n| 13 | Nearly all (>85%) | Universal need |\n\n### Impact (I)\n\n**Question:** How much does this improve the user experience?\n\n| Score | Meaning | Kano Category |\n|-------|---------|---------------|\n| 1 | Minimal improvement | Basic (expected) |\n| 2 | Slight improvement | Basic |\n| 3 | Noticeable improvement | Performance |\n| 5 | Significant improvement | Performance |\n| 8 | Major improvement | Performance |\n| 13 | Transformative | Delighter |\n\n### Business Value (BV)\n\n**Question:** How does this contribute to business goals/OKRs?\n\n| Score | Meaning | OKR Alignment |\n|-------|---------|---------------|\n| 1 | Tangential | No direct OKR connection |\n| 2 | Supporting | Supports an initiative |\n| 3 | Contributing | Contributes to Key Result |\n| 5 | Advancing | Directly advances Key Result |\n| 8 | Critical | Required for Key Result |\n| 13 | Strategic | Core to company Objective |\n\n### Time Criticality (TC)\n\n**Question:** What's the cost of delay?\n\n| Score | Meaning | Urgency |\n|-------|---------|---------|\n| 1 | Can wait indefinitely | Nice to have |\n| 2 | Can wait 6+ months | Low urgency |\n| 3 | Should do this quarter | Moderate urgency |\n| 5 | Should do this month | High urgency |\n| 8 | Should do this sprint | Very high urgency |\n| 13 | Must do immediately | Blocking/critical |\n\n## Cost Factors\n\n### Effort (E)\n\n**Question:** How much work is this?\n\n| Score | Meaning | Time Estimate |\n|-------|---------|---------------|\n| 1 | Trivial | < 1 day |\n| 2 | Small | 1-3 days |\n| 3 | Moderate | 3-5 days |\n| 5 | Large | 1-2 weeks |\n| 8 | Very large | 2-4 weeks |\n| 13 | Huge | > 1 month |\n\n### Risk (Rk)\n\n**Question:** What could go wrong?\n\n| Score | Meaning | Risk Level |\n|-------|---------|------------|\n| 1 | Very low risk | Well-understood, no dependencies |\n| 2 | Low risk | Minor unknowns |\n| 3 | Moderate risk | Some unknowns or dependencies |\n| 5 | High risk | Significant unknowns |\n| 8 | Very high risk | Many unknowns, critical dependencies |\n| 13 | Extreme risk | Uncharted territory |\n\n### Complexity (Cx)\n\n**Question:** How hard is this to build correctly?\n\n| Score | Meaning | Complexity Level |\n|-------|---------|------------------|\n| 1 | Simple | Single component, clear requirements |\n| 2 | Straightforward | Few components, clear interfaces |\n| 3 | Moderate | Multiple components, some edge cases |\n| 5 | Complex | Cross-cutting concerns, many edge cases |\n| 8 | Very complex | Architectural changes, distributed state |\n| 13 | Extremely complex | Novel algorithms, fundamental changes |\n\n## Confidence Scoring\n\nRate your confidence in the estimates:\n\n| Confidence | Meaning | When to Use |\n|------------|---------|-------------|\n| 0.9-1.0 | High | Clear requirements, similar past work |\n| 0.7-0.9 | Moderate | Some unknowns, reasonable estimates |\n| 0.5-0.7 | Low | Many unknowns, rough estimates |\n| 0.3-0.5 | Very low | Mostly guessing |\n| < 0.3 | Speculative | Requires spike/research first |\n\n**Guardrail:** Features with confidence < 0.5 should be flagged for research before commitment.\n\n## Kano Classification\n\nAfter scoring, classify the feature:\n\n### Basic (Must-Have)\n\n- Users expect this; absence causes dissatisfaction\n- Doesn't increase satisfaction when present\n- **Action:** validate these exist before anything else\n\n### Performance (Linear)\n\n- More is better; satisfaction scales with quality\n- Competitive differentiator\n- **Action:** Optimize based on ROI\n\n### Delighters (Wow Factors)\n\n- Unexpected features that create joy\n- Absence doesn't hurt; presence delights\n- **Action:** Build after basics and key performers\n\n### Indifferent\n\n- Users don't care either way\n- **Action:** Deprioritize or cut\n\n### Reverse\n\n- Feature that some users actively dislike\n- **Action:** Make optional or remove\n\n## Calculation Example\n\n```yaml\nFeature: Auto-save drafts\n\n# Value Factors\nReach: 8          # Most users write drafts\nImpact: 5         # Significant UX improvement\nBusiness Value: 3 # Supports retention KR\nTime Criticality: 3 # Should do this quarter\n\nValue Score = (8 + 5 + 3 + 3) / 4 = 4.75\n\n# Cost Factors\nEffort: 3         # 3-5 days\nRisk: 2           # Low risk, understood problem\nComplexity: 3     # Moderate, needs state management\n\nCost Score = (3 + 2 + 3) / 3 = 2.67\n\n# Confidence\nConfidence: 0.8   # Similar features built before\n\n# Final Score\nFeature Score = (4.75 / 2.67) * 0.8 = 1.42\n\n# Classification\nKano: Performance (more saving = better UX)\nPriority: Medium (1.42 is between 1.5-2.5 threshold)\n```\n\n## Interpreting Scores\n\n| Score Range | Priority | Recommendation |\n|-------------|----------|----------------|\n| > 2.5 | High | Schedule for next sprint |\n| 1.5 - 2.5 | Medium | Add to roadmap, plan timing |\n| 1.0 - 1.5 | Low | Backlog, revisit quarterly |\n| < 1.0 | Very Low | Defer indefinitely or reject |\n\n## Custom Weights\n\nDefault weights can be customized in `.feature-review.yaml`:\n\n```yaml\nweights:\n  value:\n    reach: 0.25           # Equal weighting\n    impact: 0.30          # Slightly favor user impact\n    business_value: 0.25  # Equal to reach\n    time_criticality: 0.20 # Slightly less weight\n  cost:\n    effort: 0.40          # Effort matters most\n    risk: 0.30            # Risk is significant\n    complexity: 0.30      # Complexity matters\n\n# REQUIRED: Document weight derivation\nderivation:\n  method: \"expert_judgment\"  # or \"AHP\" or \"empirical\"\n  experts: 3\n  date: \"2025-01-07\"\n  rationale: \"Impact weighted slightly higher based on user feedback\"\n```\n\n**Guardrails**:\n- Weights within each category must sum to 1.0\n- Document how weights were derived (not arbitrary)\n- Run sensitivity analysis before finalizing\n- Flag weights that cause ranking instability\n\n## Comparison with Pure RICE\n\n| Aspect | RICE | Feature Review |\n|--------|------|----------------|\n| Value factors | Reach, Impact | Reach, Impact, BV, TC |\n| Cost factors | Effort only | Effort, Risk, Complexity |\n| Time sensitivity | Not explicit | Time Criticality factor |\n| Business alignment | Not explicit | Business Value factor |\n| Uncertainty | Confidence | Confidence |\n| Classification | None | Kano model |\n| **MCDA Compliance** | Basic | **Full** (normalization, weighting, sensitivity) |\n\nFeature Review extends RICE with WSJF's time criticality and business value, plus Kano classification for strategic context, all grounded in MCDA best practices.\n\n## Sensitivity Analysis\n\nBefore committing to prioritization, test robustness:\n\n```python\ndef priority_sensitivity_analysis(features, weights, variation=0.20):\n    \"\"\"\n    Tests if rankings are stable to weight variations.\n\n    Args:\n        features: List of features with scores\n        weights: Current weight configuration\n        variation: Test ±20% changes\n\n    Returns:\n        Dict with sensitivity metrics\n    \"\"\"\n    base_ranking = rank_features(features, weights)\n    sensitivity = {}\n\n    for category in [\"value\", \"cost\"]:\n        for factor in weights[category].keys():\n            # Test weight increase\n            weights_plus = adjust_weight(weights, factor, +variation)\n            ranking_plus = rank_features(features, weights_plus)\n            correlation_plus = spearman_correlation(base_ranking, ranking_plus)\n\n            # Test weight decrease\n            weights_minus = adjust_weight(weights, factor, -variation)\n            ranking_minus = rank_features(features, weights_minus)\n            correlation_minus = spearman_correlation(base_ranking, ranking_minus)\n\n            sensitivity[factor] = {\n                \"avg_correlation\": (correlation_plus + correlation_minus) / 2,\n                \"sensitive\": correlation_plus < 0.8 or correlation_minus < 0.8\n            }\n\n    return sensitivity\n```\n\n**Interpretation**:\n- Correlation > 0.9: Ranking stable to this weight variation\n- Correlation 0.8-0.9: Moderately sensitive\n- Correlation < 0.8: Highly sensitive, weight is critical\n\nFile v1.9.17:modules/tradeoff-dimensions.md\n\n# Tradeoff Dimensions\n\nQuality attributes for evaluating features. Based on ISO 25010 software quality model, CAP/PACELC theorems, and practical engineering concerns.\n\n## Overview\n\nEvery feature makes tradeoffs. This module provides structured evaluation across nine dimensions, each with specific criteria and scoring guidance.\n\n**Scoring Scale:** 1-5 for each dimension\n\n| Score | Meaning |\n|-------|---------|\n| 1 | Poor - Significant issues |\n| 2 | Below average - Notable gaps |\n| 3 | Adequate - Meets basic needs |\n| 4 | Good - Above expectations |\n| 5 | Excellent - Best in class |\n\n## Dimension 1: Quality of Results\n\n**Question:** Does the feature deliver correct, accurate results?\n\n### Evaluation Criteria\n\n| Score | Criteria |\n|-------|----------|\n| 5 | Always correct, handles all edge cases, validated against ground truth |\n| 4 | Correct in normal cases, handles most edge cases |\n| 3 | Usually correct, some known edge case issues |\n| 2 | Occasionally incorrect, multiple edge case failures |\n| 1 | Frequently incorrect, unreliable outputs |\n\n### Considerations\n\n- **Correctness:** Does it produce right answers?\n- **Completeness:** Does it handle all expected inputs?\n- **Precision:** How accurate are numerical/search results?\n- **Recall:** Does it find everything it should?\n\n### Tradeoff Partners\n\n- Quality often trades against **Latency** (more checks = slower)\n- Quality often trades against **Resource Usage** (validation costs)\n\n---\n\n## Dimension 2: Latency\n\n**Question:** Does the feature meet timing requirements for its classification?\n\n### Evaluation Criteria by Type\n\n**For Reactive Features:**\n\n| Score | Criteria |\n|-------|----------|\n| 5 | < 50ms response, feels instant |\n| 4 | 50-100ms response, very responsive |\n| 3 | 100-300ms response, acceptable |\n| 2 | 300ms-1s response, noticeable delay |\n| 1 | > 1s response, frustrating delay |\n\n**For Proactive Features:**\n\n| Score | Criteria |\n|-------|----------|\n| 5 | Completes before needed, no user awareness |\n| 4 | Usually ready when needed |\n| 3 | Sometimes user waits briefly |\n| 2 | Often not ready, visible loading |\n| 1 | Rarely ready, defeats purpose |\n\n### Considerations\n\n- **P50 latency:** Typical case\n- **P99 latency:** Worst case (matters for reliability)\n- **Cold start:** First invocation time\n- **Warm path:** Subsequent invocations\n\n### Tradeoff Partners\n\n- Latency trades against **Quality** (PACELC theorem)\n- Latency trades against **Consistency** (eventual vs strong)\n\n---\n\n## Dimension 3: Token Usage\n\n**Question:** Is the feature context-efficient for LLM interactions?\n\n### Evaluation Criteria\n\n| Score | Criteria |\n|-------|----------|\n| 5 | Minimal tokens, highly compressed, efficient prompts |\n| 4 | Reasonable tokens, well-structured |\n| 3 | Average tokens, some verbosity |\n| 2 | High token usage, could be optimized |\n| 1 | Excessive tokens, bloated context |\n\n### Considerations\n\n- **Input tokens:** How much context needed?\n- **Output tokens:** How verbose are results?\n- **Caching potential:** Can results be reused?\n- **Streaming:** Can partial results reduce perception?\n\n### Measurement\n\n```\nToken Efficiency = Useful Output / Total Tokens\n```\n\nTarget: > 0.5 for most features\n\n### Tradeoff Partners\n\n- Token usage trades against **Quality** (more context = better results)\n- Token usage trades against **Readability** (compression reduces clarity)\n\n---\n\n## Dimension 4: Resource Usage (CPU/Memory)\n\n**Question:** Is CPU and memory consumption reasonable?\n\n### Evaluation Criteria\n\n| Score | Criteria |\n|-------|----------|\n| 5 | Minimal footprint, efficient algorithms |\n| 4 | Low resource usage, well-optimized |\n| 3 | Moderate usage, acceptable overhead |\n| 2 | High usage, performance impact on system |\n| 1 | Excessive usage, causes degradation |\n\n### Considerations\n\n- **Peak memory:** Maximum allocation\n- **Sustained memory:** Ongoing consumption\n- **CPU intensity:** Processing load\n- **I/O patterns:** Disk/network usage\n\n### Measurement Guidance\n\n| Resource | Good | Acceptable | Poor |\n|----------|------|------------|------|\n| Memory delta | < 10MB | 10-50MB | > 50MB |\n| CPU spike | < 100ms | 100-500ms | > 500ms |\n| Sustained CPU | < 5% | 5-20% | > 20% |\n\n### Tradeoff Partners\n\n- Resources trade against **Latency** (caching uses memory)\n- Resources trade against **Scalability** (per-user costs)\n\n---\n\n## Dimension 5: Redundancy (Fault Tolerance)\n\n**Question:** Does the feature handle failures gracefully?\n\n### Evaluation Criteria\n\n| Score | Criteria |\n|-------|----------|\n| 5 | Full redundancy, automatic failover, no data loss |\n| 4 | Good failover, minimal disruption |\n| 3 | Basic error handling, recoverable failures |\n| 2 | Some failure handling, may require retry |\n| 1 | No redundancy, failures cause data loss or crashes |\n\n### Considerations\n\n- **Graceful degradation:** Does it fail partially vs completely?\n- **Retry logic:** Does it handle transient failures?\n- **Data durability:** Is data protected from loss?\n- **Recovery time:** How fast to recover?\n\n### CAP Theorem Implications\n\nFor distributed features:\n- **CP systems:** May sacrifice availability for consistency\n- **AP systems:** May sacrifice consistency for availability\n\n### Tradeoff Partners\n\n- Redundancy trades against **Complexity** (more failure modes)\n- Redundancy trades against **Latency** (replication delays)\n\n---\n\n## Dimension 6: Readability (Maintainability)\n\n**Question:** Can others understand and modify this feature?\n\n### Evaluation Criteria\n\n| Score | Criteria |\n|-------|----------|\n| 5 | Self-documenting, clear abstractions, easy to extend |\n| 4 | Well-structured, good comments, learnable |\n| 3 | Understandable with effort, some complexity |\n| 2 | Hard to follow, requires tribal knowledge |\n| 1 | Opaque, only original author understands |\n\n### Considerations\n\n- **Code clarity:** Is logic obvious?\n- **Documentation:** Are complex parts explained?\n- **Naming:** Are variables/functions descriptive?\n- **Structure:** Is code well-organized?\n\n### Measurement Proxies\n\n- Cyclomatic complexity < 10\n- Functions < 50 lines\n- Clear separation of concerns\n- Test coverage > 80%\n\n### Tradeoff Partners\n\n- Readability trades against **Token Usage** (verbose = clearer)\n- Readability trades against **Resource Usage** (abstractions have cost)\n\n---\n\n## Dimension 7: Scalability\n\n**Question:** Will the feature handle 10x load?\n\n### Evaluation Criteria\n\n| Score | Criteria |\n|-------|----------|\n| 5 | Linear or sub-linear scaling, handles massive load |\n| 4 | Good scaling, handles significant growth |\n| 3 | Adequate scaling, may need attention at 5x |\n| 2 | Poor scaling, issues at 2-3x load |\n| 1 | Doesn't scale, breaks under modest increase |\n\n### Considerations\n\n- **Horizontal scaling:** Can add more instances?\n- **Vertical scaling:** Can add more resources?\n- **Bottlenecks:** Where does it fail first?\n- **State management:** How is state distributed?\n\n### Scaling Patterns\n\n| Pattern | Scalability | Complexity |\n|---------|-------------|------------|\n| Stateless | Excellent | Low |\n| Cached | Very good | Medium |\n| Sharded | Good | High |\n| Stateful single | Poor | Low |\n\n### Tradeoff Partners\n\n- Scalability trades against **Complexity** (distributed systems are hard)\n- Scalability trades against **Redundancy** (more nodes = more failure modes)\n\n---\n\n## Dimension 8: Integration (Interoperability)\n\n**Question:** Does the feature play well with existing systems?\n\n### Evaluation Criteria\n\n| Score | Criteria |\n|-------|----------|\n| 5 | smooth integration, follows all conventions, composable |\n| 4 | Good integration, minor adaptations needed |\n| 3 | Integrates with effort, some friction |\n| 2 | Difficult integration, significant workarounds |\n| 1 | Isolated, doesn't integrate without major changes |\n\n### Considerations\n\n- **API consistency:** Matches existing patterns?\n- **Data formats:** Uses standard formats?\n- **Dependencies:** Minimal coupling?\n- **Extension points:** Can be extended/wrapped?\n\n### Integration Checklist\n\n- [ ] Uses project's standard data formats\n- [ ] Follows naming conventions\n- [ ] Integrates with existing logging/metrics\n- [ ] Works with existing auth/permissions\n- [ ] Compatible with existing tooling\n\n### Tradeoff Partners\n\n- Integration trades against **Innovation** (conventions limit novelty)\n- Integration trades against **Optimization** (generic > specialized)\n\n---\n\n## Dimension 9: API Surface\n\n**Question:** Is the API backward compatible and well-designed?\n\n### Evaluation Criteria\n\n| Score | Criteria |\n|-------|----------|\n| 5 | Stable API, versioned, excellent documentation, no breaking changes |\n| 4 | Good API, rare breaking changes with migration path |\n| 3 | Adequate API, occasional breaking changes |\n| 2 | Unstable API, frequent breaking changes |\n| 1 | No stable API, constant churn |\n\n### Considerations\n\n- **Breaking changes:** How often do consumers break?\n- **Deprecation policy:** Are changes communicated?\n- **Versioning:** Is there a versioning strategy?\n- **Documentation:** Is the contract clear?\n\n### API Design Checklist\n\n- [ ] Additive changes only (no removal)\n- [ ] Optional new fields with defaults\n- [ ] Deprecation warnings before removal\n- [ ] Semantic versioning\n- [ ] Clear error contracts\n\n### Tradeoff Partners\n\n- API stability trades against **Innovation** (can't change freely)\n- API stability trades against **Quality** (may keep suboptimal designs)\n\n---\n\n## Composite Scoring\n\nCalculate overall tradeoff score:\n\n```\nTradeoff Score = Σ(dimension_score * dimension_weight) / Σ(weights)\n```\n\n### Default Weights\n\n| Dimension | Default Weight | Rationale |\n|-----------|---------------|-----------|\n| Quality | 1.0 | Core requirement |\n| Latency | 1.0 | User experience |\n| Token Usage | 1.0 | LLM efficiency |\n| Resource Usage | 0.8 | Important but secondary |\n| Redundancy | 0.5 | Context-dependent |\n| Readability | 1.0 | Maintainability |\n| Scalability | 0.8 | Future-proofing |\n| Integration | 1.0 | Ecosystem fit |\n| API Surface | 1.0 | Contract stability |\n\n### Guardrail\n\n**Minimum 5 dimensions must be evaluated.** Cannot skip all tradeoff analysis.\n\n---\n\n## Using Tradeoffs in Review\n\n### For Existing Features\n\n1. Score each dimension\n2. Identify dimensions below 3\n3. Determine if improvement is feasible\n4. Prioritize based on impact\n\n### For Proposed Features\n\n1. Estimate scores for each dimension\n2. Compare against existing features\n3. Identify which tradeoffs are acceptable\n4. Document accepted tradeoffs explicitly\n\n### Red Flag Combinations\n\n| Pattern | Concern | Action |\n|---------|---------|--------|\n| High Quality and High Latency | May frustrate users | Optimize or classify as Proactive |\n| Low Readability and Low API Surface | Maintenance nightmare | Refactor before extending |\n| High Token and Low Quality | Wasteful | Optimize prompts |\n| Low Redundancy and High Integration | Cascading failures | Add fault tolerance |\n\nFile v1.9.17:skill-card.md\n\n## Description: <br>\nScores backlog items with RICE/WSJF/Kano and files GitHub issues for top candidates. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[athola](https://clawhub.ai/user/athola) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and engineering teams use this skill to inventory roadmap features, score backlog candidates, analyze tradeoffs, and prepare GitHub issues for accepted suggestions. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: The skill can instruct the agent to run a local deferred-capture script automatically after planning decisions. <br>\nMitigation: Use it only in repositories where the local deferred_capture.py behavior is trusted, or disable deferred capture and require explicit confirmation before any local write. <br>\nRisk: Roadmap scores and suggested issues may be incomplete or misleading if feature inventory, confidence, or research inputs are weak. <br>\nMitigation: Review generated priorities and issue drafts before acting on them, especially low-confidence scores or changes that affect API surfaces. <br>\n\n\n## Reference(s): <br>\n- [ClawHub Skill Listing](https://clawhub.ai/athola/skills/nm-imbue-feature-review) <br>\n- [Source Homepage](https://github.com/athola/claude-night-market/tree/master/plugins/imbue) <br>\n- [Classification System](modules/classification-system.md) <br>\n- [Configuration](modules/configuration.md) <br>\n- [Multi-Metric Evaluation Methodology](modules/multi-metric-evaluation-methodology.md) <br>\n- [Research Enrichment](modules/research-enrichment.md) <br>\n- [Scoring Framework](modules/scoring-framework.md) <br>\n- [Tradeoff Dimensions](modules/tradeoff-dimensions.md) <br>\n- [External Multi-Metric Evaluation Methodology](https://claude-night-market/plugins/abstract/skills/skills-eval/modules/multi-metric-evaluation-methodology.md) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [text, markdown, shell commands, configuration, guidance] <br>\n**Output Format:** [Markdown reports, tables, issue drafts, and inline shell commands] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [May create GitHub issue content when explicitly requested and may trigger deferred local capture for skipped high-priority suggestions.] <br>\n\n## Skill Version(s): <br>\n1.9.17 (source: ClawHub release evidence; artifact frontmatter reports 1.9.8) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nArchive v1.9.16: 9 files, 28261 bytes\n\nFiles: modules/classification-system.md (8999b), modules/configuration.md (6593b), modules/multi-metric-evaluation-methodology.md (8759b), modules/research-enrichment.md (5903b), modules/scoring-framework.md (10049b), modules/tradeoff-dimensions.md (10886b), skill-card.md (2550b), SKILL.md (12170b), _meta.json (143b)\n\nFile v1.9.16:SKILL.md\n\n---\nname: feature-review\ndescription: |\n  Scores backlog items with RICE/WSJF/Kano and files GitHub issues for top candidates\nversion: 1.9.8\ntriggers:\n  - feature-prioritization\n  - backlog-triage\n  - RICE\n  - WSJF\n  - Kano\n  - roadmap\n  - triaging a roadmap or prioritizing features for a sprint\nmetadata: {\"openclaw\": {\"homepage\": \"https://github.com/athola/claude-night-market/tree/master/plugins/imbue\", \"emoji\": \"\\ud83e\\udd9e\", \"requires\": {\"config\": [\"night-market.imbue:scope-guard\"]}}}\nsource: claude-night-market\nsource_plugin: imbue\n---\n\n> **Night Market Skill** — ported from [claude-night-market/imbue](https://github.com/athola/claude-night-market/tree/master/plugins/imbue). For the full experience with agents, hooks, and commands, install the Claude Code plugin.\n\n\n## Table of Contents\n\n- [Philosophy](#philosophy)\n- [When to Use](#when-to-use)\n- [When NOT to Use](#when-not-to-use)\n- [Quick Start](#quick-start)\n- [1. Inventory Current Features](#1-inventory-current-features)\n- [2. Score and Classify](#2-score-and-classify)\n- [3. Generate Suggestions](#3-generate-suggestions)\n\n## Verification\n\nRun `make test-feature-review` to verify scoring logic after changes.\n- [4. Upload to GitHub](#4-upload-to-github)\n- [Workflow](#workflow)\n- [Phase 1: Feature Discovery (`feature-review:inventory-complete`)](#phase-1:-feature-discovery-(feature-review:inventory-complete))\n- [Phase 2: Classification (`feature-review:classified`)](#phase-2:-classification-(feature-review:classified))\n- [Phase 3: Scoring (`feature-review:scored`)](#phase-3:-scoring-(feature-review:scored))\n- [Phase 4: Tradeoff Analysis (`feature-review:tradeoffs-analyzed`)](#phase-4:-tradeoff-analysis-(feature-review:tradeoffs-analyzed))\n- [Phase 5: Gap Analysis & Suggestions (`feature-review:suggestions-generated`)](#phase-5:-gap-analysis-&-suggestions-(feature-review:suggestions-generated))\n- [Phase 6: GitHub Integration (`feature-review:issues-created`)](#phase-6:-github-integration-(feature-review:issues-created))\n- [Configuration](#configuration)\n- [Configuration File](#configuration-file)\n- [Guardrails](#guardrails)\n- [Required TodoWrite Items](#required-todowrite-items)\n- [Integration Points](#integration-points)\n- [Output Format](#output-format)\n- [Feature Inventory Table](#feature-inventory-table)\n- [Suggestion Report](#suggestion-report)\n- [Feature Suggestions](#feature-suggestions)\n- [High Priority (Score > 2.5)](#high-priority-(score->-25))\n- [Related Skills](#related-skills)\n- [Reference](#reference)\n\n\n# Feature Review\n\nReview implemented features and suggest new ones using evidence-based prioritization. Create GitHub issues for accepted suggestions.\n\n## Philosophy\n\nFeature decisions rely on data. Every feature involves tradeoffs that require evaluation. This skill uses hybrid RICE+WSJF scoring with Kano classification to prioritize work and generates actionable GitHub issues for accepted suggestions.\n\n## When To Use\n\n- Roadmap reviews (sprint planning, quarterly reviews).\n- Retrospective evaluations.\n- Planning new development cycles.\n\n## When NOT To Use\n\n- Emergency bug fixes.\n- Simple documentation updates.\n- Active implementation (use `scope-guard`).\n\n## Quick Start\n\n### 1. Inventory Current Features\n\nDiscover and categorize existing features:\n```bash\n/feature-review --inventory\n```\n\n### 2. Score and Classify\n\nEvaluate features against the prioritization framework:\n```bash\n/feature-review\n```\n\n### 3. Generate Suggestions\n\nReview gaps and suggest new features:\n```bash\n/feature-review --suggest\n```\n\n### 4. Research-Enriched Scoring\n\nUse tome plugin to adjust scores with external evidence:\n```bash\n/feature-review --research\n```\n\n### 5. Upload to GitHub\n\nCreate issues for accepted suggestions:\n```bash\n/feature-review --suggest --create-issues\n```\n\n## Workflow\n\n### Phase 1: Feature Discovery (`feature-review:inventory-complete`)\n\nIdentify features by analyzing:\n\n1. **Code artifacts**: Entry points, public APIs, and configuration surfaces.\n2. **Documentation**: README lists, CHANGELOG entries, and user docs.\n3. **Git history**: Recent feature commits and branches.\n\n**Output:** Feature inventory table.\n\n### Phase 2: Classification (`feature-review:classified`)\n\nClassify each feature along two axes:\n\n**Axis 1: Proactive vs Reactive**\n\n| Type | Definition | Examples |\n|------|------------|----------|\n| **Proactive** | Anticipates user needs. | Suggestions, prefetching. |\n| **Reactive** | Responds to explicit input. | Form handling, click actions. |\n\n**Axis 2: Static vs Dynamic**\n\n| Type | Update Pattern | Storage Model |\n|------|---------------|---------------|\n| **Static** | Incremental, versioned. | File-based, cached. |\n| **Dynamic** | Continuous, streaming. | Database, real-time. |\n\nSee [classification-system.md](modules/classification-system.md) for details.\n\n### Phase 3: Scoring (`feature-review:scored`)\n\nApply hybrid RICE+WSJF scoring:\n\n```\nFeature Score = Value Score / Cost Score\n\nValue Score = (Reach + Impact + Business Value + Time Criticality) / 4\nCost Score = (Effort + Risk + Complexity) / 3\n\nAdjusted Score = Feature Score * Confidence\n```\n\n**Scoring Scale:** Fibonacci (1, 2, 3, 5, 8, 13).\n\n**Thresholds:**\n- **> 2.5**: High priority.\n- **1.5 - 2.5**: Medium priority.\n- **< 1.5**: Low priority.\n\nSee [scoring-framework.md](modules/scoring-framework.md) for the framework.\nSee [multi-metric-evaluation-methodology.md](modules/multi-metric-evaluation-methodology.md)\nwhen one model is not enough: it covers how to combine\nRICE, WSJF, and Kano, where each model fits, and how to\nreconcile conflicting signals.\n\n### Phase 4: Tradeoff Analysis (`feature-review:tradeoffs-analyzed`)\n\nEvaluate each feature across quality dimensions:\n\n| Dimension | Question | Scale |\n|-----------|----------|-------|\n| **Quality** | Does it deliver correct results? | 1-5 |\n| **Latency** | Does it meet timing requirements? | 1-5 |\n| **Token Usage** | Is it context-efficient? | 1-5 |\n| **Resource Usage** | Is CPU/memory reasonable? | 1-5 |\n| **Redundancy** | Does it handle failures gracefully? | 1-5 |\n| **Readability** | Can others understand it? | 1-5 |\n| **Scalability** | Will it handle 10x load? | 1-5 |\n| **Integration** | Does it play well with others? | 1-5 |\n| **API Surface** | Is it backward compatible? | 1-5 |\n\nSee [tradeoff-dimensions.md](modules/tradeoff-dimensions.md) for criteria.\n\n### Phase 4.5: Research Enrichment (`feature-review:research-enriched`)\n\n**Triggered by:** `--research` flag. Requires tome plugin.\n\nUse tome's multi-source research to adjust scoring factors\nwith external evidence. This phase runs between tradeoff\nanalysis and gap analysis.\n\n1. **Dispatch research**: For each feature, construct\n   research topics and dispatch tome channels (code-search,\n   discourse, papers, triz) in parallel.\n2. **Synthesize findings**: Merge results across channels\n   using `tome:synthesize`.\n3. **Calculate deltas**: Map findings to scoring factor\n   adjustments using channel-to-factor mapping.\n4. **Apply deltas**: Adjust initial scores by research\n   deltas, clamp to Fibonacci scale, respect max_delta.\n5. **Present evidence**: Show adjustment table with\n   evidence sources and rationale.\n\nSee [research-enrichment.md](modules/research-enrichment.md)\nfor the full enrichment protocol, delta calculation, and\ngraceful degradation behavior.\n\n**Graceful degradation**: If tome is not installed, prints\na warning and proceeds with initial scores unchanged.\n\n### Phase 5: Gap Analysis & Suggestions (`feature-review:suggestions-generated`)\n\n1. **Identify gaps**: Missing Kano basics.\n2. **Surface opportunities**: High-value, low-effort features.\n3. **Flag technical debt**: Features with declining scores.\n4. **Recommend actions**: Build, improve, deprecate, or maintain.\n\n### Phase 6: GitHub Integration (`feature-review:issues-created`)\n\n1. Generate issue title and body from suggestions.\n2. Apply labels (feature, enhancement, priority/*).\n3. Link to related issues.\n4. Confirm with user before creation.\n\n**Deferred capture for high-scoring suggestions:**\nAfter the user confirms which suggestions to act on, any\nhigh-scoring suggestion (score > 2.5) that is not acted on\nshould be preserved as a deferred item.\nRun once per skipped high-scoring suggestion:\n\n```bash\npython3 scripts/deferred_capture.py \\\n  --title \"<suggestion title>\" \\\n  --source feature-review \\\n  --context \"RICE score: <score>. <description>\"\n```\n\nThis runs automatically without prompting the user.\nSuggestions with scores of 2.5 or below do not need\nto be captured.\n\n## Configuration\n\nFeature-review uses opinionated defaults but allows customization.\n\n### Configuration File\n\nCreate `.feature-review.yaml` in project root:\n\n```yaml\n# .feature-review.yaml\nversion: 1.9.3\n\n# Scoring weights (m\n\nArchive v1.9.14: 9 files, 28507 bytes\n\nFiles: modules/classification-system.md (8999b), modules/configuration.md (6593b), modules/multi-metric-evaluation-methodology.md (8759b), modules/research-enrichment.md (5903b), modules/scoring-framework.md (10049b), modules/tradeoff-dimensions.md (10886b), skill-card.md (3331b), SKILL.md (12170b), _meta.json (143b)\n\nArchive v1.9.13: 9 files, 28395 bytes\n\nFiles: modules/classification-system.md (8999b), modules/configuration.md (6593b), modules/multi-metric-evaluation-methodology.md (8759b), modules/research-enrichment.md (5903b), modules/scoring-framework.md (10049b), modules/tradeoff-dimensions.md (10886b), skill-card.md (3026b), SKILL.md (12170b), _meta.json (143b)\n\nArchive v1.9.12: 9 files, 28294 bytes\n\nFiles: modules/classification-system.md (8999b), modules/configuration.md (6593b), modules/multi-metric-evaluation-methodology.md (8759b), modules/research-enrichment.md (5903b), modules/scoring-framework.md (10049b), modules/tradeoff-dimensions.md (10886b), skill-card.md (2719b), SKILL.md (12170b), _meta.json (143b)\n\nArchive v1.0.3: 9 files, 28341 bytes\n\nFiles: modules/classification-system.md (8999b), modules/configuration.md (6593b), modules/multi-metric-evaluation-methodology.md (8759b), modules/research-enrichment.md (5903b), modules/scoring-framework.md (10049b), modules/tradeoff-dimensions.md (10886b), skill-card.md (2859b), SKILL.md (12170b), _meta.json (142b)\n\nArchive v1.0.2: 8 files, 26933 bytes\n\nFiles: modules/classification-system.md (8975b), modules/configuration.md (19834b), modules/research-enrichment.md (5903b), modules/scoring-framework.md (10049b), modules/tradeoff-dimensions.md (10878b), skill-card.md (2554b), SKILL.md (11950b), _meta.json (142b)\n\nArchive v1.0.1: 7 files, 25624 bytes\n\nFiles: modules/classification-system.md (8975b), modules/configuration.md (19834b), modules/research-enrichment.md (5903b), modules/scoring-framework.md (10049b), modules/tradeoff-dimensions.md (10878b), SKILL.md (11950b), _meta.json (142b)\n\nArchive v1.0.0: 7 files, 25624 bytes\n\nFiles: modules/classification-system.md (8975b), modules/configuration.md (19834b), modules/research-enrichment.md (5903b), modules/scoring-framework.md (10049b), modules/tradeoff-dimensions.md (10878b), SKILL.md (11950b), _meta.json (142b)","readmeExcerpt":"Skill: feature-review Owner: athola Summary: Scores backlog items with RICE/WSJF/Kano and files GitHub issues for top candidates Tags: latest:1.9.19 Version history: v1.9.19 | 2026-08-26T13:12:29.298Z | user Release v1.9.19 v1.9.17 | 2026-07-30T05:34:08.720Z | user Release v1.9.17 v1.9.16 | 2026-07-14T19:50:52.775Z | user Release v1.9.16 v1.9.14 | 2026-06-30T17:59:58.487Z | user Release v1.9.14 v1.9.13 | 2026-06-27T1","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"/feature-review --inventory"},{"language":"bash","snippet":"/feature-review"},{"language":"bash","snippet":"/feature-review --suggest"},{"language":"bash","snippet":"/feature-review --research"},{"language":"bash","snippet":"/feature-review --suggest --create-issues"},{"language":"text","snippet":"Feature Score = Value Score / Cost Score\n\nValue Score = (Reach + Impact + Business Value + Time Criticality) / 4\nCost Score = (Effort + Risk + Complexity) / 3\n\nAdjusted Score = Feature Score * Confidence"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: feature-review\ndescription: |\n  Scores backlog items with RICE/WSJF/Kano and files GitHub issues for top candidates\nversion: 1.9.8\ntriggers:\n  - feature-prioritization\n  - backlog-triage\n  - RICE\n  - WSJF\n  - Kano\n  - roadmap\n  - triaging a roadmap or prioritizing features for a sprint\nmetadata: {\"openclaw\": {\"homepage\": \"https://github.com/athola/claude-night-market/tree/master/plugins/imbue\", \"emoji\": \"\\ud83e\\udd9e\", \"requires\": {\"config\": [\"night-market.imbue:scope-guard\"]}}}\nsource: claude-night-market\nsource_plugin: imbue\n---\n\n> **Night Market Skill** — ported from [claude-night-market/imbue](https://github.com/athola/claude-night-market/tree/master/plugins/imbue). For the full experience with agents, hooks, and commands, install the Claude Code plugin.\n\n\n## Table of Contents\n\n- [Philosophy](#philosophy)\n- [When to Use](#when-to-use)\n- [When NOT to Use](#when-not-to-use)\n- [Quick Start](#quick-start)\n- [1. Inventory Current Features](#1-inventory-current-features)\n- [2. Score and Classify](#2-score-and-classify)\n- [3. Generate Suggestions](#3-generate-suggestions)\n\n## Verification\n\nRun `make test-feature-review` to verify scoring logic after changes.\n- [4. Upload to GitHub](#4-upload-to-github)\n- [Workflow](#workflow)\n- [Phase 1: Feature Discovery (`feature-review:inventory-complete`)](#phase-1:-feature-discovery-(feature-review:inventory-complete))\n- [Phase 2: Classification (`feature-review:classified`)](#phase-2:-classification-(feature-review:classified))\n- [Phase 3: Scoring (`feature-review:scored`)](#phase-3:-scoring-(feature-review:scored))\n- [Phase 4: Tradeoff Analysis (`feature-review:tradeoffs-analyzed`)](#phase-4:-tradeoff-analysis-(feature-review:tradeoffs-analyzed))\n- [Phase 5: Gap Analysis & Suggestions (`feature-review:suggestions-generated`)](#phase-5:-gap-analysis-&-suggestions-(feature-review:suggestions-generated))\n- [Phase 6: GitHub Integration (`feature-review:issues-created`)](#phase-6:-github-integration-(feature-review:issues-created))\n- [Configuration](#configuration)\n- [Configuration File](#configuration-file)\n- [Guardrails](#guardrails)\n- [Required TodoWrite Items](#required-todowrite-items)\n- [Integration Points](#integration-points)\n- [Output Format](#output-format)\n- [Feature Inventory Table](#feature-inventory-table)\n- [Suggestion Report](#suggestion-report)\n- [Feature Suggestions](#feature-suggestions)\n- [High Priority (Score > 2.5)](#high-priority-(score->-25))\n- [Related Skills](#related-skills)\n- [Reference](#reference)\n\n\n# Feature Review\n\nReview implemented features and suggest new ones using evidence-based prioritization. Create GitHub issues for accepted suggestions.\n\n## Philosophy\n\nFeature decisions rely on data. Every feature involves tradeoffs that require evaluation. This skill uses hybrid RICE+WSJF scoring with Kano classification to prioritize work and generates actionable GitHub issues for accepted suggestions.\n\n## When To Use\n\n- Roadmap reviews (sprint planning, quarterly reviews).\n- Re"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn7d107jg9jv602h9ytsegydq184a42s\",\n  \"slug\": \"nm-imbue-feature-review\",\n  \"version\": \"1.9.19\",\n  \"publishedAt\": 1787749949298\n}"},{"path":"modules/classification-system.md","content":"# Classification System\n\nFeatures are classified along two orthogonal axes that determine architectural and UX implications.\n\n## Axis 1: Proactive vs Reactive\n\nThis axis describes **when** the feature acts relative to user intent.\n\n### Proactive Features\n\n**Definition:** Anticipates user needs and acts before explicit request.\n\n**Characteristics:**\n- Runs in background or ahead of user action\n- Requires prediction/inference of user intent\n- May consume resources speculatively\n- Higher latency tolerance (users don't wait)\n\n**Latency Tolerance:**\n- Background processing acceptable (seconds to minutes)\n- User doesn't perceive delay directly\n- Can be batched or deferred\n\n**Examples:**\n| Feature | How It's Proactive |\n|---------|-------------------|\n| Auto-save | Saves before user requests |\n| Prefetching | Loads data before navigation |\n| Suggestions | Offers options before user types |\n| Health checks | Monitors before problems occur |\n| Cache warming | Prepares data before access |\n\n**Tradeoffs:**\n| Pro | Con |\n|-----|-----|\n| Reduces user effort | May waste resources |\n| Feels \"smart\" | Can be wrong/intrusive |\n| Prevents problems | Requires more data |\n| Smoother UX | Higher complexity |\n\n**Architecture Patterns:**\n- Event-driven / pub-sub\n- Background workers\n- Predictive models\n- Eventual consistency acceptable\n\n### Reactive Features\n\n**Definition:** Responds to explicit user input or system events.\n\n**Characteristics:**\n- Triggered by user action\n- Must feel immediate\n- Resources used on-demand\n- Correctness over speculation\n\n**Latency Tolerance:**\n- Sub-100ms for UI feedback\n- Sub-1s for completion\n- User actively waiting\n\n**Examples:**\n| Feature | How It's Reactive |\n|---------|------------------|\n| Form submission | User clicks submit |\n| Search | User types query |\n| Navigation | User clicks link |\n| Validation | User enters input |\n| Commands | User invokes action |\n\n**Tradeoffs:**\n| Pro | Con |\n|-----|-----|\n| User in control | User must initiate |\n| Predictable behavior | No anticipation |\n| Lower resource waste | Perceived latency |\n| Simpler to implement | Less \"magical\" UX |\n\n**Architecture Patterns:**\n- Request/response\n- Synchronous processing\n- Strong consistency\n- Direct invocation\n\n### Classification Decision Tree\n\n```\nIs the feature triggered by explicit user action?\n├── Yes → Is immediate response critical?\n│   ├── Yes → REACTIVE\n│   └── No → Could be either (consider UX goals)\n└── No → Does it require user data/context?\n    ├── Yes → PROACTIVE (with data)\n    └── No → PROACTIVE (autonomous)\n```\n\n## Axis 2: Static vs Dynamic\n\nThis axis describes **how** feature data changes over time.\n\n### Static Features\n\n**Definition:** Data changes incrementally through explicit updates.\n\n**Characteristics:**\n- Version-controlled or release-based updates\n- Can be cached aggressively\n- Deterministic lookups\n- Stale data possible but predictable\n\n**Update Pattern:**\n- Deploy-time updates\n- Batch processing\n- Periodic refresh\n- Manual triggers"},{"path":"modules/configuration.md","content":"# Configuration\n\nFeature-review uses opinionated defaults but allows\nproject-specific customization through a YAML\nconfiguration file.\n\n## Configuration File Location\n\nCreate `.feature-review.yaml` in your project root:\n\n```\nproject/\n├── .feature-review.yaml    # Configuration file\n├── src/\n└── ...\n```\n\n## Full Configuration Schema\n\n```yaml\n# .feature-review.yaml\n# All values shown are defaults - only specify what you want to change\n\nversion: 1  # Schema version (required if file exists)\n\nweights:\n  value:\n    reach: 0.25              # How many users affected\n    impact: 0.30             # How much improvement per user\n    business_value: 0.25     # OKR/strategic alignment\n    time_criticality: 0.20   # Cost of delay\n  cost:\n    effort: 0.40             # Development time\n    risk: 0.30               # Uncertainty/unknowns\n    complexity: 0.30         # Technical difficulty\n\nthresholds:\n  high_priority: 2.5         # Score > 2.5 = implement soon\n  medium_priority: 1.5       # Score > 1.5 = roadmap candidate\n  confidence_warning: 0.5    # Scores below this get flagged\n\nclassification:\n  default_type: reactive     # proactive | reactive\n  default_data: static       # static | dynamic\n  patterns:\n    proactive_patterns: [\"*auto*\", \"*suggest*\", \"*predict*\", \"*prefetch*\"]\n    dynamic_patterns: [\"*session*\", \"*realtime*\", \"*live*\", \"*stream*\"]\n\ntradeoffs:\n  quality: 1.0               # Correctness of results\n  latency: 1.0               # Response time\n  token_usage: 1.0           # Context efficiency (LLM-specific)\n  resource_usage: 0.8        # CPU/memory consumption\n  redundancy: 0.5            # Fault tolerance\n  readability: 1.0           # Code maintainability\n  scalability: 0.8           # Growth handling\n  integration: 1.0           # Ecosystem fit\n  api_surface: 1.0           # Contract stability\n\ngithub:\n  enabled: true\n  auto_label: true\n  label_prefix: \"priority/\"\n  default_labels: [enhancement, feature-review]\n  priority_labels:\n    high: \"priority/high\"\n    medium: \"priority/medium\"\n    low: \"priority/low\"\n\ninventory:\n  scan_paths: [\"commands/\", \"skills/\", \"agents/\", \"src/\"]\n  exclude_patterns: [\"**/test*\", \"**/mock*\", \"**/__pycache__/**\"]\n\noutput:\n  format: markdown           # markdown | json | yaml\n  include_rationale: true\n  include_tradeoffs: true\n  max_suggestions: 10\n\nbacklog:\n  max_items: 25              # Maximum items (guardrail, cannot exceed 25)\n  stale_days: 30\n  auto_archive: false\n  file: \"docs/backlog/feature-queue.md\"\n```\n\n## Minimal Configuration Examples\n\n### Startup (Move Fast)\n\n```yaml\nversion: 1\nthresholds:\n  high_priority: 2.0\n  medium_priority: 1.0\ntradeoffs:\n  redundancy: 0.3\n  scalability: 0.5\nbacklog:\n  max_items: 15\n```\n\n### Enterprise (Stability First)\n\n```yaml\nversion: 1\nthresholds:\n  high_priority: 3.0\n  confidence_warning: 0.7\ntradeoffs:\n  api_surface: 1.5\n  redundancy: 1.2\n  readability: 1.2\n```\n\n## Project-Type Templates\n\nAdjust tradeoff weights based on project type:\n\n| Project Type | Key Weight Adjustm"},{"path":"modules/multi-metric-evaluation-methodology.md","content":"# Multi-Metric Evaluation Methodology\n\nHow to combine RICE, WSJF, Kano, and related models when\nprioritizing a feature backlog. Each model encodes a\ndifferent assumption about what makes a feature worth\nbuilding. This module shows the formulas, where each model\nfits, and how to combine them when no single model is\nenough on its own.\n\n## The Models at a Glance\n\n| Model | Origin | Output | Captures |\n|-------|--------|--------|----------|\n| RICE | Intercom (Sean McBride, 2017) | Number | Reach * Impact * Confidence / Effort |\n| WSJF | SAFe (Scaled Agile) | Number | (Value, Time, and Risk) / Effort |\n| Kano | Noriaki Kano (1984) | Category | Basic, Performance, Delighter, Indifferent, Reverse |\n| MoSCoW | DSDM Consortium (1994) | Bucket | Must, Should, Could, Won't |\n| Cost-of-Delay | Don Reinertsen (2009) | $ / week | Value lost per week of delay |\n\nSingle-model use is rare in practice. Most teams converge\non a hybrid: RICE for a base score, WSJF to raise time-\ncritical items, Kano to gate basics. The rest of this\nmodule explains why and how.\n\n## Model 1: RICE\n\n```\nRICE = (Reach * Impact * Confidence) / Effort\n```\n\n| Factor | Unit | Typical scale |\n|--------|------|---------------|\n| Reach | users / period | absolute count |\n| Impact | satisfaction delta | 0.25, 0.5, 1, 2, 3 |\n| Confidence | probability | 0.5, 0.8, 1.0 |\n| Effort | person-months | 0.5, 1, 2, 5, 10 |\n\n**Best for**: large user-facing roadmaps where reach is\nmeasurable and a single team can absorb most items.\n\n**Worst for**: backlogs dominated by infrastructure or\ncompliance work where \"reach\" is meaningless or every item\nshares similar reach.\n\n**Worked example**:\n\n```\nFeature: Auto-save drafts\n  Reach:      8,000 users / quarter\n  Impact:     1.0 (significant satisfaction)\n  Confidence: 0.8\n  Effort:     2 person-months\n\nRICE = (8000 * 1.0 * 0.8) / 2 = 3200\n```\n\n## Model 2: WSJF\n\nWeighted Shortest Job First. From SAFe; treats\nprioritization as a cost-of-delay optimization.\n\n```\nWSJF = Cost_of_Delay / Job_Size\n\nCost_of_Delay = User_Value + Time_Criticality + Risk_Reduction\nJob_Size      = Effort estimate\n```\n\nEach input uses a Fibonacci scale: 1, 2, 3, 5, 8, 13, 20.\n\n| Factor | Question |\n|--------|----------|\n| User_Value | How much does the user/business gain? |\n| Time_Criticality | What does delay cost? Does the value decay? |\n| Risk_Reduction | Does this open future options or de-risk? |\n| Job_Size | How much work? |\n\n**Best for**: backlogs with strong time pressure and many\nitems where deferral has measurable cost. Common in\nSAFe-aligned organizations.\n\n**Worst for**: small teams without explicit\ncost-of-delay numbers; reduces to \"gut feel times Fibonacci\".\n\n**Worked example**:\n\n```\nFeature: GDPR consent banner\n  User_Value:        5\n  Time_Criticality:  20  (regulatory deadline)\n  Risk_Reduction:    13\n  Job_Size:          3\n\nWSJF = (5 + 20 + 13) / 3 = 12.67\n```\n\nCompare with the auto-save example: WSJF would put\nauto-save at roughly (8 + 3 + 2) / 5 = 2.6, far below the\nGDPR ite"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Scores backlog items with RICE/WSJF/Kano and files GitHub issues for top candidates Skill: feature-review Owner: athola Summary: Scores backlog items with RICE/WSJF/Kano and files GitHub issues for top candidates Tags: latest:1.9.19 Version history: v1.9.19 | 2026-08-26T13:12:29.298Z | user Release v1.9.19 v1.9.17 | 2026-07-30T05:34:08.720Z | user Release v1.9.17 v1.9.16 | 2026-07-14T19:50:52.775Z | user Release v1.9.16 v1.9.14 | 2026-06-30T17:59:58.487Z | user Release v1.9.14 v1.9.13 | 2026-06-27T1","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1377,"uniquenessScore":56,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-10T07:51:01.745Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-10T07:51:01.745Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T08:12:50.985Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}