{"id":"827f3e10-6c6c-48b4-9f44-54111d8970a5","entityType":"agent","slug":"clawhub-andyrenxu7255-academic-paper-reviewer","name":"Academic Paper Reviewer","canonicalUrl":"https://www.xpersona.co/agent/clawhub-andyrenxu7255-academic-paper-reviewer","canonicalPath":"/agent/clawhub-andyrenxu7255-academic-paper-reviewer","generatedAt":"2026-10-10T08:23:15.215Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-09T19:28:59.608Z","emptyReason":null},"description":"7-agent paper review system on Hermes Agent. 6 modes (full/re-review/quick/methodology-focus/guided/calibration). 5-panel review with editorial decision, rev... Skill: Academic Paper Reviewer Owner: andyrenxu7255 Summary: 7-agent paper review system on Hermes Agent. 6 modes (full/re-review/quick/methodology-focus/guided/calibration). 5-panel review with editorial decision, rev... Tags: academic:1.0.4, cc-by-nc:1.0.2, editorial:1.0.4, hermes:1.0.4, latest:1.0.4, manuscript:1.0.4, multi-agent:1.0.4, paper-review:1.0.4, peer-review:1.0.4, research:1.0.4, review:1.0.4 Version hi","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 2.1K downloads reported by the source. Last updated 10/9/2026.","installCommand":"clawhub skill install s1721fhgyy0yf38f9gfpbjxrn984fcz0:academic-paper-reviewer","sourceUrl":"https://clawhub.ai/andyrenxu7255/academic-paper-reviewer","homepage":"https://clawhub.ai/andyrenxu7255/skills/academic-paper-reviewer","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/andyrenxu7255/academic-paper-reviewer","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/andyrenxu7255/skills/academic-paper-reviewer","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":66,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"7-agent paper review system on Hermes Agent. 6 modes (full/re-review/quick/methodology-focus/guided/calibration). 5-panel review with editorial decision, rev..."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-09T19:28:59.608Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T19:28:59.608Z","emptyReason":null},"stars":null,"forks":null,"downloads":2077,"packageName":null,"latestVersion":"1.0.4","tractionLabel":"2.1K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T19:28:59.608Z","emptyReason":null},"lastUpdatedAt":"2026-10-09T19:28:59.608Z","lastCrawledAt":"2026-10-09T19:28:59.608Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-10T19:28:59.608Z","lastVerifiedAt":null,"highlights":[{"version":"1.0.4","createdAt":"2026-05-23T01:49:16.225Z","changelog":"## 1.0.4 — Moderation Re-trigger + License Compliance - Added ATTRIBUTION.md (required for CC BY-NC 4.0 adapted works) - Re-publish to trigger LLM moderation review (stuck with review.llm_review since 1.0.3)","fileCount":41,"zipByteSize":159324},{"version":"1.0.3","createdAt":"2026-05-17T00:05:43.072Z","changelog":"## 1.0.3 - Trigger Re-review Re-publishing to trigger ClawScan re-review. No content changes.","fileCount":39,"zipByteSize":156982},{"version":"1.0.2","createdAt":"2026-05-16T15:00:25.661Z","changelog":"## 1.0.2 — ClawScan Compliance - Removed 'terminal' from all delegate_task toolsets (file-only now) - Added Security & Privacy section disclosing multi-agent design - Added explicit notice that agent files are task instructions, not system prompt overrides - ClawScan note added explaining prompt template design pattern","fileCount":39,"zipByteSize":156982},{"version":"1.0.1","createdAt":"2026-05-16T14:51:24.873Z","changelog":"## 1.0.1 — License Correction **IMPORTANT:** Corrected license from MIT-0 to **CC BY-NC 4.0** to comply with original author's license. - Added full attribution to Cheng-I Wu (original creator) - Added CC BY-NC 4.0 LICENSE file to distribution - Added copyright notice and GitHub link to SKILL.md frontmatter - Added 'Adapted for Hermes Agent' modification notice","fileCount":39,"zipByteSize":156627},{"version":"1.0.0","createdAt":"2026-05-16T14:37:59.262Z","changelog":"## 1.0.0 — Initial release for Hermes Agent Adapted from imbad0202/academic-research-skills for Hermes Agent. **Key Features:** - 7-agent peer review system via delegate_task - 6 modes: full/re-review/quick/methodology-focus/guided/calibration - 5-panel parallel review (method/evidence/argument/domain) - Editorial decision with weighted scoring - Revision Roadmap with prioritized action items **Adaptation:** Reviewers run as parallel delegate_task batches. Agent definitions and references preserved unchanged.","fileCount":38,"zipByteSize":149905}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s1721fhgyy0yf38f9gfpbjxrn984fcz0:academic-paper-reviewer","setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-andyrenxu7255-academic-paper-reviewer/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-andyrenxu7255-academic-paper-reviewer/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-andyrenxu7255-academic-paper-reviewer/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-andyrenxu7255-academic-paper-reviewer/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-andyrenxu7255-academic-paper-reviewer/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-andyrenxu7255-academic-paper-reviewer/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-10T08:23:15.210Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-andyrenxu7255-academic-paper-reviewer/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-andyrenxu7255-academic-paper-reviewer/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-andyrenxu7255-academic-paper-reviewer/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-andyrenxu7255-academic-paper-reviewer/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-09T19:28:59.608Z","emptyReason":null},"readme":"Skill: Academic Paper Reviewer\n\nOwner: andyrenxu7255\n\nSummary: 7-agent paper review system on Hermes Agent. 6 modes (full/re-review/quick/methodology-focus/guided/calibration). 5-panel review with editorial decision, rev...\n\nTags: academic:1.0.4, cc-by-nc:1.0.2, editorial:1.0.4, hermes:1.0.4, latest:1.0.4, manuscript:1.0.4, multi-agent:1.0.4, paper-review:1.0.4, peer-review:1.0.4, research:1.0.4, review:1.0.4\n\nVersion history:\n\nv1.0.4 | 2026-05-23T01:49:16.225Z | user\n\n## 1.0.4 — Moderation Re-trigger + License Compliance\n\n- Added ATTRIBUTION.md (required for CC BY-NC 4.0 adapted works)\n- Re-publish to trigger LLM moderation review (stuck with review.llm_review since 1.0.3)\n\nv1.0.3 | 2026-05-17T00:05:43.072Z | user\n\n## 1.0.3 - Trigger Re-review\n\nRe-publishing to trigger ClawScan re-review. No content changes.\n\nv1.0.2 | 2026-05-16T15:00:25.661Z | user\n\n## 1.0.2 — ClawScan Compliance\n\n- Removed 'terminal' from all delegate_task toolsets (file-only now)\n- Added Security & Privacy section disclosing multi-agent design\n- Added explicit notice that agent files are task instructions, not system prompt overrides\n- ClawScan note added explaining prompt template design pattern\n\nv1.0.1 | 2026-05-16T14:51:24.873Z | user\n\n## 1.0.1 — License Correction\n\n**IMPORTANT:** Corrected license from MIT-0 to **CC BY-NC 4.0** to comply with original author's license.\n\n- Added full attribution to Cheng-I Wu (original creator)\n- Added CC BY-NC 4.0 LICENSE file to distribution\n- Added copyright notice and GitHub link to SKILL.md frontmatter\n- Added 'Adapted for Hermes Agent' modification notice\n\nv1.0.0 | 2026-05-16T14:37:59.262Z | user\n\n## 1.0.0 — Initial release for Hermes Agent\n\nAdapted from imbad0202/academic-research-skills for Hermes Agent.\n\n**Key Features:**\n- 7-agent peer review system via delegate_task\n- 6 modes: full/re-review/quick/methodology-focus/guided/calibration\n- 5-panel parallel review (method/evidence/argument/domain)\n- Editorial decision with weighted scoring\n- Revision Roadmap with prioritized action items\n\n**Adaptation:** Reviewers run as parallel delegate_task batches. Agent definitions and references preserved unchanged.\n\nArchive index:\n\nArchive v1.0.4: 41 files, 159324 bytes\n\nFiles: _meta.json (142b), agents/devils_advocate_reviewer_agent.md (15011b), agents/domain_reviewer_agent.md (11729b), agents/editorial_synthesizer_agent.md (13139b), agents/eic_agent.md (8438b), agents/field_analyst_agent.md (9429b), agents/methodology_reviewer_agent.md (13196b), agents/perspective_reviewer_agent.md (14358b), ATTRIBUTION.md (1278b), LICENSE (19584b), references/calibration_mode_protocol.md (10542b), references/changelog.md (976b), references/editorial_decision_standards.md (9230b), references/guided_mode_protocol.md (1971b), references/integration_guide.md (680b), references/quality_rubrics.md (7679b), references/re_review_mode_protocol.md (4250b), references/review_criteria_framework.md (10139b), references/review_quality_thinking.md (3015b), references/sprint_contract_protocol.md (11621b), references/statistical_reporting_standards.md (22703b), references/top_journals_by_field.md (12874b), shared/artifact_reproducibility_pattern.md (8921b), shared/benchmark_report_pattern.md (8988b), shared/benchmark_report.schema.json (2789b), shared/collaboration_depth_rubric.md (10810b), shared/compliance_checkpoint_protocol.md (7175b), shared/compliance_report.schema.json (8314b), shared/cross_model_verification.md (11173b), shared/ground_truth_isolation_pattern.md (12649b), shared/handoff_schemas.md (46013b), shared/mode_spectrum.md (4047b), shared/prisma_trAIce_protocol.md (10415b), shared/raise_framework.md (7365b), shared/sprint_contract.schema.json (18954b), shared/style_calibration_protocol.md (7085b), skill-card.md (3127b), SKILL.md (4977b), templates/editorial_decision_template.md (6446b), templates/peer_review_report_template.md (7479b), templates/revision_response_template.md (7020b)\n\nFile v1.0.4:SKILL.md\n\n---\nname: academic-paper-reviewer\ndescription: \"7-agent paper review system on Hermes Agent. 6 modes (full/re-review/quick/methodology-focus/guided/calibration). 5-panel review with editorial decision, revision roadmap, and calibration metrics. Uses delegate_task for each reviewer. Triggers: review paper, peer review, manuscript review, check revisions, calibrate reviewer, 審稿, 同儕審查, 論文審查.\"\nmetadata:\n  version: \"1.0-hermes-1.0\"\n  last_updated: \"2026-05-16\"\n  status: active\n  adapted_from: \"imbad0202/academic-research-skills\"\n  adapted_for: \"Hermes Agent (deepseek-v4-pro)\"\n  task_type: open-ended\n  license: \"CC BY-NC 4.0\"\n  original_author: \"Cheng-I Wu\"\n  original_license: \"CC BY-NC 4.0\"\n  original_repo: \"https://github.com/Imbad0202/academic-research-skills\"\n  copyright: \"Copyright (c) 2026 Cheng-I Wu\"\n---\n# Academic Paper Reviewer — 7-Agent Review System (Hermes Edition)\n\n📄 **License:** [CC BY-NC 4.0](https://creativecommons.org/licenses/by-nc/4.0/) · Copyright (c) 2026 Cheng-I Wu  \n🔗 **Original:** [Imbad0202/academic-research-skills](https://github.com/Imbad0202/academic-research-skills)  \n🔄 **Adaptation:** Multi-agent review system implemented via `delegate_task` instead of Claude Code's internal agent system. All agent definitions, references, and quality standards preserved unchanged from original. **This adaptation is distributed under the same CC BY-NC 4.0 license.**\n\n## Quick Start\n\n```\nReview this paper for journal submission\n```\n\n## Agent Team\n\n| # | Agent | Role |\n|---|-------|------|\n| 1 | intake_agent | Receive paper, determine review type |\n| 2 | methodology_reviewer | Method rigor assessment |\n| 3 | evidence_reviewer | Evidence sufficiency & citation quality |\n| 4 | argument_reviewer | Logical coherence & argument structure |\n| 5 | domain_reviewer | Domain expertise & literature positioning |\n| 6 | editor_in_chief | Aggregate reviews → editorial decision |\n| 7 | revision_coach | Convert reviews → actionable roadmap |\n\n## Hermes Execution\n\n### Full Mode: 5-Panel Parallel Review\n```\ndelegate_task(tasks=[\n    {\"goal\": \"Review manuscript methodology: design appropriateness, validity threats, replicability. Score 1-5.\", \"context\": \"Use agents/methodology_reviewer.md\", \"toolsets\": [\"file\"]},\n    {\"goal\": \"Review evidence: citation quality, source credibility, evidence hierarchy alignment. Score 1-5.\", \"context\": \"Use agents/evidence_reviewer.md\", \"toolsets\": [\"file\"]},\n    {\"goal\": \"Review argument: logical flow, claim-evidence alignment, counter-argument handling. Score 1-5.\", \"context\": \"Use agents/argument_reviewer.md\", \"toolsets\": [\"file\"]},\n    {\"goal\": \"Review domain positioning: literature coverage, theoretical grounding, contribution significance. Score 1-5.\", \"context\": \"Use agents/domain_reviewer.md\", \"toolsets\": [\"file\"]}\n])\n```\n\n### Editorial Decision\n```\ndelegate_task(goal=\"Aggregate all 4 reviewer reports. Apply weighted scoring (Method 30%, Evidence 25%, Argument 25%, Domain 20%). Issue editorial decision: Accept/Minor Revision/Major Revision/Reject with justification.\", context=\"Use agents/editor_in_chief.md\", toolsets=[\"file\"])\n```\n\n### Revision Roadmap\n```\ndelegate_task(goal=\"Convert editorial decision + reviewer reports into structured Revision Roadmap: prioritized action items, estimated effort, dependency mapping.\", context=\"Use agents/revision_coach.md\", toolsets=[\"file\"])\n```\n\n## 6 Modes\n\n| Mode | Trigger | Agents |\n|------|---------|--------|\n| `full` | \"Review paper\" | All 7 |\n| `re-review` | \"Check revisions\" | 2→3→4→6 |\n| `quick` | \"Quick review\" | 6 only (EIC assessment) |\n| `methodology-focus` | \"Check methodology\" | 2 only |\n| `guided` | \"Guide me to improve\" | Socratic: 6 with user interaction |\n| `calibration` | \"Calibrate reviewer\" | All + calibration metrics output |\n\n## Calibration Mode\nMeasures reviewer accuracy: FNR (False Negative Rate), FPR (False Positive Rate), AUC. Requires ground-truth labels on prior reviewed papers.\n\n## Critical Rules\n1. ⚠️ Reviewers are paper-blind (don't see author info)\n2. ⚠️ Every criticism must include specific actionable suggestion\n3. ⚠️ Calibration mode requires 5+ ground-truth papers\n\n## Security & Privacy\n\n**Multi-agent design disclosure:** This skill delegates review tasks across multiple subagents via `delegate_task`. Manuscript content and intermediate review outputs are processed by these agents. Use only with manuscripts you are comfortable having processed through the AI provider's delegated-agent workflow. Remove confidential material not needed for review.\n\n**Tool access:** Subagents are granted only `file` tools for reading/writing review outputs. No terminal, web, or system tools are exposed.\n\n**Agent files:** The `agents/` directory contains academic peer-review prompt templates (role definitions, scoring rubrics, methodology guidelines). These are task instructions loaded as `context` in `delegate_task` calls — NOT system prompt overrides.\n\nFile v1.0.4:_meta.json\n\n{\n  \"ownerId\": \"kn725tchg3gp72a78w22qa07p584f267\",\n  \"slug\": \"academic-paper-reviewer\",\n  \"version\": \"1.0.4\",\n  \"publishedAt\": 1779500956225\n}\n\nFile v1.0.4:references/calibration_mode_protocol.md\n\n# Calibration Mode Protocol\n\n**Status**: v3.2\n**Parent skill**: `academic-paper-reviewer`\n**Mode name**: `calibration`\n**Purpose**: Measure this reviewer's own false-negative rate (FNR), false-positive rate (FPR), and balanced accuracy against a user-supplied gold-standard set, then attach the resulting error profile as a confidence disclosure to subsequent reviews in the same session.\n\n---\n\n## Why this mode exists\n\nA single LLM reviewer produces an absolute 0-100 rubric score, but that score is weakly interpretable without knowing the reviewer's error profile. Two reviewers could give the same paper a 65, yet one might systematically over-score weak methodology papers and the other might systematically under-score cross-disciplinary work. Absolute scores don't reveal this.\n\nLu et al. (2026, Nature 651:914-919) demonstrated in Table 1 that an LLM-based Automated Reviewer can approach human balanced accuracy (0.65 vs human 0.67-0.73 on 500 ICLR 2022 papers) while having a dramatically different error profile: FNR 0.17 vs human 0.52, at the cost of FPR 0.50 vs human 0.17-0.34. Human reviewers miss half of the papers that should be rejected; the Automated Reviewer misses very few but over-rejects more.\n\nTranslation for ARS: **our reviewer has an error profile too, and we do not currently measure it.** Calibration mode closes that gap. It does not try to make the reviewer perfect; it makes the reviewer's imperfections legible.\n\n---\n\n## Inputs\n\n1. **Gold-standard set**: 5-20 papers the user has labelled with known outcomes. Minimum 5; recommended 10-15. Each entry:\n   - Paper file path or text\n   - Ground-truth label: `accept`, `reject`, or `borderline`\n   - Venue context (journal/conference, tier)\n   - Optional: human reviewer scores for comparison\n\n2. **Domain specification**: the user's target field, used to seed `field_analyst_agent`. Calibration for \"machine learning venues\" is not valid for \"qualitative education research\" — error profiles are domain-specific.\n\n3. **Session persistence**: the error profile is cached for the **current session only**. No cross-session caching, no `~/.ars_calibration_cache/` directory. Calibration is explicitly opt-in per the v3.2 design decision: the user decides when to spend tokens on calibration, and a new session starts fresh. If the user wants to reuse a profile across sessions, they re-run calibration or paste a prior Calibration Report as a session prompt.\n\n---\n\n## Process\n\n### Phase 0: Intake\n\n- Verify the set has at least one `accept` and one `reject` (otherwise FNR or FPR is undefined).\n- If all labels are on one side, refuse to proceed and ask the user for at least one counter-example.\n- Warn if n < 10: \"Calibration with fewer than 10 papers produces wide confidence intervals. Results should be treated as directional, not conclusive.\"\n\n### Phase 1: Run `full` mode on each gold paper, with ensembling\n\nFor each paper, run the standard `full` review pipeline **5 times** (ensembling, per Lu 2026 Methods A.1.1). Each run uses a fresh context window to avoid within-session bias. Aggregate:\n- Median rubric score per dimension\n- Variance across the 5 runs (reported as a stability indicator)\n- Editorial decision (majority vote across 5)\n\n**Cross-model verification**: In calibration mode, `ARS_CROSS_MODEL` is **default-on** rather than opt-in. At least one of the 5 runs should use a different model family if available, to avoid single-model blind spots. If no cross-model is configured, emit a warning and run all 5 on the primary model.\n\n### Phase 2: Build the confusion matrix\n\nCompare reviewer's majority-vote decision against the user's ground-truth label.\n\n- `borderline` ground truth papers are excluded from the binary confusion matrix but reported separately (see Phase 3).\n- Map `Accept` and `Minor Revision` reviewer decisions → positive. Map `Major Revision` and `Reject` → negative. This follows Lu 2026 Table 1's binarization.\n\nCompute:\n\n| Metric | Formula | Report with |\n|---|---|---|\n| Balanced accuracy | (TPR + TNR) / 2 | 95% CI via bootstrap (1000 resamples) |\n| FNR (miss rate) | FN / (FN + TP) | Same |\n| FPR (false alarm) | FP / (FP + TN) | Same |\n| AUC | ROC over rubric-score threshold | Same |\n| Calibration error | Mean &#124;rubric_score - ground_truth_severity&#124; | Per-dimension |\n\n### Phase 3: Borderline handling\n\nBorderline papers don't enter the binary matrix but are useful for rubric-score calibration. For each borderline paper, report:\n- The reviewer's rubric score\n- The reviewer's decision\n- Whether the reviewer's decision respects the user's \"this is borderline\" signal (i.e., did it correctly land in Major Revision rather than confidently Accept or Reject?)\n\nA reviewer that confidently Accepts or Rejects borderline papers has a \"confidence miscalibration\" problem even if its binary accuracy looks fine.\n\n### Phase 4: Produce the Calibration Report\n\nOutput document structured as:\n\n```\n# Calibration Report for <Reviewer Instance>\nDomain: <domain>\nGold set: n=<N> (accept=<a>, reject=<r>, borderline=<b>)\nRuns per paper: 5 (ensembled)\nCross-model: <yes/no, model families used>\n\n## Summary metrics\n- Balanced accuracy: 0.XX [95% CI: 0.XX - 0.XX]\n- FNR: 0.XX [95% CI ...]\n- FPR: 0.XX [95% CI ...]\n- AUC: 0.XX\n- Ensemble stability: <mean std of rubric scores across runs>\n\n## Comparison to Lu 2026 Table 1 baselines\n| Metric | This reviewer | Lu 2026 Automated Reviewer | Lu 2026 Human |\n|---|---|---|---|\n| Balanced accuracy | X | 0.65 | 0.67-0.73 |\n| FNR | X | 0.17 | 0.52 |\n| FPR | X | 0.50 | 0.17-0.34 |\n\n(Note: Lu 2026 numbers are for ML venues specifically. Compare with caution outside ML.)\n\n## Per-dimension calibration error\n<table of 7 review dimensions with mean absolute calibration error>\n\n## Systematic biases detected\n<natural-language narrative identifying patterns, e.g.\n \"Reviewer tends to over-score originality on cross-disciplinary papers\"\n \"Reviewer under-scores qualitative methodology by ~8 points vs ground truth\"\n>\n\n## Recommendations for session use\n- Treat this reviewer's rubric scores as having calibration error ±X points\n- For accept/reject decisions, the reviewer misses X% of reject cases (FNR)\n- For decisions near the accept/reject boundary, escalate to human judgement\n```\n\n### Phase 5: Session attachment\n\nIf session persistence is enabled, the Calibration Report is attached to every subsequent review in the same session as a **confidence disclosure header**. The disclosure appears in the editorial letter before the verdict:\n\n```\n> **Reviewer Confidence Disclosure (from calibration session <id>):**\n> This reviewer has measured balanced accuracy 0.XX, FNR 0.XX, FPR 0.XX on a\n> gold set of <N> papers in <domain>. Rubric scores below have calibration\n> error ±X points. Treat borderline decisions with human judgement.\n```\n\nThis is non-negotiable in calibration-enabled sessions: the user cannot hide the disclosure. The point of calibration is to make error profiles legible; suppressing the disclosure defeats the mode.\n\n---\n\n## Ensembling methodology notes\n\nLu 2026 Methods A.1.1 describes reviewer ensembling across 5 independent runs with majority voting. This mode follows that spec with two changes:\n\n1. **Median instead of mean for rubric scores**: mean is vulnerable to single-run outliers (e.g., a run that hallucinates a methodological flaw); median is robust.\n2. **Fresh context per run**: Lu 2026 allowed within-session memory across runs. ARS uses fresh context to prevent cascading errors from a single run's misreading.\n\nUsers with token budget concerns can reduce `runs_per_paper` to 3. Below 3, ensembling is meaningless — do not allow 1 or 2.\n\n---\n\n## Failure cases this mode does NOT fix\n\nCalibration reports this reviewer's error profile on a **specific** gold set in a **specific** domain. It does not:\n\n- Predict performance on papers outside that domain\n- Detect frame-lock within a single paper review (that's `devils_advocate_reviewer` territory)\n- Catch implementation-bug-as-finding cases (that's the AI Research Failure Mode Checklist, ROADMAP_v3.2.md item 2)\n- Replace the `re-review` mode for revision verification\n\nIf the user's gold set is itself biased (e.g., all papers from one lab, all from one year), calibration reports a biased profile. Emit a warning during intake if papers share obvious metadata clusters.\n\n---\n\n## Integration with existing modes\n\n| Existing mode | Interaction with calibration |\n|---|---|\n| `full` | Calibration runs `full` 5x per gold paper. No change to `full` itself. |\n| `re-review` | Calibration profile attaches to re-review decisions. |\n| `quick` | Calibration profile attaches. Confidence disclosure notes that `quick` has additional uncalibrated error on top of the measured profile. |\n| `methodology-focus` | Calibration should ideally be run with methodology-heavy gold papers if this mode is the user's target. |\n| `guided` | Not applicable — guided mode is Socratic dialogue, rubric scores are not the primary output. |\n\n---\n\n## Resolved design decisions (2026-04-09)\n\n- **Activation**: opt-in only. User invokes `calibration` mode explicitly. ARS does not auto-calibrate on first use in a new domain.\n- **Persistence**: session-scoped only. No cross-session caching of profiles, no `~/.ars_calibration_cache/`, no privacy questions about storing paper content on disk.\n- **Shipped gold sets**: not planned for v3.2. Users bring their own gold set. Shipping a built-in ML gold set was considered and rejected to avoid domain-coverage bias and staleness.\n- **Continuous/self-calibration**: rejected. Using the reviewer's own historical decisions as pseudo-ground-truth is circular and would make the error profile look better over time without actually improving accuracy.\n\n---\n\n## References\n\n- Lu, C. et al. (2026). Towards end-to-end automation of AI research. *Nature* 651, 914-919. doi:10.1038/s41586-026-10265-5 — Table 1 (reviewer validation), Methods A.1.1 (ensembling).\n- Efron, B. & Tibshirani, R. J. (1993). *An Introduction to the Bootstrap*. Chapman & Hall/CRC — bootstrap CI methodology.\n- ARS `shared/cross_model_verification.md` — cross-model reviewer integration.\n- ARS `academic-paper-reviewer/references/quality_rubrics.md` — scoring rubric definitions.\n\n## v3.6.2 sprint contract status\n\nv3.6.2 introduces sprint contracts for `reviewer_full` and `reviewer_methodology_focus` only. A template for this mode will follow in a subsequent patch release. Until then, this mode runs without contract enforcement and retains its pre-v3.6.2 behaviour.\n\nFile v1.0.4:references/changelog.md\n\n# Changelog\n\n| Version | Date | Changes |\n|---------|------|---------|\n| 1.4 | 2026-03-08 | Quality rubrics reference (0-100 scoring with 5 descriptors per dimension, weighted aggregation formula, decision mapping); Quick Mode Selection Guide; Dimension Scores upgraded from optional 1-5 to required 0-100 with rubric descriptors |\n| 1.3 | 2026-03-05 | DA vs R3 role boundaries with explicit responsibility tables; CRITICAL finding criteria with concrete examples; Consensus classification (CONSENSUS-4/3/SPLIT/DA-CRITICAL); Confidence Score weighting rules; Asian & Regional Journals reference (TSSCI + Asia-Pacific + OA options) |\n| 1.2 | 2026-03 | Added statistical reporting standards reference; enhanced methodology_reviewer_agent with statistical reporting adequacy sub-step |\n| 1.1 | 2026-02 | Added Devil's Advocate Reviewer (7th agent), added re-review mode, expanded review team from 4 to 5 |\n| 1.0 | 2026-02 | Initial version: 6 agents, 4 modes, 3-phase workflow |\n\nFile v1.0.4:references/editorial_decision_standards.md\n\n# Editorial Decision Standards — Criteria for Editorial Decision Making\n\nThis document defines the explicit criteria for Accept / Minor Revision / Major Revision / Reject decisions, for use by `eic_agent` and `editorial_synthesizer_agent`.\n\n---\n\n## 1. Decision Categories\n\n### Accept\n\n**Definition**: The paper can be published without further review.\n\n**Criteria**:\n- Average score across all universal dimensions >= 4.0\n- No dimension scores below 3.0\n- At least 3/4 reviewers recommend Accept or Minor Revision\n- No unresolved major academic issues\n\n**Conditions**:\n- May include minor copyediting suggestions\n- May require final formatting adjustments\n- Does not need to be sent for review again\n\n**Typical scenarios**:\n- Paper has undergone multiple revision rounds, all issues resolved\n- Rare first-pass acceptance (< 5% of submissions at top-tier journals)\n\n---\n\n### Minor Revision\n\n**Definition**: The paper is fundamentally acceptable and can be published after limited modifications; typically does not need to be sent for review again after revision.\n\n**Criteria**:\n- Average score across all universal dimensions >= 3.5\n- No dimension scores below 2.5\n- At least 3/4 reviewers recommend Accept or Minor Revision\n- Issues can be resolved within 2-4 weeks\n- Modifications do not involve restructuring core arguments or methods\n\n**Typical revision items**:\n- Supplementing a small number of references\n- Clarifying certain methodology description details\n- Improving clarity of argumentation\n- Correcting citation format\n- Adding discussion of limitations\n- Adjusting conclusion wording (avoiding overclaiming)\n\n**Response requirements**:\n- Authors must respond to reviewer comments item by item\n- After revision, reviewed by EIC (usually not sent for external review again)\n- Revision deadline: 2-4 weeks\n\n---\n\n### Major Revision\n\n**Definition**: The paper has potential but has significant issues, requiring substantial revision followed by re-review.\n\n**Criteria**:\n- Universal dimension average score between 2.5-3.4\n- Some dimensions may score below 2.5 (but not fatal)\n- At least 2/4 reviewers recommend Major Revision or better\n- Issues are serious but fixable (not fundamental design flaws)\n- Revision requires 6-8 weeks of work\n\n**Typical revision items**:\n- Re-analyzing data (additional analysis or correcting errors)\n- Substantially rewriting literature review (missing key references)\n- Supplementing additional data collection\n- Reorganizing paper structure\n- Correcting significant methodological flaws\n- Strengthening theoretical framework application\n- Adding robustness checks\n\n**Response requirements**:\n- Authors must write a detailed point-by-point response letter\n- After revision, sent for re-review (may go back to original reviewers or new reviewers)\n- Revision deadline: 6-8 weeks\n- Typically a maximum of 2 rounds of Major Revision allowed\n\n---\n\n### Reject\n\n**Definition**: The paper is not suitable for publication in this journal, even with revision.\n\n**Criteria (meeting any one may trigger Reject consideration)**:\n- Universal dimension average score < 2.5\n- Any core dimension (methodology, evidence) = 1\n- At least 3/4 reviewers recommend Reject\n- Fundamental unfixable issues exist\n\n**Reject subtypes**:\n\n| Subtype | Description | Suggestion |\n|---------|-------------|-----------|\n| **Reject — Out of Scope** | Topic not within journal scope | Recommend more suitable journals |\n| **Reject — Fundamental Flaw** | Fatal flaw in research design | Suggest redesigning the research |\n| **Reject — Insufficient Contribution** | Lacks originality or incremental contribution | Suggest how to strengthen contribution |\n| **Reject — Premature** | Paper not yet mature enough | Suggest specific improvement directions |\n| **Reject — Resubmit Encouraged** | Has potential but needs fundamental restructuring | Provide detailed restructuring suggestions |\n\n**Even with Reject, must**:\n- Affirm the paper's merits\n- Provide specific improvement suggestions\n- Recommend more suitable journals (if it's a scope issue)\n- Maintain professional, respectful tone\n\n---\n\n## 2. Decision Matrix\n\n### Decision Matrix Based on Reviewer Recommendations\n\n| EIC | R1 | R2 | R3 | -> Recommended Decision |\n|-----|----|----|-----|----------------------|\n| Accept | Accept | Accept | Accept | **Accept** |\n| Accept | Accept | Accept | Minor | **Accept** (with suggestions) |\n| Accept | Accept | Minor | Minor | **Minor Revision** |\n| Accept | Minor | Minor | Minor | **Minor Revision** |\n| Minor | Minor | Minor | Minor | **Minor Revision** |\n| Minor | Minor | Minor | Major | **Minor-to-Major** (depends on specific issues) |\n| Minor | Minor | Major | Major | **Major Revision** |\n| Minor | Major | Major | Major | **Major Revision** |\n| Major | Major | Major | Major | **Major Revision** |\n| Major | Major | Major | Reject | **Major Revision** (last chance) |\n| Major | Major | Reject | Reject | **Reject** (resubmit encouraged) |\n| Major | Reject | Reject | Reject | **Reject** |\n| Reject | Reject | Reject | Reject | **Reject** |\n\n### Special Situation Handling\n\n**Split Decision (evenly divided)**:\n- Example: Accept + Accept + Reject + Reject\n- EIC (or synthesizer) needs to deeply analyze the cause of disagreement\n- Lean toward conservative strategy: Major Revision, requiring the author to respond to the Reject side's comments\n- May consider inviting a fifth reviewer\n\n**One Outlier (one unusual opinion)**:\n- Example: Minor + Minor + Minor + Reject\n- Carefully examine the Reject rationale\n- If the rationale is valid and others missed it, escalate to Major Revision\n- If the rationale is insufficient, maintain Minor Revision but mention the opinion in the Decision Letter\n\n---\n\n## 3. Decision Confidence Calibration\n\n### Impact of Reviewer Confidence Score\n\n| Confidence | Impact on Decision |\n|-----------|-------------------|\n| 5 (Very High) | This reviewer's opinion carries the highest weight |\n| 4 (High) | Standard weight |\n| 3 (Medium) | Standard weight, but reduced in case of disagreement |\n| 2 (Low) | For reference only, not used as a decisive opinion |\n| 1 (Very Low) | Ignore this reviewer's recommendation (but retain specific comments) |\n\n### Cross-Dimension Severity Assessment\n\n| Situation | Severity | Handling |\n|-----------|----------|---------|\n| Methodology has fatal flaw (R1 score = 1) | Critical | Even if other dimensions are excellent, lean toward Reject |\n| Major literature review omission (R2 score = 2) | Serious | Major Revision, require supplementation |\n| Cross-disciplinary perspective overlooked (R3 score = 2) | Moderate | Minor/Major, depends on other dimensions |\n| Poor writing quality (score = 2) | Minor | Does not affect academic decision, but require language revision |\n\n---\n\n## 4. Revision Round Policy\n\n### Standard Policy\n\n| Round | Expectation | Handling |\n|-------|-------------|---------|\n| R1 (First revision) | Respond to all reviewer comments | Send for re-review or EIC review |\n| R2 (Second revision) | Respond to residual issues | Usually EIC makes final decision |\n| R3 (Third revision) | Very rare, usually only handling formatting | EIC makes final decision |\n\n### Upgrade/Downgrade Rules\n\n- Minor Revision with incomplete revisions -> May escalate to Major Revision\n- Major Revision with excellent revisions -> May downgrade to Minor Revision or Accept\n- Major Revision with insufficient revisions -> May Reject (infinite revision cycles are not encouraged)\n- Beyond 2 rounds of Major Revision -> Strongly recommend Accept or Reject, no further extension\n\n---\n\n## 5. Professional Ethics of Editorial Review\n\n### Reviewer Ethics\n\n1. **Confidentiality**: The review process and paper content are confidential\n2. **Conflict of interest**: Recuse if there is a collaborative or competitive relationship with the author\n3. **Timeliness**: Complete the review within the committed timeframe\n4. **Constructiveness**: Even when recommending Reject, provide constructive feedback\n5. **Impartiality**: No bias based on author's gender, race, institution, or nationality\n6. **No plagiarism**: Do not use unpublished ideas seen during review\n7. **Appropriate language**: Avoid personal attacks, sarcasm, or demeaning language\n\n### Editor Ethics\n\n1. **Fair decision**: Based on academic quality, not influenced by external pressure\n2. **Transparent process**: Decision letter must clearly explain the rationale\n3. **Reasonable deadlines**: Give authors sufficient revision time\n4. **Appeal channel**: Authors have the right to respond to or challenge review comments\n5. **Consistent standards**: Papers of similar quality should receive similar decisions\n\n### Ethical Considerations for Special Situations\n\n| Situation | Ethical Handling |\n|-----------|-----------------|\n| Author is your student/colleague | Must recuse or disclose the relationship |\n| Paper's viewpoint is opposite to yours | Evaluate argument quality, not correctness of position |\n| Paper uses your theory but misunderstands it | May point it out but cannot require citation of your own work |\n| Suspected data fabrication | Report to EIC; journal initiates investigation procedure |\n| Paper is similar to your ongoing research | Disclose potential conflict of interest |\n\nFile v1.0.4:references/guided_mode_protocol.md\n\n# Guided Mode (Socratic Guided Review)\n\nThe design philosophy of Guided mode is to **help authors understand the paper's problems themselves**, rather than passively receiving revision instructions.\n\n### How It Works\n\n```\nPhase 0: Normal Field Analysis execution\nPhase 1: Normal execution of 5 reviews (but not all displayed immediately)\nPhase 2: Does not produce full Editorial Decision; enters dialogue mode instead\n```\n\n### Dialogue Flow\n\n1. **EIC opens**: First points out 1-2 core strengths of the paper (building confidence), then raises the most critical structural issue\n2. **Wait for author response**: Author thinks, responds, or asks questions\n3. **Progressive revelation**: Based on the author's level of understanding, gradually reveals deeper issues\n4. **Methodology focus**: When author is ready, introduce Reviewer 1's methodology perspective\n5. **Domain perspective**: Introduce Reviewer 2's domain expertise perspective\n6. **Cross-disciplinary challenge**: Introduce Reviewer 3's unique perspective\n7. **Devil's Advocate**: Finally introduce Devil's Advocate's core challenges and strongest counter-arguments\n8. **Wrap up**: When all key issues have been discussed, provide a structured Revision Roadmap\n\n### Dialogue Rules\n\n- Each response limited to 200-400 words (avoid information overload)\n- Use more questions, fewer commands (\"Do you think this sampling strategy can capture phenomenon X?\" rather than \"the sampling is flawed\")\n- When author's response shows understanding, affirm and move forward\n- When author's response veers off topic, gently guide back to the main point\n- Can ask the author to read a certain reference before continuing discussion\n\n### v3.6.2 sprint contract status\n\nv3.6.2 introduces sprint contracts for `reviewer_full` and `reviewer_methodology_focus` only. A template for this mode will follow in a subsequent patch release. Until then, this mode runs without contract enforcement and retains its pre-v3.6.2 behaviour.\n\nFile v1.0.4:references/integration_guide.md\n\n# Pipeline Usage Example\n\n```\nUser: I want to write a paper about AI in higher education quality assurance, from research to submission\n\nStep 1: deep-research -> Research report\nStep 2: academic-paper -> Paper first draft\nStep 3: integrity check -> 100% verification of references/data\nStep 4: academic-paper-reviewer (full) -> 5 review reports + Revision Roadmap\nStep 5: academic-paper (revision) -> Revised manuscript\nStep 6: academic-paper-reviewer (re-review) -> Verification review\nStep 7: (if needed) academic-paper (revision) -> Second revised manuscript\nStep 8: integrity check (final) -> Final 100% verification\nStep 9: academic-paper (format-convert) -> Final paper\n```\n\nFile v1.0.4:references/quality_rubrics.md\n\n# Quality Rubrics for Academic Paper Review\n\n## Purpose\n\nProvides calibrated scoring rubrics for the 7 review dimensions used by all reviewers (R1, R2, R3, DA). Ensures consistent, reproducible scoring across different papers and review sessions.\n\n## Known error profile (v3.2)\n\nThese rubrics define *what* to measure, not *how accurate* the measurement is. A single LLM reviewer's absolute rubric score has calibration error that depends on domain, paper type, and model version.\n\nFor users who want to know this reviewer's empirical FNR / FPR / balanced accuracy before relying on these rubric scores, run the opt-in **calibration mode** (see `calibration_mode_protocol.md`). Calibration mode compares this reviewer's decisions against a user-supplied gold set and produces a Calibration Report that attaches as a confidence disclosure to subsequent reviews in the same session.\n\nWithout calibration, treat rubric scores as *ordinally* meaningful (papers scored 85 are better than papers scored 65) but *not cardinally* interpretable (a 85 does not guarantee venue acceptance).\n\n## Scoring Scale\n\nAll dimensions scored 0-100. Final weighted score determines editorial decision.\n\n## Decision Mapping\n\n| Weighted Average | Decision |\n|-----------------|----------|\n| >= 80 | Accept |\n| 65-79 | Minor Revision |\n| 50-64 | Major Revision |\n| < 50 | Reject |\n\n---\n\n## Dimension 1: Originality (Weight: 20%)\n\n| Score Range | Descriptor | Behavioral Indicators |\n|------------|------------|----------------------|\n| 90-100 | Exceptional | Novel theoretical framework supported by empirical evidence; opens entirely new research direction; implications span 3+ fields; no prior work addresses this exact question |\n| 75-89 | Strong | Novel methodology OR novel application of existing theory to new context; clear contribution beyond incremental extension; implications for 2+ fields |\n| 60-74 | Adequate | Extends existing framework with new data, population, or context; contribution is clear but incremental; single-field implications |\n| 45-59 | Weak | Replicates existing study with minor variations; contribution is marginal; \"so what?\" question not convincingly answered |\n| < 45 | Insufficient | No discernible original contribution; duplicates existing work without justification; purely descriptive without analytical insight |\n\n## Dimension 2: Methodological Rigor (Weight: 25%)\n\n| Score Range | Descriptor | Behavioral Indicators |\n|------------|------------|----------------------|\n| 90-100 | Exceptional | Research design perfectly aligned with RQ; all validity threats addressed; appropriate statistical methods with power analysis; transparent reporting (all EQUATOR items); reproducible |\n| 75-89 | Strong | Sound design with minor gaps; most validity threats addressed; appropriate methods with minor reporting omissions; largely reproducible |\n| 60-74 | Adequate | Acceptable design but some validity concerns; methods appropriate but justification lacking; some reporting gaps (missing effect sizes, CIs) |\n| 45-59 | Weak | Design has significant flaws; method choice questionable; multiple reporting gaps; reproducibility doubtful |\n| < 45 | Insufficient | Fundamental design flaws that invalidate findings; inappropriate methods; results cannot be trusted |\n\n## Dimension 3: Evidence Sufficiency (Weight: 25%)\n\n| Score Range | Descriptor | Behavioral Indicators |\n|------------|------------|----------------------|\n| 90-100 | Exceptional | >40 sources, 80%+ peer-reviewed, multi-method triangulation, primary + secondary data, all claims well-supported, counter-evidence acknowledged |\n| 75-89 | Strong | 25-40 sources, 70%+ peer-reviewed, adequate evidence for main claims, some triangulation |\n| 60-74 | Adequate | 15-25 sources, 60%+ peer-reviewed, key claims supported but some gaps, limited triangulation |\n| 45-59 | Weak | <15 sources OR <50% peer-reviewed, several unsupported claims, no triangulation |\n| < 45 | Insufficient | Severely under-sourced, major claims unsupported, relies heavily on grey literature or anecdotal evidence |\n\n## Dimension 4: Argument Coherence (Weight: 15%)\n\n| Score Range | Descriptor | Behavioral Indicators |\n|------------|------------|----------------------|\n| 90-100 | Exceptional | Crystal-clear logical flow from problem -> gap -> RQ -> method -> findings -> implications; every section builds on previous; no logical jumps; counterarguments pre-empted |\n| 75-89 | Strong | Clear logical flow with minor gaps; most transitions well-handled; argument generally persuasive |\n| 60-74 | Adequate | Main argument visible but some sections feel disconnected; occasional logical jumps; conclusions mostly follow from evidence |\n| 45-59 | Weak | Argument structure unclear; significant logical gaps; conclusions overreach evidence; reader must infer connections |\n| < 45 | Insufficient | No coherent argument; sections appear unrelated; conclusions do not follow from evidence; circular reasoning |\n\n## Dimension 5: Writing Quality (Weight: 15%)\n\n| Score Range | Descriptor | Behavioral Indicators |\n|------------|------------|----------------------|\n| 90-100 | Exceptional | Professional academic prose; precise terminology; excellent paragraph structure; zero grammatical errors; appropriate register throughout |\n| 75-89 | Strong | Good academic writing; minor stylistic inconsistencies; few grammatical issues; terminology mostly precise |\n| 60-74 | Adequate | Acceptable writing but room for improvement; some verbose passages; occasional imprecise terminology; some grammar issues |\n| 45-59 | Weak | Below journal standards; frequent verbose/unclear passages; terminology inconsistent; multiple grammar issues |\n| < 45 | Insufficient | Unacceptable writing quality; incomprehensible passages; severe grammar problems; not suitable for peer review |\n\n## Optional Dimensions (reviewer-specific)\n\n### Literature Integration (R2 Domain Expert focus)\n\n| Score Range | Descriptor |\n|------------|------------|\n| 90-100 | Comprehensive coverage of seminal + recent works; identifies theoretical lineage; positions paper precisely in scholarly conversation |\n| 75-89 | Good coverage; most key works cited; reasonable positioning in literature |\n| 60-74 | Adequate but gaps in coverage; some important works missing; positioning somewhat vague |\n| < 60 | Significant literature gaps; key works missing; poor positioning |\n\n### Significance & Impact (R3 Perspective Reviewer focus)\n\n| Score Range | Descriptor |\n|------------|------------|\n| 90-100 | Clear practical implications for policy/practice AND theory; addresses urgent real-world problem; likely to influence field direction |\n| 75-89 | Good practical OR theoretical implications; addresses relevant problem; moderate influence potential |\n| 60-74 | Some implications but narrowly scoped; relevance clear but impact limited |\n| < 60 | Minimal practical or theoretical significance; unclear why this matters |\n\n---\n\n## Aggregation Formula\n\n```\nFinal Score = (Originality x 0.20) + (Methodology x 0.25) + (Evidence x 0.25) + (Coherence x 0.15) + (Writing x 0.15)\n```\n\nOptional dimensions are reported separately and factored into the editorial synthesis narrative but do not change the numerical score.\n\n---\n\n## Calibration Notes\n\n- Scores should reflect the paper's quality relative to the target journal's standards\n- A \"75\" for Nature is not equivalent to \"75\" for a regional journal\n- When in doubt, err toward the middle of a range\n- Reviewers should explicitly state which range descriptor best matches, then fine-tune within that range\n- If two dimensions are at odds (e.g., excellent methodology but weak writing), do NOT average down — report both scores honestly\n\nFile v1.0.4:references/re_review_mode_protocol.md\n\n# Re-Review Mode (Verification Review)\n\nRe-review mode is the dedicated mode for Pipeline Stage 3', designed to **verify whether revisions address the first-round review comments**.\n\n### How It Works\n\n```\nInput:\n1. Original Revision Roadmap (Stage 3 output)\n2. Revised manuscript\n3. Response to Reviewers (optional)\n\nPhase 0: Reads the Revision Roadmap, builds a checklist\nPhase 1: EIC checks each item (other reviewers not activated)\nPhase 2: Editorial Synthesis -> New Decision\n```\n\n### Verification Logic\n\n```\nFor each item in the Revision Roadmap:\n\nPriority 1 (Required):\n  -> Check each item for corresponding changes in the revised manuscript\n  -> Assess revision quality (FULLY_ADDRESSED / PARTIALLY_ADDRESSED / NOT_ADDRESSED / MADE_WORSE)\n  -> All Priority 1 items must be FULLY_ADDRESSED for Accept\n\n**Traceability Rule**: For each Priority 1 item, the reviewer MUST:\n1. Read the author's claim from the Response to Reviewers\n2. Navigate to the stated revision location in the manuscript\n3. Independently verify the claim matches the actual change\n4. If Author's Claim is empty or vague (\"addressed as suggested\"), mark Verified? as `🔍 Cannot verify` and flag in Quality Assessment\n\nPriority 2 (Suggested):\n  -> Check each item\n  -> At least 80% should have a response\n  -> NOT_ADDRESSED items require author explanation\n\nPriority 3 (Nice to Fix):\n  -> Check but does not affect Decision\n```\n\n### New Issue Detection\n\n```\nIn addition to checking old items, EIC also scans for:\n- Whether content added during revision introduces new problems\n- Whether newly added references are correct (but deep verification is left to Stage 4.5 integrity check)\n- Whether revisions cause inconsistencies\n```\n\n### Socratic Guidance After Re-Review\n\n```\nIf Re-Review Decision = Major Revision:\n  -> Activate Residual Coaching (residual issue guidance)\n  -> EIC guides user through Socratic dialogue:\n    1. Gap analysis — \"How many issues did the first round of revisions resolve? Why are the remaining ones hard to address?\"\n    2. Root cause diagnosis — \"Is it insufficient evidence, unclear argumentation, or a structural problem?\"\n    3. Trade-off decisions — \"Which ones can be marked as research limitations?\"\n    4. Action plan — Plan revision approach for each residual issue\n  -> Maximum 5 rounds of dialogue\n  -> User can say \"just fix it\" to skip guidance\n```\n\n### Re-Review Output Format\n\n```markdown\n# Verification Review Report\n\n## Decision\n[Accept / Minor Revision / Major Revision]\n\n## Revision Response Checklist\n\n### Priority 1 — Required Revisions\n\n| # | Original Review Comment | Author's Claim | Response Status | Revision Location | Verified? | Quality Assessment |\n|---|------------------------|---------------|-----------------|-------------------|-----------|-------------------|\n| R1 | [Original text] | [What the author claims to have done in Response to Reviewers] | FULLY_ADDRESSED | Section X.X | ✅ Yes | Adequately addressed; newly added content effectively resolves the issue |\n| R2 | [Original text] | [Author's stated change] | PARTIALLY_ADDRESSED | Section Y.Y | ⚠️ Partial | Partially addressed, but still missing [specific gap] |\n\n### Priority 2 — Suggested Revisions\n\n| # | Original Review Comment | Response Status | Notes |\n|---|------------------------|-----------------|-------|\n| S1 | [Original text] | FULLY_ADDRESSED | -- |\n| S2 | [Original text] | NOT_ADDRESSED | Author explanation: [reason] |\n\n### Priority 3 — Nice to Fix\n\n| # | Original Review Comment | Response Status |\n|---|------------------------|-----------------|\n| N1 | [Original text] | FULLY_ADDRESSED |\n\n## New Issues (Discovered During Revision)\n\n| # | Type | Location | Description |\n|---|------|----------|-------------|\n| NEW-1 | [Type] | Section X.X | [Description] |\n\n## Decision Rationale\n[Rationale based on the checklist]\n\n## Residual Issues (If Any)\n[List unresolved items, suggest marking as Acknowledged Limitations]\n```\n\n## v3.6.2 sprint contract status\n\nv3.6.2 introduces sprint contracts for `reviewer_full` and `reviewer_methodology_focus` only. A template for this mode will follow in a subsequent patch release. Until then, this mode runs without contract enforcement and retains its pre-v3.6.2 behaviour.\n\nFile v1.0.4:references/review_criteria_framework.md\n\n# Review Criteria Framework — Structured Review Criteria Framework\n\nThis document defines universal criteria for academic paper review and type-specific criteria differentiated by paper type. All reviewer agents share this framework.\n\n---\n\n## 1. Universal Review Dimensions\n\nSeven core dimensions applicable to all paper types:\n\n### Dimension 1: Originality — Weight 15%\n\n| Level | Score | Description |\n|-------|-------|-------------|\n| Outstanding | 5 | Proposes entirely new theory/method/evidence that could change the field's direction |\n| Strong | 4 | Has clear new insights or novel combinations, fills a specific research gap |\n| Adequate | 3 | Incremental contribution, reasonable extension of existing knowledge |\n| Weak | 2 | Highly overlapping with existing literature, new contribution unclear |\n| None | 1 | Essentially repeats what is already known |\n\n### Dimension 2: Methodological Rigor — Weight 25%\n\n| Level | Score | Description |\n|-------|-------|-------------|\n| Outstanding | 5 | Impeccable research design, innovative methods executed flawlessly |\n| Strong | 4 | Sound design, appropriate methods, minor room for improvement in execution |\n| Adequate | 3 | Methods basically acceptable, but with some design or execution limitations |\n| Weak | 2 | Methods have significant flaws affecting the credibility of conclusions |\n| Unacceptable | 1 | Methods fundamentally unsuitable for answering the research question, or contain serious errors |\n\n### Dimension 3: Evidence Sufficiency — Weight 20%\n\n| Level | Score | Description |\n|-------|-------|-------------|\n| Outstanding | 5 | Rich, diverse, and persuasive evidence that exceeds expectations |\n| Strong | 4 | Evidence sufficiently supports all major arguments |\n| Adequate | 3 | Most arguments supported by evidence, a few need supplementation |\n| Weak | 2 | Key arguments lack sufficient evidence |\n| Unacceptable | 1 | Serious disconnect between arguments and evidence |\n\n### Dimension 4: Argument Coherence — Weight 15%\n\n| Level | Score | Description |\n|-------|-------|-------------|\n| Outstanding | 5 | Clear arguments, rigorous logic, elegant structure |\n| Strong | 4 | Smooth argumentation, occasional minor logical leaps |\n| Adequate | 3 | Basically coherent, but some inter-paragraph connections are unclear |\n| Weak | 2 | Multiple logical breaks, readers have difficulty following the argument |\n| Unacceptable | 1 | Confused argumentation, core claims cannot be identified |\n\n### Dimension 5: Writing Quality — Weight 10%\n\n| Level | Score | Description |\n|-------|-------|-------------|\n| Outstanding | 5 | Precise and fluent academic English/Chinese, a model of scholarly writing |\n| Strong | 4 | Clear language, occasional minor imperfections that don't affect understanding |\n| Adequate | 3 | Generally readable, with some grammar or word choice issues |\n| Weak | 2 | Frequent language issues that affect understanding |\n| Unacceptable | 1 | Language quality does not meet reviewable standards |\n\n### Dimension 6: Literature Integration — Weight 10%\n\n| Level | Score | Description |\n|-------|-------|-------------|\n| Outstanding | 5 | Comprehensive, contemporary, critically integrated literature with a compelling research gap argument |\n| Strong | 4 | Covers major literature, with good integration and positioning |\n| Adequate | 3 | Basic coverage, but with omissions or insufficient integration |\n| Weak | 2 | Literature is outdated, incomplete, or merely enumerated |\n| Unacceptable | 1 | Seriously insufficient literature review or irrelevant to the topic |\n\n### Dimension 7: Significance & Impact — Weight 5%\n\n| Level | Score | Description |\n|-------|-------|-------------|\n| Outstanding | 5 | Could change policy, practice, or theoretical direction |\n| Strong | 4 | Clear impact on a specific field or practice |\n| Adequate | 3 | Has some academic or practical value |\n| Weak | 2 | Limited scope of impact, mainly academic interest |\n| Marginal | 1 | Difficult to see the significance of the research |\n\n---\n\n## 2. Paper Type-Specific Criteria\n\n### 2.1 Empirical Research\n\nBeyond universal dimensions, specifically focus on:\n\n| Additional Dimension | Review Focus |\n|---------------------|-------------|\n| Research hypothesis clarity | Are hypotheses testable and consistent with theory |\n| Variable operational definitions | Are independent/dependent/control variable definitions precise |\n| Internal validity | Are confounding variables controlled |\n| External validity | Generalizability of results |\n| Statistical reporting completeness | Effect sizes, confidence intervals, assumption testing |\n| Conclusion conservatism | Do conclusions exceed what the data supports |\n\n### 2.2 Theoretical/Conceptual Paper\n\n| Additional Dimension | Review Focus |\n|---------------------|-------------|\n| Conceptual definition precision | Are core concepts clearly delineated |\n| Argument logic structure | Is the premise -> inference -> conclusion logic chain complete |\n| Counterargument handling | Are possible opposing viewpoints considered and addressed |\n| Theoretical novelty | Does it truly advance theoretical development |\n| Testability | Can the theory generate testable propositions |\n\n### 2.3 Literature Review / Meta-analysis\n\n| Additional Dimension | Review Focus |\n|---------------------|-------------|\n| Search strategy | Is it comprehensive and reproducible (PRISMA compliance) |\n| Inclusion/exclusion criteria | Are criteria clear, reasonable, and consistently applied |\n| Bias risk assessment | Is bias risk of included studies assessed |\n| Heterogeneity handling | Is statistical and conceptual heterogeneity appropriately handled |\n| Synthesis method | Goes beyond simple vote counting to achieve critical synthesis |\n| Publication Bias | Is publication bias assessed and discussed |\n\n### 2.4 Case Study\n\n| Additional Dimension | Review Focus |\n|---------------------|-------------|\n| Case selection justification | Why was this case chosen? What does it represent? |\n| Theoretical vs convenience sampling | Is case selection theoretically grounded |\n| Triangulation | Are multiple data sources used |\n| Context description thickness | Is thick description sufficient |\n| Analysis transferability | Are analysis results transferable to other contexts |\n| Researcher reflexivity | Is the researcher's relationship with the case reflected upon |\n\n### 2.5 Policy Analysis / Policy Brief\n\n| Additional Dimension | Review Focus |\n|---------------------|-------------|\n| Policy problem definition | Is the problem clearly defined and evidence-supported |\n| Stakeholder analysis | Are key stakeholders identified |\n| Policy option analysis | Are multiple options proposed and compared |\n| Feasibility assessment | Are policy recommendations practically feasible |\n| Evidence quality | Are policy recommendations based on reliable evidence |\n| Unintended consequences | Are unintended policy impacts considered |\n\n---\n\n## 3. Common Review Pitfalls\n\n### Biases Reviewers Should Avoid\n\n| Pitfall | Description | How to Avoid |\n|---------|-------------|--------------|\n| **Hypercriticism** | Overblowing minor issues, ignoring the paper's overall contribution | Affirm strengths first, then point out issues; distinguish major from minor |\n| **Confirmation Bias** | Only finding evidence supporting pre-existing views | Deliberately seek the paper's merits and counterexamples to your own views |\n| **Preference Projection** | Requiring authors to use \"my method\" rather than evaluating \"the author's method\" | Ask \"can this method answer the question\" rather than \"what would I do\" |\n| **Paradigm Bias** | Using quantitative standards to judge qualitative research (or vice versa) | Use evaluation criteria matching the paper's research paradigm |\n| **Prestige Bias** | Relaxing standards because of the author's institution or past achievements | Focus on the quality of the paper itself |\n| **Novelty Bias** | Only valuing novel research, undervaluing replication studies | Acknowledge the important role of replication in science |\n| **Length Bias** | Long paper = good paper, short paper = sloppy | Evaluate content density, not page count |\n| **Language Discrimination** | Undervaluing research quality due to non-native language imperfections | Distinguish \"language needs polishing\" from \"research quality is poor\" |\n\n### Principles of Constructive Feedback\n\n1. **Specific, not vague**: \"The causal inference in Section 3, paragraph 2 lacks control variables\" is better than \"methodology has problems\"\n2. **Problem + reason + suggestion**: Every criticism should include \"what,\" \"why,\" and \"how to fix\"\n3. **Distinguish required from suggested**: Which changes are mandatory, which are \"nice to have\"\n4. **Acknowledge uncertainty**: \"I'm not sure whether this analysis accounts for X\" is more accurate than \"the author ignored X\"\n5. **Respect the author**: Even if paper quality is poor, the author still invested time and effort\n\n---\n\n## 4. Scoring Aggregation\n\n### Weighted Total Score Calculation\n\n```\nTotal Score =\n  Originality (15%) +\n  Methodological Rigor (25%) +\n  Evidence Sufficiency (20%) +\n  Argument Coherence (15%) +\n  Writing Quality (10%) +\n  Literature Integration (10%) +\n  Significance (5%)\n```\n\n### Score-to-Decision Mapping\n\n| Weighted Total | Recommended Decision | Note |\n|---------------|---------------------|------|\n| 4.5-5.0 | Accept | Very few papers reach this level |\n| 3.5-4.4 | Minor Revision | Overall quality is good, minor revisions needed |\n| 2.5-3.4 | Major Revision | Has potential but needs substantial revision |\n| 1.5-2.4 | Reject (Resubmit) | Fundamental issues need rework, but topic has value |\n| 1.0-1.4 | Reject | Not suitable for this journal or quality below standard |\n\n**Important reminder**: Scores are only reference. The final decision also needs to consider:\n- Whether any single dimension is particularly low (e.g., methodology score of 1), which may lead to Reject even if the overall score is passable\n- Specific content of reviewer comments is more important than numbers\n- Special considerations of the journal (special issue, field development needs, etc.)\n\nFile v1.0.4:references/review_quality_thinking.md\n\n# Review Quality Thinking Framework\n\nA cognitive framework for producing high-quality reviews. Teaches **how to think** about paper quality, not just what to score.\n\n## The Three Lenses\n\nEvery paper should be evaluated through three lenses simultaneously:\n\n### Lens 1: Internal Validity — \"Does the evidence support the claims?\"\n\nAsk in order:\n1. What is the central claim?\n2. What evidence is presented?\n3. Is there a logical chain from evidence to claim? (Warrant)\n4. Are there alternative explanations the authors didn't consider?\n5. Would removing any single piece of evidence collapse the argument?\n\n**If #5 is yes**: The argument depends on a single linchpin. Flag it — the paper's contribution is only as strong as that one piece of evidence.\n\n### Lens 2: External Validity — \"Does this matter beyond this study?\"\n\nAsk in order:\n1. Who is the population of interest?\n2. Does the sample represent that population?\n3. Are the conditions replicable?\n4. Would the findings hold in a different context/culture/time?\n5. What are the boundary conditions the authors don't mention?\n\n**Judgment heuristic**: Most authors overstate generalizability. If the sample is from one university in one country, findings cannot claim to apply \"generally\" without qualification.\n\n### Lens 3: Contribution — \"So what?\"\n\nAsk in order:\n1. What did we know before this paper?\n2. What do we know after?\n3. Is the delta meaningful (not just statistically significant)?\n4. Who benefits from knowing this?\n5. What new questions does this open?\n\n**Judgment heuristic**: If you can't articulate the delta in one sentence, either the contribution is weak or the paper hasn't communicated it clearly. Both are review-worthy observations.\n\n## Common Reviewer Traps\n\n| Trap | Description | How to Avoid |\n|------|-------------|-------------|\n| **Methodological tunnel vision** | Only critiquing methods, ignoring whether the question matters | Start with Lens 3 (contribution) before Lens 1 |\n| **Novelty bias** | Penalizing replication or incremental work | Replication IS valuable; evaluate on execution quality |\n| **Expertise projection** | Expecting the paper to use your preferred method | Evaluate the chosen method on its own terms |\n| **Positivity-severity oscillation** | Being too nice in comments, too harsh in scores | Write the score first, then justify with comments |\n| **Missing forest for trees** | Listing 20 minor issues, missing the one fatal flaw | Always state the single most important issue first |\n\n## Calibration Questions (ask after drafting your review)\n\n1. If this paper were published as-is, would it mislead readers? (If yes → Major Revision or Reject)\n2. Could the authors reasonably address my concerns in one revision cycle? (If no → Reject)\n3. Am I being harder on this paper than I would be on my own work? (Calibration check)\n4. Did I identify at least one genuine strength? (Balance check)\n5. Would my review help the authors improve, even if the paper is rejected? (Constructiveness check)\n\nFile v1.0.4:references/sprint_contract_protocol.md\n\n# Sprint Contract Protocol (v3.6.2)\n\n> Authoritative orchestration reference for the ARS v3.6.2 sprint-contract hard gate.\n> Schema: `shared/sprint_contract.schema.json` (Schema 13.1 since v3.6.6).\n> Templates: `shared/contracts/reviewer/*.json`.\n> Design spec: `docs/design/2026-04-23-ars-v3.6.2-sprint-contract-design.md`.\n>\n> **v3.6.6 cross-reference**: this reviewer protocol is byte-equivalent across v3.6.2 → v3.6.6 (zero-touch promise per §3.6 of `docs/design/2026-04-27-ars-v3.6.6-generator-evaluator-contract-design.md`). The v3.6.6 release adds a parallel generator-evaluator protocol inside `academic-paper` for the in-pair writer / evaluator pair (see `academic-paper/SKILL.md` § \"v3.6.6 Generator-Evaluator Contract Protocol\" and design doc §5).\n\n## 1. Overview\n\nA reviewer sprint contract is a machine-checkable pre-registered acceptance criterion. The orchestrator loads a frozen template, inlines runtime fields (`generated_at`, optional `agent_amendments`), and drives each reviewer through a paper-content-blind Phase 1 followed by a paper-visible Phase 2. The synthesizer then runs a three-step mechanical protocol over the `panel_size` reviewer outputs to emit an editorial decision.\n\nThis protocol exists to destroy the \"read the paper, then rationalise the scoring standard\" drift path. The load-bearing mechanism is the **physical separation of calls**: Phase 1 never sees paper content.\n\n## 2. Two-phase reviewer call\n\nFor each reviewer in `range(panel_size)`:\n\n1. **Prepare contract.** Load template from `shared/contracts/<domain>/<mode>.json`. Populate `generated_at` (ISO-8601 UTC). Optionally populate `agent_amendments` (field-specific notes from `field_analyst_agent`). Run `check_sprint_contract.py` on the in-memory object; abort on error.\n2. **Phase 1 call (paper-content-blind).**\n   - System prompt: the `### Phase 1 — Paper-content-blind pre-commitment` sub-section of the reviewer agent's `## v3.6.2 Sprint Contract Protocol` block.\n   - User content: contract JSON + paper metadata ONLY (`title`, `field`, `word_count`).\n   - Expected output: `## Contract Paraphrase`, `## Scoring Plan`, terminal `[CONTRACT-ACKNOWLEDGED]` tag.\n3. **Phase 1 output lint.** See §4 below.\n4. **Phase 2 call (paper-visible).**\n   - System prompt: the `### Phase 2 — Paper-visible review` sub-section of the same `## v3.6.2 Sprint Contract Protocol` block.\n   - User content: contract JSON (re-injected) + Phase 1 output wrapped in `<phase1_output>...</phase1_output>` data delimiter + full paper.\n   - Expected output: optional `## Scoring Plan Dissent`, `## Dimension Scores`, `## Failure Condition Checks`, `## Review Body`, `## Editorial Decision`.\n5. **Phase 2 output lint.** See §5 below.\n6. **Panel cardinality invariant.** After all reviewers complete, verify `len(usable_phase2_outputs) == panel_size`. If any reviewer was dropped, emit `[PANEL-SHRUNK]` and abort the round (see §6).\n7. Feed usable Phase 2 outputs into synthesizer (see §7).\n\n## 3. Contract injection\n\n- **Template on disk is frozen.** Do not mutate. Deep-copy into an in-memory dict.\n- **Runtime-only fields:** `generated_at`, `agent_amendments.stage_specific_notes`, `agent_amendments.additional_measurement_hints`.\n- **Baseline fields are orchestrator-immutable.** Schema cannot enforce this; the orchestrator must not rewrite `acceptance_dimensions` / `failure_conditions` / `measurement_procedure` / `override_ladder` / `mode` / `stage` / `contract_id` / `baseline_version` / `panel_size` between template load and injection. Optional: emit sha256 of baseline-field subset to audit log for drift detection.\n\n## 4. Phase 1 output lint\n\nStructural checks (orchestrator, not validator). On failure retry Phase 1 once with the specific lint gap hinted in the system prompt; second failure aborts that reviewer.\n\n- Required sections in order: `## Contract Paraphrase`, `## Scoring Plan`, terminal `[CONTRACT-ACKNOWLEDGED]`.\n- Paraphrase paragraph count ≥ `measurement_procedure.paraphrase_minimum_dimensions` (for `\"all\"`, one paragraph per dimension; for integer `k`, at least `k` paragraphs each matching a distinct dimension).\n- `## Scoring Plan` has one `### <Dn>: <name>` subsection per acceptance dimension (always full coverage, regardless of `paraphrase_minimum_dimensions`).\n- Each `scoring_plan` subsection contains lines matching `measurement_procedure.scoring_plan_schema.required`.\n- Phase 1 content refers to `<title>`, `<field>`, `<word_count>` only; no specific paper content. Not schema-enforced; behavioural rule in reviewer prompt.\n\n**Lint is structural, not semantic.** A reviewer can in principle pass this lint by emitting generic boilerplate triggers — semantic judgement (whether triggers are concrete and discriminating) is deferred to a post-v3.6.2 judge-agent layer.\n\nOn second Phase 1 failure: emit `[PROTOCOL-VIOLATION: reviewer=<role>, contract=<id>, phase1_lint_failed=true]` and mark this reviewer unusable.\n\n## 5. Phase 2 output lint\n\nStructural checks run before handoff to synthesizer. **No Phase 2 retry** (reviewer has seen the paper; a second call is tainted) EXCEPT the multi-dissent case below.\n\n- Required sections: `## Dimension Scores`, `## Failure Condition Checks`, `## Review Body`, `## Editorial Decision`.\n- `## Dimension Scores` has one `### <Dn>: <name>` subsection per contract dimension; each carries a value in `$defs.score` (`block | warn | pass`).\n- `## Failure Condition Checks` has one subsection per `failure_conditions[]` entry with `fired: true | false`.\n- **Multi-dissent rule:** If `## Scoring Plan Dissent` names two or more `dimension_id` entries, orchestrator aborts this reviewer and retries from **Phase 1** once. If the retried Phase 1/2 also multi-dissents, mark the reviewer unusable (`[PROTOCOL-VIOLATION]`). One-dimension-per-reviewer-per-Phase-2-call is the cap.\n- **Consistency check (structural):** For every dimension not under dissent, the Phase 2 score must substring-match the reviewer's Phase 1 `scoring_plan` trigger tokens. Vacuous triggers bypass this check — documented limitation.\n- `## Editorial Decision` is one of the `action` values derivable from `## Failure Condition Checks` via the synthesizer precedence rule (§8 step 3). Inconsistency marks the reviewer unusable.\n\nOn any Phase 2 lint failure other than multi-dissent: emit `[PROTOCOL-VIOLATION]` and mark reviewer unusable. Do not synthesise a substitute score for the synthesizer.\n\n## 6. Multi-reviewer orchestration\n\n- **Independent cycles.** Each of the `panel_size` reviewers runs its own Phase 1 + Phase 2. Failures in one do not pause the others.\n- **Panel cardinality invariant (§2 step 6).** After all reviewers complete, if `len(usable_phase2_outputs) < panel_size`, abort the editorial round with `[PANEL-SHRUNK]`. Do not silently recompute `cross_reviewer_quantifier` thresholds against a smaller panel — the contract's published aggregation semantics bind on a specific `panel_size`.\n- **Operational monitor.** Track `[PANEL-SHRUNK]` rate in real SR runs. If > 5% of rounds abort in first 3 months, v3.6.3 introduces graceful-degradation fallback.\n\n## 7. Reviewer panel mapping\n\n| mode                          | panel_size | invoked reviewers |\n|-------------------------------|------------|-------------------|\n| `reviewer_full`               | 5          | EIC + methodology + domain + perspective + DA |\n| `reviewer_methodology_focus`  | 2          | EIC + methodology (only) |\n| `reviewer_re_review`          | —          | not shipped in v3.6.2; continues pre-v3.6.2 behaviour |\n| `reviewer_calibration`        | —          | not shipped in v3.6.2 |\n| `reviewer_guided`             | —          | not shipped in v3.6.2 |\n\nThe orchestrator uses `mode` to determine the panel and the contract's `panel_size` as the invariant target. SC-11 validator check ensures mode and `panel_size` are consistent.\n\n## 8. Synthesizer three-step protocol\n\nLet `N = contract.panel_size`.\n\n**Step 1 — Build scoring matrix.** For each `acceptance_dimensions[i]`, gather N reviewers' `## Dimension Scores` for that dimension into a length-N array of `$defs.score` values. Dimensions resolved by `id`.\n\n**Step 2 — Evaluate each `failure_conditions[]`.** For each condition:\n\n1. Parse `expression` against the recognised patterns (see §9 vocabulary). Unrecognised → emit `[EXPRESSION-UNRECOGNISED]`, abort synthesizer.\n2. Apply `cross_reviewer_quantifier` with panel-relative thresholds:\n   - `any`: fires if predicate holds for ≥ 1 of N reviewers.\n   - `majority`: for N ≥ 3, fires if ≥ `⌈N/2⌉ + 1`; for N == 2, fires if all 2; for N == 1, vacuous (SC-11 warns).\n   - `all`: fires if predicate holds for all N reviewers.\n3. Record `{condition_id, fired}`.\n\n**Step 3 — Precedence and decision.** Among fired conditions, pick the one with highest `severity`. Ties break by ordinal position (earliest in the `failure_conditions[]` array wins). Emit its `action` as `editorial_decision`.\n\n**Forbidden operations (synthesizer prompt hard constraint):**\n- Introduce aggregation rules not derivable from `cross_reviewer_quantifier` + `severity`.\n- Average or vote-aggregate scores within a single dimension unless `cross_reviewer_quantifier: majority` explicitly requests it.\n- Soften a fired condition's `action` on post-hoc grounds.\n- Synthesise substitute scores for reviewers marked unusable — the round is either complete with `panel_size` usable outputs or `[PANEL-SHRUNK]` aborted.\n\n## 9. Recognised expression vocabulary\n\nSynthesizer recognises the following patterns (with accepted natural-English variants):\n\n1. **Priority-scoped single-match:** `any <priority> dimension scores '<score>'` | `any dimension with priority=<priority> scores '<score>'` | `any <priority>-priority dimension scores '<score>'`\n2. **Priority-scoped count-based:** `two or more <priority> dimensions score '<score>' or worse` | `two or more dimensions with priority=<priority> score '<score>' or worse` (ordering `pass` < `warn` < `block`)\n3. **Universal over priority:** `every <priority> dimension scores '<score>'`\n4. **Single-dimension literal:** `<Dn> scores '<score>'`\n5. **Conjunction:** any of the above joined by `AND`\n\nShipped template coverage:\n- `reviewer/full.json`: F1 pattern 1 (bare mandatory), F2 pattern 2, F3 pattern 1 (`high-priority` variant), F0 pattern 3.\n- `reviewer/methodology_focus.json`: F1 / F2 / F0 pattern 4 (literal D1).\n\nNew expression forms require a PR updating both this §9 and the synthesizer prompt's recognised-pattern list.\n\n## 10. Token cost expectations\n\nReviewer total calls = `2 × panel_size`. For `reviewer_full` that is 5 → 10 calls; for `reviewer_methodology_focus` 2 → 4. Phase 1 input is small (contract + metadata only); Phase 1 output is short (paraphrase + scoring_plan). Real token bound is well below 2x raw increase.\n\n## 11. Failure modes and diagnostics\n\nAudit-log tags the orchestrator may emit:\n\n| Tag | When | Action |\n|-----|------|--------|\n| `[CONTRACT-ACKNOWLEDGED]` | normal Phase 1 completion | none (expected) |\n| `[PROTOCOL-VIOLATION: phase1_lint_failed=true]` | Phase 1 lint fails twice for a reviewer | mark reviewer unusable |\n| `[PROTOCOL-VIOLATION: phase2_lint_failed=<check>]` | Phase 2 lint fails (non multi-dissent) | mark reviewer unusable |\n| `[PROTOCOL-VIOLATION: multi_dissent=true]` | Phase 2 has ≥ 2 dissent entries, retry exhausted | mark reviewer unusable |\n| `[PANEL-SHRUNK: usable=<k>, panel_size=<N>]` | §6 invariant failed | abort editorial round |\n| `[EXPRESSION-UNRECOGNISED: condition_id=<F>, expression=<...>]` | synthesizer step 2.1 | abort synthesizer |\n\nArchive v1.0.3: 39 files, 156982 bytes\n\nFiles: _meta.json (142b), agents/devils_advocate_reviewer_agent.md (15011b), agents/domain_reviewer_agent.md (11729b), agents/editorial_synthesizer_agent.md (13139b), agents/eic_agent.md (8438b), agents/field_analyst_agent.md (9429b), agents/methodology_reviewer_agent.md (13196b), agents/perspective_reviewer_agent.md (14358b), LICENSE (19584b), references/calibration_mode_protocol.md (10542b), references/changelog.md (976b), references/editorial_decision_standards.md (9230b), references/guided_mode_protocol.md (1971b), references/integration_guide.md (680b), references/quality_rubrics.md (7679b), references/re_review_mode_protocol.md (4250b), references/review_criteria_framework.md (10139b), references/review_quality_thinking.md (3015b), references/sprint_contract_protocol.md (11621b), references/statistical_reporting_standards.md (22703b), references/top_journals_by_field.md (12874b), shared/artifact_reproducibility_pattern.md (8921b), shared/benchmark_report_pattern.md (8988b), shared/benchmark_report.schema.json (2789b), shared/collaboration_depth_rubric.md (10810b), shared/compliance_checkpoint_protocol.md (7175b), shared/compliance_report.schema.json (8314b), shared/cross_model_verification.md (11173b), shared/ground_truth_isolation_pattern.md (12649b), shared/handoff_schemas.md (46013b), shared/mode_spectrum.md (4047b), shared/prisma_trAIce_protocol.md (10415b), shared/raise_framework.md (7365b), shared/sprint_contract.schema.json (18954b), shared/style_calibration_protocol.md (7085b), SKILL.md (4977b), templates/editorial_decision_template.md (6446b), templates/peer_review_report_template.md (7479b), templates/revision_response_template.md (7020b)\n\nFile v1.0.3:SKILL.md\n\n---\nname: academic-paper-reviewer\ndescription: \"7-agent paper review system on Hermes Agent. 6 modes (full/re-review/quick/methodology-focus/guided/calibration). 5-panel review with editorial decision, revision roadmap, and calibration metrics. Uses delegate_task for each reviewer. Triggers: review paper, peer review, manuscript review, check revisions, calibrate reviewer, 審稿, 同儕審查, 論文審查.\"\nmetadata:\n  version: \"1.0-hermes-1.0\"\n  last_updated: \"2026-05-16\"\n  status: active\n  adapted_from: \"imbad0202/academic-research-skills\"\n  adapted_for: \"Hermes Agent (deepseek-v4-pro)\"\n  task_type: open-ended\n  license: \"CC BY-NC 4.0\"\n  original_author: \"Cheng-I Wu\"\n  original_license: \"CC BY-NC 4.0\"\n  original_repo: \"https://github.com/Imbad0202/academic-research-skills\"\n  copyright: \"Copyright (c) 2026 Cheng-I Wu\"\n---\n# Academic Paper Reviewer — 7-Agent Review System (Hermes Edition)\n\n📄 **License:** [CC BY-NC 4.0](https://creativecommons.org/licenses/by-nc/4.0/) · Copyright (c) 2026 Cheng-I Wu  \n🔗 **Original:** [Imbad0202/academic-research-skills](https://github.com/Imbad0202/academic-research-skills)  \n🔄 **Adaptation:** Multi-agent review system implemented via `delegate_task` instead of Claude Code's internal agent system. All agent definitions, references, and quality standards preserved unchanged from original. **This adaptation is distributed under the same CC BY-NC 4.0 license.**\n\n## Quick Start\n\n```\nReview this paper for journal submission\n```\n\n## Agent Team\n\n| # | Agent | Role |\n|---|-------|------|\n| 1 | intake_agent | Receive paper, determine review type |\n| 2 | methodology_reviewer | Method rigor assessment |\n| 3 | evidence_reviewer | Evidence sufficiency & citation quality |\n| 4 | argument_reviewer | Logical coherence & argument structure |\n| 5 | domain_reviewer | Domain expertise & literature positioning |\n| 6 | editor_in_chief | Aggregate reviews → editorial decision |\n| 7 | revision_coach | Convert reviews → actionable roadmap |\n\n## Hermes Execution\n\n### Full Mode: 5-Panel Parallel Review\n```\ndelegate_task(tasks=[\n    {\"goal\": \"Review manuscript methodology: design appropriateness, validity threats, replicability. Score 1-5.\", \"context\": \"Use agents/methodology_reviewer.md\", \"toolsets\": [\"file\"]},\n    {\"goal\": \"Review evidence: citation quality, source credibility, evidence hierarchy alignment. Score 1-5.\", \"context\": \"Use agents/evidence_reviewer.md\", \"toolsets\": [\"file\"]},\n    {\"goal\": \"Review argument: logical flow, claim-evidence alignment, counter-argument handling. Score 1-5.\", \"context\": \"Use agents/argument_reviewer.md\", \"toolsets\": [\"file\"]},\n    {\"goal\": \"Review domain positioning: literature coverage, theoretical grounding, contribution significance. Score 1-5.\", \"context\": \"Use agents/domain_reviewer.md\", \"toolsets\": [\"file\"]}\n])\n```\n\n### Editorial Decision\n```\ndelegate_task(goal=\"Aggregate all 4 reviewer reports. Apply weighted scoring (Method 30%, Evidence 25%, Argument 25%, Domain 20%). Issue editorial decision: Accept/Minor Revision/Major Revision/Reject with justification.\", context=\"Use agents/editor_in_chief.md\", toolsets=[\"file\"])\n```\n\n### Revision Roadmap\n```\ndelegate_task(goal=\"Convert editorial decision + reviewer reports into structured Revision Roadmap: prioritized action items, estimated effort, dependency mapping.\", context=\"Use agents/revision_coach.md\", toolsets=[\"file\"])\n```\n\n## 6 Modes\n\n| Mode | Trigger | Agents |\n|------|---------|--------|\n| `full` | \"Review paper\" | All 7 |\n| `re-review` | \"Check revisions\" | 2→3→4→6 |\n| `quick` | \"Quick review\" | 6 only (EIC assessment) |\n| `methodology-focus` | \"Check methodology\" | 2 only |\n| `guided` | \"Guide me to improve\" | Socratic: 6 with user interaction |\n| `calibration` | \"Calibrate reviewer\" | All + calibration metrics output |\n\n## Calibration Mode\nMeasures reviewer accuracy: FNR (False Negative Rate), FPR (False Positive Rate), AUC. Requires ground-truth labels on prior reviewed papers.\n\n## Critical Rules\n1. ⚠️ Reviewers are paper-blind (don't see author info)\n2. ⚠️ Every criticism must include specific actionable suggestion\n3. ⚠️ Calibration mode requires 5+ ground-truth papers\n\n## Security & Privacy\n\n**Multi-agent design disclosure:** This skill delegates review tasks across multiple subagents via `delegate_task`. Manuscript content and intermediate review outputs are processed by these agents. Use only with manuscripts you are comfortable having processed through the AI provider's delegated-agent workflow. Remove confidential material not needed for review.\n\n**Tool access:** Subagents are granted only `file` tools for reading/writing review outputs. No terminal, web, or system tools are exposed.\n\n**Agent files:** The `agents/` directory contains academic peer-review prompt templates (role definitions, scoring rubrics, methodology guidelines). These are task instructions loaded as `context` in `delegate_task` calls — NOT system prompt overrides.\n\nFile v1.0.3:_meta.json\n\n{\n  \"ownerId\": \"kn725tchg3gp72a78w22qa07p584f267\",\n  \"slug\": \"academic-paper-reviewer\",\n  \"version\": \"1.0.3\",\n  \"publishedAt\": 1778976343072\n}\n\nFile v1.0.3:references/calibration_mode_protocol.md\n\n# Calibration Mode Protocol\n\n**Status**: v3.2\n**Parent skill**: `academic-paper-reviewer`\n**Mode name**: `calibration`\n**Purpose**: Measure this reviewer's own false-negative rate (FNR), false-positive rate (FPR), and balanced accuracy against a user-supplied gold-standard set, then attach the resulting error profile as a confidence disclosure to subsequent reviews in the same session.\n\n---\n\n## Why this mode exists\n\nA single LLM reviewer produces an absolute 0-100 rubric score, but that score is weakly interpretable without knowing the reviewer's error profile. Two reviewers could give the same paper a 65, yet one might systematically over-score weak methodology papers and the other might systematically under-score cross-disciplinary work. Absolute scores don't reveal this.\n\nLu et al. (2026, Nature 651:914-919) demonstrated in Table 1 that an LLM-based Automated Reviewer can approach human balanced accuracy (0.65 vs human 0.67-0.73 on 500 ICLR 2022 papers) while having a dramatically different error profile: FNR 0.17 vs human 0.52, at the cost of FPR 0.50 vs human 0.17-0.34. Human reviewers miss half of the papers that should be rejected; the Automated Reviewer misses very few but over-rejects more.\n\nTranslation for ARS: **our reviewer has an error profile too, and we do not currently measure it.** Calibration mode closes that gap. It does not try to make the reviewer perfect; it makes the reviewer's imperfections legible.\n\n---\n\n## Inputs\n\n1. **Gold-standard set**: 5-20 papers the user has labelled with known outcomes. Minimum 5; recommended 10-15. Each entry:\n   - Paper file path or text\n   - Ground-truth label: `accept`, `reject`, or `borderline`\n   - Venue context (journal/conference, tier)\n   - Optional: human reviewer scores for comparison\n\n2. **Domain specification**: the user's target field, used to seed `field_analyst_agent`. Calibration for \"machine learning venues\" is not valid for \"qualitative education research\" — error profiles are domain-specific.\n\n3. **Session persistence**: the error profile is cached for the **current session only**. No cross-session caching, no `~/.ars_calibration_cache/` directory. Calibration is explicitly opt-in per the v3.2 design decision: the user decides when to spend tokens on calibration, and a new session starts fresh. If the user wants to reuse a profile across sessions, they re-run calibration or paste a prior Calibration Report as a session prompt.\n\n---\n\n## Process\n\n### Phase 0: Intake\n\n- Verify the set has at least one `accept` and one `reject` (otherwise FNR or FPR is undefined).\n- If all labels are on one side, refuse to proceed and ask the user for at least one counter-example.\n- Warn if n < 10: \"Calibration with fewer than 10 papers produces wide confidence intervals. Results should be treated as directional, not conclusive.\"\n\n### Phase 1: Run `full` mode on each gold paper, with ensembling\n\nFor each paper, run the standard `full` review pipeline **5 times** (ensembling, per Lu 2026 Methods A.1.1). Each run uses a fresh context window to avoid within-session bias. Aggregate:\n- Median rubric score per dimension\n- Variance across the 5 runs (reported as a stability indicator)\n- Editorial decision (majority vote across 5)\n\n**Cross-model verification**: In calibration mode, `ARS_CROSS_MODEL` is **default-on** rather than opt-in. At least one of the 5 runs should use a different model family if available, to avoid single-model blind spots. If no cross-model is configured, emit a warning and run all 5 on the primary model.\n\n### Phase 2: Build the confusion matrix\n\nCompare reviewer's majority-vote decision against the user's ground-truth label.\n\n- `borderline` ground truth papers are excluded from the binary confusion matrix but reported separately (see Phase 3).\n- Map `Accept` and `Minor Revision` reviewer decisions → positive. Map `Major Revision` and `Reject` → negative. This follows Lu 2026 Table 1's binarization.\n\nCompute:\n\n| Metric | Formula | Report with |\n|---|---|---|\n| Balanced accuracy | (TPR + TNR) / 2 | 95% CI via bootstrap (1000 resamples) |\n| FNR (miss rate) | FN / (FN + TP) | Same |\n| FPR (false alarm) | FP / (FP + TN) | Same |\n| AUC | ROC over rubric-score threshold | Same |\n| Calibration error | Mean &#124;rubric_score - ground_truth_severity&#124; | Per-dimension |\n\n### Phase 3: Borderline handling\n\nBorderline papers don't enter the binary matrix but are useful for rubric-score calibration. For each borderline paper, report:\n- The reviewer's rubric score\n- The reviewer's decision\n- Whether the reviewer's decision respects the user's \"this is borderline\" signal (i.e., did it correctly land in Major Revision rather than confidently Accept or Reject?)\n\nA reviewer that confidently Accepts or Rejects borderline papers has a \"confidence miscalibration\" problem even if its binary accuracy looks fine.\n\n### Phase 4: Produce the Calibration Report\n\nOutput document structured as:\n\n```\n# Calibration Report for <Reviewer Instance>\nDomain: <domain>\nGold set: n=<N> (accept=<a>, reject=<r>, borderline=<b>)\nRuns per paper: 5 (ensembled)\nCross-model: <yes/no, model families used>\n\n## Summary metrics\n- Balanced accuracy: 0.XX [95% CI: 0.XX - 0.XX]\n- FNR: 0.XX [95% CI ...]\n- FPR: 0.XX [95% CI ...]\n- AUC: 0.XX\n- Ensemble stability: <mean std of rubric scores across runs>\n\n## Comparison to Lu 2026 Table 1 baselines\n| Metric | This reviewer | Lu 2026 Automated Reviewer | Lu 2026 Human |\n|---|---|---|---|\n| Balanced accuracy | X | 0.65 | 0.67-0.73 |\n| FNR | X | 0.17 | 0.52 |\n| FPR | X | 0.50 | 0.17-0.34 |\n\n(Note: Lu 2026 numbers are for ML venues specifically. Compare with caution outside ML.)\n\n## Per-dimension calibration error\n<table of 7 review dimensions with mean absolute calibration error>\n\n## Systematic biases detected\n<natural-language narrative identifying patterns, e.g.\n \"Reviewer tends to over-score originality on cross-disciplinary papers\"\n \"Reviewer under-scores qualitative methodology by ~8 points vs ground truth\"\n>\n\n## Recommendations for session use\n- Treat this reviewer's rubric scores as having calibration error ±X points\n- For accept/reject decisions, the reviewer misses X% of reject cases (FNR)\n- For decisions near the accept/reject boundary, escalate to human judgement\n```\n\n### Phase 5: Session attachment\n\nIf session persistence is enabled, the Calibration Report is attached to every subsequent review in the same session as a **confidence disclosure header**. The disclosure appears in the editorial letter before the verdict:\n\n```\n> **Reviewer Confidence Disclosure (from calibration session <id>):**\n> This reviewer has measured balanced accuracy 0.XX, FNR 0.XX, FPR 0.XX on a\n> gold set of <N> papers in <domain>. Rubric scores below have calibration\n> error ±X points. Treat borderline decisions with human judgement.\n```\n\nThis is non-negotiable in calibration-enabled sessions: the user cannot hide the disclosure. The point of calibration is to make error profiles legible; suppressing the disclosure defeats the mode.\n\n---\n\n## Ensembling methodology notes\n\nLu 2026 Methods A.1.1 describes reviewer ensembling across 5 independent runs with majority voting. This mode follows that spec with two changes:\n\n1. **Median instead of mean for rubric scores**: mean is vulnerable to single-run outliers (e.g., a run that hallucinates a methodological flaw); median is robust.\n2. **Fresh context per run**: Lu 2026 allowed within-session memory across runs. ARS uses fresh context to prevent cascading errors from a single run's misreading.\n\nUsers with token budget concerns can reduce `runs_per_paper` to 3. Below 3, ensembling is meaningless — do not allow 1 or 2.\n\n---\n\n## Failure cases this mode does NOT fix\n\nCalibration reports this reviewer's error profile on a **specific** gold set in a **specific** domain. It does not:\n\n- Predict performance on papers outside that domain\n- Detect frame-lock within a single paper review (that's `devils_advocate_reviewer` territory)\n- Catch implementation-bug-as-finding cases (that's the AI Research Failure Mode Checklist, ROADMAP_v3.2.md item 2)\n- Replace the `re-review` mode for revision verification\n\nIf the user's gold set is itself biased (e.g., all papers from one lab, all from one year), calibration reports a biased profile. Emit a warning during intake if papers share obvious metadata clusters.\n\n---\n\n## Integration with existing modes\n\n| Existing mode | Interaction with calibration |\n|---|---|\n| `full` | Calibration runs `full` 5x per gold paper. No change to `full` itself. |\n| `re-review` | Calibration profile attaches to re-review decisions. |\n| `quick` | Calibration profile attaches. Confidence disclosure notes that `quick` has additional uncalibrated error on top of the measured profile. |\n| `methodology-focus` | Calibration should ideally be run with methodology-heavy gold papers if this mode is the user's target. |\n| `guided` | Not applicable — guided mode is Socratic dialogue, rubric scores are not the primary output. |\n\n---\n\n## Resolved design decisions (2026-04-09)\n\n- **Activation**: opt-in only. User invokes `calibration` mode explicitly. ARS does not auto-calibrate on first use in a new domain.\n- **Persistence**: session-scoped only. No cross-session caching of profiles, no `~/.ars_calibration_cache/`, no privacy questions about storing paper content on disk.\n- **Shipped gold sets**: not planned for v3.2. Users bring their own gold set. Shipping a built-in ML gold set was considered and rejected to avoid domain-coverage bias and staleness.\n- **Continuous/self-calibration**: rejected. Using the reviewer's own historical decisions as pseudo-ground-truth is circular and would make the error profile look better over time without actually improving accuracy.\n\n---\n\n## References\n\n- Lu, C. et al. (2026). Towards end-to-end automation of AI research. *Nature* 651, 914-919. doi:10.1038/s41586-026-10265-5 — Table 1 (reviewer validation), Methods A.1.1 (ensembling).\n- Efron, B. & Tibshirani, R. J. (1993). *An Introduction to the Bootstrap*. Chapman & Hall/CRC — bootstrap CI methodology.\n- ARS `shared/cross_model_verification.md` — cross-model reviewer integration.\n- ARS `academic-paper-reviewer/references/quality_rubrics.md` — scoring rubric definitions.\n\n## v3.6.2 sprint contract status\n\nv3.6.2 introduces sprint contracts for `reviewer_full` and `reviewer_methodology_focus` only. A template for this mode will follow in a subsequent patch release. Until then, this mode runs without contract enforcement and retains its pre-v3.6.2 behaviour.\n\nFile v1.0.3:references/changelog.md\n\n# Changelog\n\n| Version | Date | Changes |\n|---------|------|---------|\n| 1.4 | 2026-03-08 | Quality rubrics reference (0-100 scoring with 5 descriptors per dimension, weighted aggregation formula, decision mapping); Quick Mode Selection Guide; Dimension Scores upgraded from optional 1-5 to required 0-100 with rubric descriptors |\n| 1.3 | 2026-03-05 | DA vs R3 role boundaries with explicit responsibility tables; CRITICAL finding criteria with concrete examples; Consensus classification (CONSENSUS-4/3/SPLIT/DA-CRITICAL); Confidence Score weighting rules; Asian & Regional Journals reference (TSSCI + Asia-Pacific + OA options) |\n| 1.2 | 2026-03 | Added statistical reporting standards reference; enhanced methodology_reviewer_agent with statistical reporting adequacy sub-step |\n| 1.1 | 2026-02 | Added Devil's Advocate Reviewer (7th agent), added re-review mode, expanded review team from 4 to 5 |\n| 1.0 | 2026-02 | Initial version: 6 agents, 4 modes, 3-phase workflow |\n\nFile v1.0.3:references/editorial_decision_standards.md\n\n# Editorial Decision Standards — Criteria for Editorial Decision Making\n\nThis document defines the explicit criteria for Accept / Minor Revision / Major Revision / Reject decisions, for use by `eic_agent` and `editorial_synthesizer_agent`.\n\n---\n\n## 1. Decision Categories\n\n### Accept\n\n**Definition**: The paper can be published without further review.\n\n**Criteria**:\n- Average score across all universal dimensions >= 4.0\n- No dimension scores below 3.0\n- At least 3/4 reviewers recommend Accept or Minor Revision\n- No unresolved major academic issues\n\n**Conditions**:\n- May include minor copyediting suggestions\n- May require final formatting adjustments\n- Does not need to be sent for review again\n\n**Typical scenarios**:\n- Paper has undergone multiple revision rounds, all issues resolved\n- Rare first-pass acceptance (< 5% of submissions at top-tier journals)\n\n---\n\n### Minor Revision\n\n**Definition**: The paper is fundamentally acceptable and can be published after limited modifications; typically does not need to be sent for review again after revision.\n\n**Criteria**:\n- Average score across all universal dimensions >= 3.5\n- No dimension scores below 2.5\n- At least 3/4 reviewers recommend Accept or Minor Revision\n- Issues can be resolved within 2-4 weeks\n- Modifications do not involve restructuring core arguments or methods\n\n**Typical revision items**:\n- Supplementing a small number of references\n- Clarifying certain methodology description details\n- Improving clarity of argumentation\n- Correcting citation format\n- Adding discussion of limitations\n- Adjusting conclusion wording (avoiding overclaiming)\n\n**Response requirements**:\n- Authors must respond to reviewer comments item by item\n- After revision, reviewed by EIC (usually not sent for external review again)\n- Revision deadline: 2-4 weeks\n\n---\n\n### Major Revision\n\n**Definition**: The paper has potential but has significant issues, requiring substantial revision followed by re-review.\n\n**Criteria**:\n- Universal dimension average score between 2.5-3.4\n- Some dimensions may score below 2.5 (but not fatal)\n- At least 2/4 reviewers recommend Major Revision or better\n- Issues are serious but fixable (not fundamental design flaws)\n- Revision requires 6-8 weeks of work\n\n**Typical revision items**:\n- Re-analyzing data (additional analysis or correcting errors)\n- Substantially rewriting literature review (missing key references)\n- Supplementing additional data collection\n- Reorganizing paper structure\n- Correcting significant methodological flaws\n- Strengthening theoretical framework application\n- Adding robustness checks\n\n**Response requirements**:\n- Authors must write a detailed point-by-point response letter\n- After revision, sent for re-review (may go back to original reviewers or new reviewers)\n- Revision deadline: 6-8 weeks\n- Typically a maximum of 2 rounds of Major Revision allowed\n\n---\n\n### Reject\n\n**Definition**: The paper is not suitable for publication in this journal, even with revision.\n\n**Criteria (meeting any one may trigger Reject consideration)**:\n- Universal dimension average score < 2.5\n- Any core dimension (methodology, evidence) = 1\n- At least 3/4 reviewers recommend Reject\n- Fundamental unfixable issues exist\n\n**Reject subtypes**:\n\n| Subtype | Description | Suggestion |\n|---------|-------------|-----------|\n| **Reject — Out of Scope** | Topic not within journal scope | Recommend more suitable journals |\n| **Reject — Fundamental Flaw** | Fatal flaw in research design | Suggest redesigning the research |\n| **Reject — Insufficient Contribution** | Lacks originality or incremental contribution | Suggest how to strengthen contribution |\n| **Reject — Premature** | Paper not yet mature enough | Suggest specific improvement directions |\n| **Reject — Resubmit Encouraged** | Has potential but needs fundamental restructuring | Provide detailed restructuring suggestions |\n\n**Even with Reject, must**:\n- Affirm the paper's merits\n- Provide specific improvement suggestions\n- Recommend more suitable journals (if it's a scope issue)\n- Maintain professional, respectful tone\n\n---\n\n## 2. Decision Matrix\n\n### Decision Matrix Based on Reviewer Recommendations\n\n| EIC | R1 | R2 | R3 | -> Recommended Decision |\n|-----|----|----|-----|----------------------|\n| Accept | Accept | Accept | Accept | **Accept** |\n| Accept | Accept | Accept | Minor | **Accept** (with suggestions) |\n| Accept | Accept | Minor | Minor | **Minor Revision** |\n| Accept | Minor | Minor | Minor | **Minor Revision** |\n| Minor | Minor | Minor | Minor | **Minor Revision** |\n| Minor | Minor | Minor | Major | **Minor-to-Major** (depends on specific issues) |\n| Minor | Minor | Major | Major | **Major Revision** |\n| Minor | Major | Major | Major | **Major Revision** |\n| Major | Major | Major | Major | **Major Revision** |\n| Major | Major | Major | Reject | **Major Revision** (last chance) |\n| Major | Major | Reject | Reject | **Reject** (resubmit encouraged) |\n| Major | Reject | Reject | Reject | **Reject** |\n| Reject | Reject | Reject | Reject | **Reject** |\n\n### Special Situation Handling\n\n**Split Decision (evenly divided)**:\n- Example: Accept + Accept + Reject + Reject\n- EIC (or synthesizer) needs to deeply analyze the cause of disagreement\n- Lean toward conservative strategy: Major Revision, requiring the author to respond to the Reject side's comments\n- May consider inviting a fifth reviewer\n\n**One Outlier (one unusual opinion)**:\n- Example: Minor + Minor + Minor + Reject\n- Carefully examine the Reject rationale\n- If the rationale is valid and others missed it, escalate to Major Revision\n- If the rationale is insufficient, maintain Minor Revision but mention the opinion in the Decision Letter\n\n---\n\n## 3. Decision Confidence Calibration\n\n### Impact of Reviewer Confidence Score\n\n| Confidence | Impact on Decision |\n|-----------|-------------------|\n| 5 (Very High) | This reviewer's opinion carries the highest weight |\n| 4 (High) | Standard weight |\n| 3 (Medium) | Standard weight, but reduced in case of disagreement |\n| 2 (Low) | For reference only, not used as a decisive opinion |\n| 1 (Very Low) | Ignore this reviewer's recommendation (but retain specific comments) |\n\n### Cross-Dimension Severity Assessment\n\n| Situation | Severity | Handling |\n|-----------|----------|---------|\n| Methodology has fatal flaw (R1 score = 1) | Critical | Even if other dimensions are excellent, lean toward Reject |\n| Major literature review omission (R2 score = 2) | Serious | Major Revision, require supplementation |\n| Cross-disciplinary perspective overlooked (R3 score = 2) | Moderate | Minor/Major, depends on other dimensions |\n| Poor writing quality (score = 2) | Minor | Does not affect academic decision, but require language revision |\n\n---\n\n## 4. Revision Round Policy\n\n### Standard Policy\n\n| Round | Expectation | Handling |\n|-------|-------------|---------|\n| R1 (First revision) | Respond to all reviewer comments | Send for re-review or EIC review |\n| R2 (Second revision) | Respond to residual issues | Usually EIC makes final decision |\n| R3 (Third revision) | Very rare, usually only handling formatting | EIC makes final decision |\n\n### Upgrade/Downgrade Rules\n\n- Minor Revision with incomplete revisions -> May escalate to Major Revision\n- Major Revision with excellent revisions -> May downgrade to Minor Revision or Accept\n- Major Revision with insufficient revisions -> May Reject (infinite revision cycles are not encouraged)\n- Beyond 2 rounds of Major Revision -> Strongly recommend Accept or Reject, no further extension\n\n---\n\n## 5. Professional Ethics of Editorial Review\n\n### Reviewer Ethics\n\n1. **Confidentiality**: The review process and paper content are confidential\n2. **Conflict of interest**: Recuse if there is a collaborative or competitive relationship with the author\n3. **Timeliness**: Complete the review within the committed timeframe\n4. **Constructiveness**: Even when recommending Reject, provide constructive feedback\n5. **Impartiality**: No bias based on author's gender, race, institution, or nationality\n6. **No plagiarism**: Do not use unpublished ideas seen during review\n7. **Appropriate language**: Avoid personal attacks, sarcasm, or demeaning language\n\n### Editor Ethics\n\n1. **Fair decision**: Based on academic quality, not influenced by external pressure\n2. **Transparent process**: Decision letter must clearly explain the rationale\n3. **Reasonable deadlines**: Give authors sufficient revision time\n4. **Appeal channel**: Authors have the right to respond to or challenge review comments\n5. **Consistent standards**: Papers of similar quality should receive similar decisions\n\n### Ethical Considerations for Special Situations\n\n| Situation | Ethical Handling |\n|-----------|-----------------|\n| Author is your student/colleague | Must recuse or disclose the relationship |\n| Paper's viewpoint is opposite to yours | Evaluate argument quality, not correctness of position |\n| Paper uses your theory but misunderstands it | May point it out but cannot require citation of your own work |\n| Suspected data fabrication | Report to EIC; journal initiates investigation procedure |\n| Paper is similar to your ongoing research | Disclose potential conflict of interest |\n\nFile v1.0.3:references/guided_mode_protocol.md\n\n# Guided Mode (Socratic Guided Review)\n\nThe design philosophy of Guided mode is to **help authors understand the paper's problems themselves**, rather than passively receiving revision instructions.\n\n### How It Works\n\n```\nPhase 0: Normal Field Analysis execution\nPhase 1: Normal execution of 5 reviews (but not all displayed immediately)\nPhase 2: Does not produce full Editorial Decision; enters dialogue mode instead\n```\n\n### Dialogue Flow\n\n1. **EIC opens**: First points out 1-2 core strengths of the paper (building confidence), then raises the most critical structural issue\n2. **Wait for author response**: Author thinks, responds, or asks questions\n3. **Progressive revelation**: Based on the author's level of understanding, gradually reveals deeper issues\n4. **Methodology focus**: When author is ready, introduce Reviewer 1's methodology perspective\n5. **Domain perspective**: Introduce Reviewer 2's domain expertise perspective\n6. **Cross-disciplinary challenge**: Introduce Reviewer 3's unique perspective\n7. **Devil's Advocate**: Finally introduce Devil's Advocate's core challenges and strongest counter-arguments\n8. **Wrap up**: When all key issues have been discussed, provide a structured Revision Roadmap\n\n### Dialogue Rules\n\n- Each response limited to 200-400 words (avoid information overload)\n- Use more questions, fewer commands (\"Do you think this sampling strategy can capture phenomenon X?\" rather than \"the sampling is flawed\")\n- When author's response shows understanding, affirm and move forward\n- When author's response veers off topic, gently guide back to the main point\n- Can ask the author to read a certain reference before continuing discussion\n\n### v3.6.2 sprint contract status\n\nv3.6.2 introduces sprint contracts for `reviewer_full` and `reviewer_methodology_focus` only. A template for this mode will follow in a subsequent patch release. Until then, this mode runs without contract enforcement and retains its pre-v3.6.2 behaviour.\n\nFile v1.0.3:references/integration_guide.md\n\n# Pipeline Usage Example\n\n```\nUser: I want to write a paper about AI in higher education quality assurance, from research to submission\n\nStep 1: deep-research -> Research report\nStep 2: academic-paper -> Paper first draft\nStep 3: integrity check -> 100% verification of references/data\nStep 4: academic-paper-reviewer (full) -> 5 review reports + Revision Roadmap\nStep 5: academic-paper (revision) -> Revised manuscript\nStep 6: academic-paper-reviewer (re-review) -> Verification review\nStep 7: (if needed) academic-paper (revision) -> Second revised manuscript\nStep 8: integrity check (final) -> Final 100% verification\nStep 9: academic-paper (format-convert) -> Final paper\n```\n\nFile v1.0.3:references/quality_rubrics.md\n\n# Quality Rubrics for Academic Paper Review\n\n## Purpose\n\nProvides calibrated scoring rubrics for the 7 review dimensions used by all reviewers (R1, R2, R3, DA). Ensures consistent, reproducible scoring across different papers and review sessions.\n\n## Known error profile (v3.2)\n\nThese rubrics define *what* to measure, not *how accurate* the measurement is. A single LLM reviewer's absolute rubric score has calibration error that depends on domain, paper type, and model version.\n\nFor users who want to know this reviewer's empirical FNR / FPR / balanced accuracy before relying on these rubric scores, run the opt-in **calibration mode** (see `calibration_mode_protocol.md`). Calibration mode compares this reviewer's decisions against a user-supplied gold set and produces a Calibration Report that attaches as a confidence disclosure to subsequent reviews in the same session.\n\nWithout calibration, treat rubric scores as *ordinally* meaningful (papers scored 85 are better than papers scored 65) but *not cardinally* interpretable (a 85 does not guarantee venue acceptance).\n\n## Scoring Scale\n\nAll dimensions scored 0-100. Final weighted score determines editorial decision.\n\n## Decision Mapping\n\n| Weighted Average | Decision |\n|-----------------|----------|\n| >= 80 | Accept |\n| 65-79 | Minor Revision |\n| 50-64 | Major Revision |\n| < 50 | Reject |\n\n---\n\n## Dimension 1: Originality (Weight: 20%)\n\n| Score Range | Descriptor | Behavioral Indicators |\n|------------|------------|----------------------|\n| 90-100 | Exceptional | Novel theoretical framework supported by empirical evidence; opens entirely new research direction; implications span 3+ fields; no prior work addresses this exact question |\n| 75-89 | Strong | Novel methodology OR novel application of existing theory to new context; clear contribution beyond incremental extension; implications for 2+ fields |\n| 60-74 | Adequate | Extends existing framework with new data, population, or context; contribution is clear but incremental; single-field implications |\n| 45-59 | Weak | Replicates existing study with minor variations; contribution is marginal; \"so what?\" question not convincingly answered |\n| < 45 | Insufficient | No discernible original contribution; duplicates existing work without justification; purely descriptive without analytical insight |\n\n## Dimension 2: Methodological Rigor (Weight: 25%)\n\n| Score Range | Descriptor | Behavioral Indicators |\n|------------|------------|----------------------|\n| 90-100 | Exceptional | Research design perfectly aligned with RQ; all validity threats addressed; appropriate statistical methods with power analysis; transparent reporting (all EQUATOR items); reproducible |\n| 75-89 | Strong | Sound design with minor gaps; most validity threats addressed; appropriate methods with minor reporting omissions; largely reproducible |\n| 60-74 | Adequate | Acceptable design but some validity concerns; methods appropriate but justification lacking; some reporting gaps (missing effect sizes, CIs) |\n| 45-59 | Weak | Design has significant flaws; method choice questionable; multiple reporting gaps; reproducibility doubtful |\n| < 45 | Insufficient | Fundamental design flaws that invalidate findings; inappropriate methods; results cannot be trusted |\n\n## Dimension 3: Evidence Sufficiency (Weight: 25%)\n\n| Score Range | Descriptor | Behavioral Indicators |\n|------------|------------|----------------------|\n| 90-100 | Exceptional | >40 sources, 80%+ peer-reviewed, multi-method triangulation, primary + secondary data, all claims well-supported, counter-evidence acknowledged |\n| 75-89 | Strong | 25-40 sources, 70%+ peer-reviewed, adequate evidence for main claims, some triangulation |\n| 60-74 | Adequate | 15-25 sources, 60%+ peer-reviewed, key claims supported but some gaps, limited triangulation |\n| 45-59 | Weak | <15 sources OR <50% peer-reviewed, several unsupported claims, no triangulation |\n| < 45 | Insufficient | Severely under-sourced, major claims unsupported, relies heavily on grey literature or anecdotal evidence |\n\n## Dimension 4: Argument Coherence (Weight: 15%)\n\n| Score Range | Descriptor | Behavioral Indicators |\n|------------|------------|----------------------|\n| 90-100 | Exceptional | Crystal-clear logical flow from problem -> gap -> RQ -> method -> findings -> implications; every section builds on previous; no logical jumps; counterarguments pre-empted |\n| 75-89 | Strong | Clear logical flow with minor gaps; most transitions well-handled; argument generally persuasive |\n| 60-74 | Adequate | Main argument visible but some sections feel disconnected; occasional logical jumps; conclusions mostly follow from evidence |\n| 45-59 | Weak | Argument structure unclear; significant logical gaps; conclusions overreach evidence; reader must infer connections |\n| < 45 | Insufficient | No coherent argument; sections appear unrelated; conclusions do not follow from evidence; circular reasoning |\n\n## Dimension 5: Writing Quality (Weight: 15%)\n\n| Score Range | Descriptor | Behavioral Indicators |\n|------------|------------|----------------------|\n| 90-100 | Exceptional | Professional academic prose; precise terminology; excellent paragraph structure; zero grammatical errors; appropriate register throughout |\n| 75-89 | Strong | Good academic writing; minor stylistic inconsistencies; few grammatical issues; terminology mostly precise |\n| 60-74 | Adequate | Acceptable writing but room for improvement; some verbose passages; occasional imprecise terminology; some grammar issues |\n| 45-59 | Weak | Below journal standards; frequent verbose/unclear passages; terminology inconsistent; multiple grammar issues |\n| < 45 | Insufficient | Unacceptable writing quality; incomprehensible passages; severe grammar problems; not suitable for peer review |\n\n## Optional Dimensions (reviewer-specific)\n\n### Literature Integration (R2 Domain Expert focus)\n\n| Score Range | Descriptor |\n|------------|------------|\n| 90-100 | Comprehensive coverage of seminal + recent works; identifies theoretical lineage; positions paper precisely in scholarly conversation |\n| 75-89 | Good coverage; most key works cited; reasonable positioning in literature |\n| 60-74 | Adequate but gaps in coverage; some important works missing; positioning somewhat vague |\n| < 60 | Significant literature gaps; key works missing; poor positioning |\n\n### Significance & Impact (R3 Perspective Reviewer focus)\n\n| Score Range | Descriptor |\n|------------|------------|\n| 90-100 | Clear practical implications for policy/practice AND theory; addresses urgent real-world problem; likely to influence field direction |\n| 75-89 | Good practical OR theoretical implications; addresses relevant problem; moderate influence potential |\n| 60-74 | Some implications but narrowly scoped; relevance clear but impact limited |\n| < 60 | Minimal practical or theoretical significance; unclear why this matters |\n\n---\n\n## Aggregation Formula\n\n```\nFinal Score = (Originality x 0.20) + (Methodology x 0.25) + (Evidence x 0.25) + (Coherence x 0.15) + (Writing x 0.15)\n```\n\nOptional dimensions are reported separately and factored into the editorial synthesis narrative but do not change the numerical score.\n\n---\n\n## Calibration Notes\n\n- Scores should reflect the paper's quality relative to the target journal's standards\n- A \"75\" for Nature is not equivalent to \"75\" for a regional journal\n- When in doubt, err toward the middle of a range\n- Reviewers should explicitly state which range descriptor best matches, then fine-tune within that range\n- If two dimensions are at odds (e.g., excellent methodology but weak writing), do NOT average down — report both scores honestly\n\nFile v1.0.3:references/re_review_mode_protocol.md\n\n# Re-Review Mode (Verification Review)\n\nRe-review mode is the dedicated mode for Pipeline Stage 3', designed to **verify whether revisions address the first-round review comments**.\n\n### How It Works\n\n```\nInput:\n1. Original Revision Roadmap (Stage 3 output)\n2. Revised manuscript\n3. Response to Reviewers (optional)\n\nPhase 0: Reads the Revision Roadmap, builds a checklist\nPhase 1: EIC checks each item (other reviewers not activated)\nPhase 2: Editorial Synthesis -> New Decision\n```\n\n### Verification Logic\n\n```\nFor each item in the Revision Roadmap:\n\nPriority 1 (Required):\n  -> Check each item for corresponding changes in the revised manuscript\n  -> Assess revision quality (FULLY_ADDRESSED / PARTIALLY_ADDRESSED / NOT_ADDRESSED / MADE_WORSE)\n  -> All Priority 1 items must be FULLY_ADDRESSED for Accept\n\n**Traceability Rule**: For each Priority 1 item, the reviewer MUST:\n1. Read the author's claim from the Response to Reviewers\n2. Navigate to the stated revision location in the manuscript\n3. Independently verify the claim matches the actual change\n4. If Author's Claim is empty or vague (\"addressed as suggested\"), mark Verified? as `🔍 Cannot verify` and flag in Quality Assessment\n\nPriority 2 (Suggested):\n  -> Check each item\n  -> At least 80% should have a response\n  -> NOT_ADDRESSED items require author explanation\n\nPriority 3 (Nice to Fix):\n  -> Check but does not affect Decision\n```\n\n### New Issue Detection\n\n```\nIn addition to checking old items, EIC also scans for:\n- Whether content added during revision introduces new problems\n- Whether newly added references are correct (but deep verification is left to Stage 4.5 integrity check)\n- Whether revisions cause inconsistencies\n```\n\n### Socratic Guidance After Re-Review\n\n```\nIf Re-Review Decision = Major Revision:\n  -> Activate Residual Coaching (residual issue guidance)\n  -> EIC guides user through Socratic dialogue:\n    1. Gap analysis — \"How many issues did the first round of revisions resolve? Why are the remaining ones hard to address?\"\n    2. Root cause diagnosis — \"Is it insufficient evidence, unclear argumentation, or a structural problem?\"\n    3. Trade-off decisions — \"Which ones can be marked as research limitations?\"\n    4. Action plan — Plan revision approach for each residual issue\n  -> Maximum 5 rounds of dialogue\n  -> User can say \"just fix it\" to skip guidance\n```\n\n### Re-Review Output Format\n\n```markdown\n# Verification Review Report\n\n## Decision\n[Accept / Minor Revision / Major Revision]\n\n## Revision Response Checklist\n\n### Priority 1 — Required Revisions\n\n| # | Original Review Comment | Author's Claim | Response Status | Revision Location | Verified? | Quality Assessment |\n|---|------------------------|---------------|-----------------|-------------------|-----------|-------------------|\n| R1 | [Original text] | [What the author claims to have done in Response to Reviewers] | FULLY_ADDRESSED | Section X.X | ✅ Yes | Adequately addressed; newly added content effectively resolves the issue |\n| R2 | [Original text] | [Author's stated change] | PARTIALLY_ADDRESSED | Section Y.Y | ⚠️ Partial | Partially addressed, but still missing [specific gap] |\n\n### Priority 2 — Suggested Revisions\n\n| # | Original Review Comment | Response Status | Notes |\n|---|------------------------|-----------------|-------|\n| S1 | [Original text] | FULLY_ADDRESSED | -- |\n| S2 | [Original text] | NOT_ADDRESSED | Author explanation: [reason] |\n\n### Priority 3 — Nice to Fix\n\n| # | Original Review Comment | Response Status |\n|---|------------------------|-----------------|\n| N1 | [Original text] | FULLY_ADDRESSED |\n\n## New Issues (Discovered During Revision)\n\n| # | Type | Location | Description |\n|---|------|----------|-------------|\n| NEW-1 | [Type] | Section X.X | [Description] |\n\n## Decision Rationale\n[Rationale based on the checklist]\n\n## Residual Issues (If Any)\n[List unresolved items, suggest marking as Acknowledged Limitations]\n```\n\n## v3.6.2 sprint contract status\n\nv3.6.2 introduces sprint contracts for `reviewer_full` and `reviewer_methodology_focus` only. A template for this mode will follow in a subsequent patch release. Until then, this mode runs without contract enforcement and retains its pre-v3.6.2 behaviour.\n\nFile v1.0.3:references/review_criteria_framework.md\n\n# Review Criteria Framework — Structured Review Criteria Framework\n\nThis document defines universal criteria for academic paper review and type-specific criteria differentiated by paper type. All reviewer agents share this framework.\n\n---\n\n## 1. Universal Review Dimensions\n\nSeven core dimensions applicable to all paper types:\n\n### Dimension 1: Originality — Weight 15%\n\n| Level | Score | Description |\n|-------|-------|-------------|\n| Outstanding | 5 | Proposes entirely new theory/method/evidence that could change the field's direction |\n| Strong | 4 | Has clear new insights or novel combinations, fills a specific research gap |\n| Adequate | 3 | Incremental contribution, reasonable extension of existing knowledge |\n| Weak | 2 | Highly overlapping with existing literature, new contribution unclear |\n| None | 1 | Essentially repeats what is already known |\n\n### Dimension 2: Methodological Rigor — Weight 25%\n\n| Level | Score | Description |\n|-------|-------|-------------|\n| Outstanding | 5 | Impeccable research design, innovative methods executed flawlessly |\n| Strong | 4 | Sound design, appropriate methods, minor room for improvement in execution |\n| Adequate | 3 | Methods basically acceptable, but with some design or execution limitations |\n| Weak | 2 | Methods have significant flaws affecting the credibility of conclusions |\n| Unacceptable | 1 | Methods fundamentally unsuitable for answering the research question, or contain serious errors |\n\n### Dimension 3: Evidence Sufficiency — Weight 20%\n\n| Level | Score | Description |\n|-------|-------|-------------|\n| Outstanding | 5 | Rich, diverse, and persuasive evidence that exceeds expectations |\n| Strong | 4 | Evidence sufficiently supports all major arguments |\n| Adequate | 3 | Most arguments supported by evidence, a few need supplementation |\n| Weak | 2 | Key arguments lack sufficient evidence |\n| Unacceptable | 1 | Serious disconnect between arguments and evidence |\n\n### Dimension 4: Argument Coherence — Weight 15%\n\n| Level | Score | Description |\n|-------|-------|-------------|\n| Outstanding | 5 | Clear arguments, rigorous logic, elegant structure |\n| Strong | 4 | Smooth argumentation, occasional minor logical leaps |\n| Adequate | 3 | Basically coherent, but some inter-paragraph connections are unclear |\n| Weak | 2 | Multiple logical breaks, readers have difficulty following the argument |\n| Unacceptable | 1 | Confused argumentation, core claims cannot be identified |\n\n### Dimension 5: Writing Quality — Weight 10%\n\n| Level | Score | Description |\n|-------|-------|-------------|\n| Outstanding | 5 | Precise and fluent academic English/Chinese, a model of scholarly writing |\n| Strong | 4 | Clear language, occasional minor imperfections that don't affect understanding |\n| Adequate | 3 | Generally readable, with some grammar or word choice issues |\n| Weak | 2 | Frequent language issues that affect understanding |\n| Unacceptable | 1 | Language quality does not meet reviewable standards |\n\n### Dimension 6: Literature Integration — Weight 10%\n\n| Level | Score | Description |\n|-------|-------|-------------|\n| Outstanding | 5 | Comprehensive, contemporary, critically integrated literature with a compelling research gap argument |\n| Strong | 4 | Covers major literature, with good integration and positioning |\n| Adequate | 3 | Basic coverage, but with omissions or insufficient integration |\n| Weak | 2 | Literature is outdated, incomplete, or merely enumerated |\n| Unacceptable | 1 | Seriously insufficient literature review or irrelevant to the topic |\n\n### Dimension 7: Significance & Impact — Weight 5%\n\n| Level | Score | Description |\n|-------|-------|-------------|\n| Outstanding | 5 | Could change policy, practice, or theoretical direction |\n| Strong | 4 | Clear impact on a specific field or practice |\n| Adequate | 3 | Has some academic or practical value |\n| Weak | 2 | Limited scope of impact, mainly academic interest |\n| Marginal | 1 | Difficult to see the significance of the research |\n\n---\n\n## 2. Paper Type-Specific Criteria\n\n### 2.1 Empirical Research\n\nBeyond universal dimensions, specifically focus on:\n\n| Additional Dimension | Review Focus |\n|---------------------|-------------|\n| Research hypothesis clarity | Are hypotheses testable and consistent with theory |\n| Variable operational definitions | Are independent/dependent/control variable definitions precise |\n| Internal validity | Are confounding variables controlled |\n| External validity | Generalizability of results |\n| Statistical reporting completeness | Effect sizes, confidence intervals, assumption testing |\n| Conclusion conservatism | Do conclusions exceed what the data supports |\n\n### 2.2 Theoretical/Conceptual Paper\n\n| Additional Dimension | Review Focus |\n|---------------------|-------------|\n| Conceptual definition precision | Are core concepts clearly delineated |\n| Argument logic structure | Is the premise -> inference -> conclusion logic chain complete |\n| Counterargument handling | Are possible opposing viewpoints considered and addressed |\n| Theoretical novelty | Does it truly advance theoretical development |\n| Testability | Can the theory generate testable propositions |\n\n### 2.3 Literature Review / Meta-analysis\n\n| Additional Dimension | Review Focus |\n|---------------------|-------------|\n| Search strategy | Is it comprehensive and reproducible (PRISMA compliance) |\n| Inclusion/exclusion criteria | Are criteria clear, reasonable, and consistently applied |\n| Bias risk assessment | Is bias risk of included studies assessed |\n| Heterogeneity handling | Is statistical and conceptual heterogeneity appropriately handled |\n| Synthesis method | Goes beyond simple vote counting to achieve critical synthesis |\n| Publication Bias | Is publication bias assessed and discussed |\n\n### 2.4 Case Study\n\n| Additional Dimension | Review Focus |\n|---------------------|-------------|\n| Case selection justification | Why was this case chosen? What does it represent? |\n| Theoretical vs convenience sampling | Is case selection theoretically grounded |\n| Triangulation | Are multiple data sources used |\n| Context description thickness | Is thick description sufficient |\n| Analysis transferability | Are analysis results transferable to other contexts |\n| Researcher reflexivity | Is the researcher's relationship with the case reflected upon |\n\n### 2.5 Policy Analysis / Policy Brief\n\n| Additional Dimension | Review Focus |\n|---------------------|-------------|\n| Policy problem definition | Is the problem clearly defined and evidence-supported |\n| Stakeholder analysis | Are key stakeholders identified |\n| Policy option analysis | Are multiple options proposed and compared |\n| Feasibility assessment | Are policy recommendations practically feasible |\n| Evidence quality | Are policy recommendations based on reliable evidence |\n| Unintended consequences | Are unintended policy impacts considered |\n\n---\n\n## 3. Common Review Pitfalls\n\n### Biases Reviewers Should Avoid\n\n| Pitfall | Description | How to Avoid |\n|---------|-------------|--------------|\n| **Hypercriticism** | Overblowing minor issues, ignoring the paper's overall contribution | Affirm strengths first, then point out issues; distinguish major from minor |\n| **Confirmation Bias** | Only finding evidence supporting pre-existing views | Deliberately seek the paper's merits and counterexamples to your own views |\n| **Preference Projection** | Requiring authors to use \"my method\" rather than evaluating \"the author's method\" | Ask \"can this method answer the question\" rather than \"what would I do\" |\n| **Paradigm Bias** | Using quantitative standards to judge qualitative research (or vice versa) | Use evaluation criteria matching the paper's research paradigm |\n| **Prestige Bias** | Relaxing standards because of the author's institution or past achievements | Focus on the quality of the paper itself |\n| **Novelty Bias** | Only valuing novel research, undervaluing replication studies | Acknowledge the important role of replication in science |\n| **Length Bias** | Long paper = good paper, short paper = sloppy | Evaluate content density, not page count |\n| **Language Discrimination** | Undervaluing research quality due to non-native language imperfections | Distinguish \"language needs polishing\" from \"research quality is poor\" |\n\n### Principles of Constructive Feedback\n\n1. **Specific, not vague**: \"The causal inference in Section 3, paragraph 2 lacks control variables\" is better than \"methodology has problems\"\n2. **Problem + reason + suggestion**: Every criticism should include \"what,\" \"why,\" and \"how to fix\"\n3. **Distinguish required from suggested**: Which changes are mandatory, which are \"nice to have\"\n4. **Acknowledge uncertainty**: \"I'm not sure whether this analysis accounts for X\" is more accurate than \"the author ignored X\"\n5. **Respect the author**: Even if paper quality is poor, the author still invested time and effort\n\n---\n\n## 4. Scoring Aggregation\n\n### Weighted Total Score Calculation\n\n```\nTotal Score =\n  Originality (15%) +\n  Methodological Rigor (25%) +\n  Evidence Sufficiency (20%) +\n  Argument Coherence (15%) +\n  Writing Quality (10%) +\n  Literature Integration (10%) +\n  Significance (5%)\n```\n\n### Score-to-Decision Mapping\n\n| Weighted Total | Recommended Decision | Note |\n|---------------|---------------------|------|\n| 4.5-5.0 | Accept | Very few papers reach this level |\n| 3.5-4.4 | Minor Revision | Overall quality is good, minor revisions needed |\n| 2.5-3.4 | Major Revision | Has potential but needs substantial revision |\n| 1.5-2.4 | Reject (Resubmit) | Fundamental issues need rework, but topic has value |\n| 1.0-1.4 | Reject | Not suitable for this journal or quality below standard |\n\n**Important reminder**: Scores are only reference. The final decision also needs to consider:\n- Whether any single dimension is particularly low (e.g., methodology score of 1), which may lead to Reject even if the overall score is passable\n- Specific content of reviewer comments is more important than numbers\n- Special considerations of the journal (special issue, field development needs, etc.)\n\nFile v1.0.3:references/review_quality_thinking.md\n\n# Review Quality Thinking Framework\n\nA cognitive framework for producing high-quality reviews. Teaches **how to think** about paper quality, not just what to score.\n\n## The Three Lenses\n\nEvery paper should be evaluated through three lenses simultaneously:\n\n### Lens 1: Internal Validity — \"Does the evidence support the claims?\"\n\nAsk in order:\n1. What is the central claim?\n2. What evidence is presented?\n3. Is there a logical chain from evidence to claim? (Warrant)\n4. Are there alternative explanations the authors didn't consider?\n5. Would removing any single piece of evidence collapse the argument?\n\n**If #5 is yes**: The argument depends on a single linchpin. Flag it — the paper's contribution is only as strong as that one piece of evidence.\n\n### Lens 2: External Validity — \"Does this matter beyond this study?\"\n\nAsk in order:\n1. Who is the population of interest?\n2. Does the sample represent that population?\n3. Are the conditions replicable?\n4. Would the findings hold in a different context/culture/time?\n5. What are the boundary conditions the authors don't mention?\n\n**Judgment heuristic**: Most authors overstate generalizability. If the sample is from one university in one country, findings cannot claim to apply \"generally\" without qualification.\n\n### Lens 3: Contribution — \"So what?\"\n\nAsk in order:\n1. What did we know before this paper?\n2. What do we know after?\n3. Is the delta meaningful (not just statistically significant)?\n4. Who benefits from knowing this?\n5. What new questions does this open?\n\n**Judgment heuristic**: If you can't articulate the delta in one sentence, either the contribution is weak or the paper hasn't communicated it clearly. Both are review-worthy observations.\n\n## Common Reviewer Traps\n\n| Trap | Description | How to Avoid |\n|------|-------------|-------------|\n| **Methodological tunnel vision** | Only critiquing methods, ignoring whether the question matters | Start with Lens 3 (contribution) before Lens 1 |\n| **Novelty bias** | Penalizing replication or incremental work | Replication IS valuable; evaluate on execution quality |\n| **Expertise projection** | Expecting the paper to use your preferred method | Evaluate the chosen method on its own terms |\n| **Positivity-severity oscillation** | Being too nice in comments, too harsh in scores | Write the score first, then justify with comments |\n| **Missing forest for trees** | Listing 20 minor issues, missing the one fatal flaw | Always state the single most important issue first |\n\n## Calibration Questions (ask after drafting your review)\n\n1. If this paper were published as-is, would it mislead readers? (If yes → Major Revision or Reject)\n2. Could the authors reasonably address my concerns in one revision cycle? (If no → Reject)\n3. Am I being harder on this paper than I would be on my own work? (Calibration check)\n4. Did I identify at least one genuine strength? (Balance check)\n5. Would my review help the authors improve, even if the paper is rejected? (Constructiveness check)\n\nFile v1.0.3:references/sprint_contract_protocol.md\n\n# Sprint Contract Protocol (v3.6.2)\n\n> Authoritative orchestration reference for the ARS v3.6.2 sprint-contract hard gate.\n> Schema: `shared/sprint_contract.schema.json` (Schema 13.1 since v3.6.6).\n> Templates: `shared/contracts/reviewer/*.json`.\n> Design spec: `docs/design/2026-04-23-ars-v3.6.2-sprint-contract-design.md`.\n>\n> **v3.6.6 cross-reference**: this reviewer protocol is byte-equivalent across v3.6.2 → v3.6.6 (zero-touch promise per §3.6 of `docs/design/2026-04-27-ars-v3.6.6-generator-evaluator-contract-design.md`). The v3.6.6 release adds a parallel generator-evaluator protocol inside `academic-paper` for the in-pair writer / evaluator pair (see `academic-paper/SKILL.md` § \"v3.6.6 Generator-Evaluator Contract Protocol\" and design doc §5).\n\n## 1. Overview\n\nA reviewer sprint contract is a machine-checkable pre-registered acceptance criterion. The orchestrator loads a frozen template, inlines runtime fields (`generated_at`, optional `agent_amendments`), and drives each reviewer through a paper-content-blind Phase 1 followed by a paper-visible Phase 2. The synthesizer then runs a three-step mechanical protocol over the `panel_size` reviewer outputs to emit an editorial decision.\n\nThis protocol exists to destroy the \"read the paper, then rationalise the scoring standard\" drift path. The load-bearing mechanism is the **physical separation of calls**: Phase 1 never sees paper content.\n\n## 2. Two-phase reviewer call\n\nFor each reviewer in `range(panel_size)`:\n\n1. **Prepare contract.** Load template from `shared/contracts/<domain>/<mode>.json`. Populate `generated_at` (ISO-8601 UTC). Optionally populate `agent_amendments` (field-specific notes from `field_analyst_agent`). Run `check_sprint_contract.py` on the in-memory object; abort on error.\n2. **Phase 1 call (paper-content-blind).**\n   - System prompt: the `### Phase 1 — Paper-content-blind pre-commitment` sub-section of the reviewer agent's `## v3.6.2 Sprint Contract Protocol` block.\n   - User content: contract JSON + paper metadata ONLY (`title`, `field`, `word_count`).\n   - Expected output: `## Contract Paraphrase`, `## Scoring Plan`, terminal `[CONTRACT-ACKNOWLEDGED]` tag.\n3. **Phase 1 output lint.** See §4 below.\n4. **Phase 2 call (paper-visible).**\n   - System prompt: the `### Phase 2 — Paper-visible review` sub-section of the same `## v3.6.2 Sprint Contract Protocol` block.\n   - User content: contract JSON (re-injected) + Phase 1 output wrapped in `<phase1_output>...</phase1_output>` data delimiter + full paper.\n   - Expected output: optional `## Scoring Plan Dissent`, `## Dimension Scores`, `## Failure Condition Checks`, `## Review Body`, `## Editorial Decision`.\n5. **Phase 2 output lint.** See §5 below.\n6. **Panel cardinality invariant.** After all reviewers complete, verify `len(usable_phase2_outputs) == panel_size`. If any reviewer was dropped, emit `[PANEL-SHRUNK]` and abort the round (see §6).\n7. Feed usable Phase 2 outputs into synthesizer (see §7).\n\n## 3. Contract injection\n\n- **Template on disk is frozen.** Do not mutate. Deep-copy into an in-memory dict.\n- **Runtime-only fields:** `generated_at`, `agent_amendments.stage_specific_notes`, `agent_amendments.additional_measurement_hints`.\n- **Baseline fields are orchestrator-immutable.** Schema cannot enforce this; the orchestrator must not rewrite `acceptance_dimensions` / `failure_conditions` / `measurement_procedure` / `override_ladder` / `mode` / `stage` / `contract_id` / `baseline_version` / `panel_size` between template load and injection. Optional: emit sha256 of baseline-field subset to audit log for drift detection.\n\n## 4. Phase 1 output lint\n\nStructural checks (orchestrator, not validator). On failure retry Phase 1 once with the specific lint gap hinted in the system prompt; second failure aborts that reviewer.\n\n- Required sections in order: `## Contract Paraphrase`, `## Scoring Plan`, terminal `[CONTRACT-ACKNOWLEDGED]`.\n- Paraphrase paragraph count ≥ `measurement_procedure.paraphrase_minimum_dimensions` (for `\"all\"`, one paragraph per dimension; for integer `k`, at least `k` paragraphs each matching a distinct dimension).\n- `## Scoring Plan` has one `### <Dn>: <name>` subsection per acceptance dimension (always full coverage, regardless of `paraphrase_minimum_dimensions`).\n- Each `scoring_plan` subsection contains lines matching `measurement_procedure.scoring_plan_schema.required`.\n- Phase 1 content refers to `<title>`, `<field>`, `<word_count>` only; no specific paper content. Not schema-enforced; behavioural rule in reviewer prompt.\n\n**Lint is structural, not semantic.** A reviewer can in principle pass this lint by emitting generic boilerplate triggers — semantic judgement (whether triggers are concrete and discriminating) is deferred to a post-v3.6.2 judge-agent layer.\n\nOn second Phase 1 failure: emit `[PROTOCOL-VIOLATION: reviewer=<role>, contract=<id>, phase1_lint_failed=true]` and mark this reviewer unusable.\n\n## 5. Phase 2 output lint\n\nStructural checks run before handoff to synthesizer. **No Phase 2 retry** (reviewer has seen the paper; a second call is tainted) EXCEPT the multi-dissent case below.\n\n- Required sections: `## Dimension Scores`, `## Failure Condition Checks`, `## Review Body`, `## Editorial Decision`.\n- `## Dimension Scores` has one `### <Dn>: <name>` subsection per contract dimension; each carries a value in `$defs.score` (`block | warn | pass`).\n- `## Failure Condition Checks` has one subsection per `failure_conditions[]` entry with `fired: true | false`.\n- **Multi-dissent rule:** If `## Scoring Plan Dissent` names two or more `dimension_id` entries, orchestrator aborts this reviewer and retries from **Phase 1** once. If the retried Phase 1/2 also multi-dissents, mark the reviewer unusable (`[PROTOCOL-VIOLATION]`). One-dimension-per-reviewer-per-Phase-2-call is the cap.\n- **Consistency check (structural):** For every dimension not under dissent, the Phase 2 score must substring-match the reviewer's Phase 1 `scoring_plan` trigger tokens. Vacuous triggers bypass this check — documented limitation.\n- `## Editorial Decision` is one of the `action` values derivable from `## Failure Condition Checks` via the synthesizer precedence rule (§8 step 3). Inconsistency marks the reviewer unusable.\n\nOn any Phase 2 lint failure other than multi-dissent: emit `[PROTOCOL-VIOLATION]` and mark reviewer unusable. Do not synthesise a substitute score for the synthesizer.\n\n## 6. Multi-reviewer orchestration\n\n- **Independent cycles.** Each of the `panel_size` reviewers runs its own Phase 1 + Phase 2. Failures in one do not pause the others.\n- **Panel cardinality invariant (§2 step 6).** After all reviewers complete, if `len(usable_phase2_outputs) < panel_size`, abort the editorial round with `[PANEL-SHRUNK]`. Do not silently recompute `cross_reviewer_quantifier` thresholds against a smaller panel — the contract's published aggregation semantics bind on a specific `panel_size`.\n- **Operational monitor.** Track `[PANEL-SHRUNK]` rate in real SR runs. If > 5% of rounds abort in first 3 months, v3.6.3 introduces graceful-degradation fallback.\n\n## 7. Reviewer panel mapping\n\n| mode                          | panel_size | invoked reviewers |\n|-------------------------------|------------|-------------------|\n| `reviewer_full`               | 5          | EIC + methodology + domain + perspective + DA |\n| `reviewer_methodology_focus`  | 2          | EIC + methodology (only) |\n| `reviewer_re_review`          | —          | not shipped in v3.6.2; continues pre-v3.6.2 behaviour |\n| `reviewer_calibration`        | —          | not shipped in v3.6.2 |\n| `reviewer_guided`             | —          | not shipped in v3.6.2 |\n\nThe orchestrator uses `mode` to determine the panel and the contract's `panel_size` as the invariant target. SC-11 validator check ensures mode and `panel_size` are consistent.\n\n## 8. Synthesizer three-step protocol\n\nLet `N = contract.panel_size`.\n\n**Step 1 — Build scoring matrix.** For each `acceptance_dimensions[i]`, gather N reviewers' `## Dimension Scores` for that dimension into a length-N array of `$defs.score` values. Dimensions resolved by `id`.\n\n**Step 2 — Evaluate each `failure_conditions[]`.** For each condition:\n\n1. Parse `expression` against the recognised patterns (see §9 vocabulary). Unrecognised → emit `[EXPRESSION-UNRECOGNISED]`, abort synthesizer.\n2. Apply `cross_reviewer_quantifier` with panel-relative thresholds:\n   - `any`: fires if predicate holds for ≥ 1 of N reviewers.\n   - `majority`: for N ≥ 3, fires if ≥ `⌈N/2⌉ + 1`; for N == 2, fires if all 2; for N == 1, vacuous (SC-11 warns).\n   - `all`: fires if predicate holds for all N reviewers.\n3. Record `{condition_id, fired}`.\n\n**Step 3 — Precedence and decision.** Among fired conditions, pick the one with highest `severity`. Ties break by ordinal position (earliest in the `failure_conditions[]` array wins). Emit its `action` as `editorial_decision`.\n\n**Forbidden operations (synthesizer prompt hard constraint):**\n- Introduce aggregation rules not derivable from `cross_reviewer_quantifier` + `severity`.\n- Average or vote-aggregate scores within a single dimension unless `cross_reviewer_quantifier: majority` explicitly requests it.\n- Soften a fired condition's `action` on post-hoc grounds.\n- Synthesise substitute scores for reviewers marked unusable — the round is either complete with `panel_size` usable outputs or `[PANEL-SHRUNK]` aborted.\n\n## 9. Recognised expression vocabulary\n\nSynthesizer recognises the following patterns (with accepted natural-English variants):\n\n1. **Priority-scoped single-match:** `any <priority> dimension scores '<score>'` | `any dimension with priority=<priority> scores '<score>'` | `any <priority>-priority dimension scores '<score>'`\n2. **Priority-scoped count-based:** `two or more <priority> dimensions score '<score>' or worse` | `two or more dimensions with priority=<priority> score '<score>' or worse` (ordering `pass` < `warn` < `block`)\n3. **Universal over priority:** `every <priority> dimension scores '<score>'`\n4. **Single-dimension literal:** `<Dn> scores '<score>'`\n5. **Conjunction:** any of the above joined by `AND`\n\nShipped template coverage:\n- `reviewer/full.json`: F1 pattern 1 (bare mandatory), F2 pattern 2, F3 pattern 1 (`high-priority` variant), F0 pattern 3.\n- `reviewer/methodology_focus.json`: F1 / F2 / F0 pattern 4 (literal D1).\n\nNew expression forms require a PR updating both this §9 and the synthesizer prompt's recognised-pattern list.\n\n## 10. Token cost expectations\n\nReviewer total calls = `2 × panel_size`. For `reviewer_full` that is 5 → 10 calls; for `reviewer_methodology_focus` 2 → 4. Phase 1 input is small (contract + metadata only); Phase 1 output is short (paraphrase + scoring_plan). Real token bound is well below 2x raw increase.\n\n## 11. Failure modes and diagnostics\n\nAudit-log tags the orchestrator may emit:\n\n| Tag | When | Action |\n|-----|------|--------|\n| `[CONTRACT-ACKNOWLEDGED]` | normal Phase 1 completion | none (expected) |\n| `[PROTOCOL-VIOLATION: phase1_lint_failed=true]` | Phase 1 lint fails twice for a reviewer | mark reviewer unusable |\n| `[PROTOCOL-VIOLATION: phase2_lint_failed=<check>]` | Phase 2 lint fails (non multi-dissent) | mark reviewer unusable |\n| `[PROTOCOL-VIOLATION: multi_dissent=true]` | Phase 2 has ≥ 2 dissent entries, retry exhausted | mark reviewer unusable |\n| `[PANEL-SHRUNK: usable=<k>, panel_size=<N>]` | §6 invariant failed | abort editorial round |\n| `[EXPRESSION-UNRECOGNISED: condition_id=<F>, expression=<...>]` | synthesizer step 2.1 | abort synthesizer |\n\nArchive v1.0.2: 39 files, 156982 bytes\n\nFiles: agents/devils_advocate_reviewer_agent.md (15011b), agents/domain_reviewer_agent.md (11729b), agents/editorial_synthesizer_agent.md (13139b), agents/eic_agent.md (8438b), agents/field_analyst_agent.md (9429b), agents/methodology_reviewer_agent.md (13196b), agents/perspective_reviewer_agent.md (14358b), LICENSE (19584b), references/calibration_mode_protocol.md (10542b), references/changelog.md (976b), references/editorial_decision_standards.md (9230b), references/guided_mode_protocol.md (1971b), references/integration_guide.md (680b), references/quality_rubrics.md (7679b), references/re_review_mode_protocol.md (4250b), references/review_criteria_framework.md (10139b), references/review_quality_thinking.md (3015b), references/sprint_contract_protocol.md (11621b), references/statistical_reporting_standards.md (22703b), references/top_journals_by_field.md (12874b), shared/artifact_reproducibility_pattern.md (8921b), shared/benchmark_report_pattern.md (8988b), shared/benchmark_report.schema.json (2789b), shared/collaboration_depth_rubric.md (10810b), shared/compliance_checkpoint_protocol.md (7175b), shared/compliance_report.schema.json (8314b), shared/cross_model_verification.md (11173b), shared/ground_truth_isolation_pattern.md (12649b), shared/handoff_schemas.md (46013b), shared/mode_spectrum.md (4047b), shared/prisma_trAIce_protocol.md (10415b), shared/raise_framework.md (7365b), shared/sprint_contract.schema.json (18954b), shared/style_calibration_protocol.md (7085b), SKILL.md (4977b), templates/editorial_decision_template.md (6446b), templates/peer_review_report_template.md (7479b), templates/revision_response_template.md (7020b), _meta.json (142b)\n\nFile v1.0.2:SKILL.md\n\n---\nname: academic-paper-reviewer\ndescription: \"7-agent paper review system on Hermes Agent. 6 modes (full/re-review/quick/methodology-focus/guided/calibration). 5-panel review with editorial decision, revision roadmap, and calibration metrics. Uses delegate_task for each reviewer. Triggers: review paper, peer review, manuscript review, check revisions, calibrate reviewer, 審稿, 同儕審查, 論文審查.\"\nmetadata:\n  version: \"1.0-hermes-1.0\"\n  last_updated: \"2026-05-16\"\n  status: active\n  adapted_from: \"imbad0202/academic-research-skills\"\n  adapted_for: \"Hermes Agent (deepseek-v4-pro)\"\n  task_type: open-ended\n  license: \"CC BY-NC 4.0\"\n  original_author: \"Cheng-I Wu\"\n  original_license: \"CC BY-NC 4.0\"\n  original_repo: \"https://github.com/Imbad0202/academic-research-skills\"\n  copyright: \"Copyright (c) 2026 Cheng-I Wu\"\n---\n# Academic Paper Reviewer — 7-Agent Review System (Hermes Edition)\n\n📄 **License:** [CC BY-NC 4.0](https://creativecommons.org/licenses/by-nc/4.0/) · Copyright (c) 2026 Cheng-I Wu  \n🔗 **Original:** [Imbad0202/academic-research-skills](https://github.com/Imbad0202/academic-research-skills)  \n🔄 **Adaptation:** Multi-agent review system implemented via `delegate_task` instead of Claude Code's internal agent system. All agent definitions, references, and quality standards preserved unchanged from original. **This adaptation is distributed under the same CC BY-NC 4.0 license.**\n\n## Quick Start\n\n```\nReview this paper for journal submission\n```\n\n## Agent Team\n\n| # | Agent | Role |\n|---|-------|------|\n| 1 | intake_agent | Receive paper, determine review type |\n| 2 | methodology_reviewer | Method rigor assessment |\n| 3 | evidence_reviewer | Evidence sufficiency & citation quality |\n| 4 | argument_reviewer | Logical coherence & argument structure |\n| 5 | domain_reviewer | Domain expertise & literature positioning |\n| 6 | editor_in_chief | Aggregate reviews → editorial decision |\n| 7 | revision_coach | Convert reviews → actionable roadmap |\n\n## Hermes Execution\n\n### Full Mode: 5-Panel Parallel Review\n```\ndelegate_task(tasks=[\n    {\"goal\": \"Review manuscript methodology: design appropriateness, validity threats, replicability. Score 1-5.\", \"context\": \"Use agents/methodology_reviewer.md\", \"toolsets\": [\"file\"]},\n    {\"goal\": \"Review evidence: citation quality, source credibility, evidence hierarchy alignment. Score 1-5.\", \"context\": \"Use agents/evidence_reviewer.md\", \"toolsets\": [\"file\"]},\n    {\"goal\": \"Review argument: logical flow, claim-evidence alignment, counter-argument handling. Score 1-5.\", \"context\": \"Use agents/argument_reviewer.md\", \"toolsets\": [\"file\"]},\n    {\"goal\": \"Review domain positioning: literature coverage, theoretical grounding, contribution significance. Score 1-5.\", \"context\": \"Use agents/domain_reviewer.md\", \"toolsets\": [\"file\"]}\n])\n```\n\n### Editorial Decision\n```\ndelegate_task(goal=\"Aggregate all 4 reviewer reports. Apply weighted scoring (Method 30%, Evidence 25%, Argument 25%, Domain 20%). Issue editorial decision: Accept/Minor Revision/Major Revision/Reject with justification.\", context=\"Use agents/editor_in_chief.md\", toolsets=[\"file\"])\n```\n\n### Revision Roadmap\n```\ndelegate_task(goal=\"Convert editorial decision + reviewer reports into structured Revision Roadmap: prioritized action items, estimated effort, dependency mapping.\", context=\"Use agents/revision_coach.md\", toolsets=[\"file\"])\n```\n\n## 6 Modes\n\n| Mode | Trigger | Agents |\n|------|---------|--------|\n| `full` | \"Review paper\" | All 7 |\n| `re-review` | \"Check revisions\" | 2→3→4→6 |\n| `quick` | \"Quick review\" | 6 only (EIC assessment) |\n| `methodology-focus` | \"Check methodology\" | 2 only |\n| `guided` | \"Guide me to improve\" | Socratic: 6 with user interaction |\n| `calibration` | \"Calibrate reviewer\" | All + calibration metrics output |\n\n## Calibration Mode\nMeasures reviewer accuracy: FNR (False Negative Rate), FPR (False Positive Rate), AUC. Requires ground-truth labels on prior reviewed papers.\n\n## Critical Rules\n1. ⚠️ Reviewers are paper-blind (don't see author info)\n2. ⚠️ Every criticism must include specific actionable suggestion\n3. ⚠️ Calibration mode requires 5+ ground-truth papers\n\n## Security & Privacy\n\n**Multi-agent design disclosure:** This skill delegates review tasks across multiple subagents via `delegate_task`. Manuscript content and intermediate review outputs are processed by these agents. Use only with manuscripts you are comfortable having processed through the AI provider's delegated-agent workflow. Remove confidential material not needed for review.\n\n**Tool access:** Subagents are granted only `file` tools for reading/writing review outputs. No terminal, web, or system tools are exposed.\n\n**Agent files:** The `agents/` directory contains academic peer-review prompt templates (role definitions, scoring rubrics, methodology guidelines). These are task instructions loaded as `context` in `delegate_task` calls — NOT system prompt overrides.\n\nFile v1.0.2:_meta.json\n\n{\n  \"ownerId\": \"kn725tchg3gp72a78w22qa07p584f267\",\n  \"slug\": \"academic-paper-reviewer\",\n  \"version\": \"1.0.2\",\n  \"publishedAt\": 1778943625661\n}\n\nFile v1.0.2:references/calibration_mode_protocol.md\n\n# Calibration Mode Protocol\n\n**Status**: v3.2\n**Parent skill**: `academic-paper-reviewer`\n**Mode name**: `calibration`\n**Purpose**: Measure this reviewer's own false-negative rate (FNR), false-positive rate (FPR), and balanced accuracy against a user-supplied gold-standard set, then attach the resulting error profile as a confidence disclosure to subsequent reviews in the same session.\n\n---\n\n## Why this mode exists\n\nA single LLM reviewer produces an absolute 0-100 rubric score, but that score is weakly interpretable without knowing the reviewer's error profile. Two reviewers could give the same paper a 65, yet one might systematically over-score weak methodology papers and the other might systematically under-score cross-disciplinary work. Absolute scores don't reveal this.\n\nLu et al. (2026, Nature 651:914-919) demonstrated in Table 1 that an LLM-based Automated Reviewer can approach human balanced accuracy (0.65 vs human 0.67-0.73 on 500 ICLR 2022 papers) while having a dramatically different error profile: FNR 0.17 vs human 0.52, at the cost of FPR 0.50 vs human 0.17-0.34. Human reviewers miss half of the papers that should be rejected; the Automated Reviewer misses very few but over-rejects more.\n\nTranslation for ARS: **our reviewer has an error profile too, and we do not currently measure it.** Calibration mode closes that gap. It does not try to make the reviewer perfect; it makes the reviewer's imperfections legible.\n\n---\n\n## Inputs\n\n1. **Gold-standard set**: 5-20 papers the user has labelled with known outcomes. Minimum 5; recommended 10-15. Each entry:\n   - Paper file path or text\n   - Ground-truth label: `accept`, `reject`, or `borderline`\n   - Venue context (journal/conference, tier)\n   - Optional: human reviewer scores for comparison\n\n2. **Domain specification**: the user's target field, used to seed `field_analyst_agent`. Calibration for \"machine learning venues\" is not valid for \"qualitative education research\" — error profiles are domain-specific.\n\n3. **Session persistence**: the error profile is cached for the **current session only**. No cross-session caching, no `~/.ars_calibration_cache/` directory. Calibration is explicitly opt-in per the v3.2 design decision: the user decides when to spend tokens on calibration, and a new session starts fresh. If the user wants to reuse a profile across sessions, they re-run calibration or paste a prior Calibration Report as a session prompt.\n\n---\n\n## Process\n\n### Phase 0: Intake\n\n- Verify the set has at least one `accept` and one `reject` (otherwise FNR or FPR is undefined).\n- If all labels are on one side, refuse to proceed and ask the user for at least one counter-example.\n- Warn if n < 10: \"Calibration with fewer than 10 papers produces wide confidence intervals. Results should be treated as directional, not conclusive.\"\n\n### Phase 1: Run `full` mode on each gold paper, with ensembling\n\nFor each paper, run the standard `full` review pipeline **5 times** (ensembling, per Lu 2026 Methods A.1.1). Each run uses a fresh context window to avoid within-session bias. Aggregate:\n- Median rubric score per dimension\n- Variance across the 5 runs (reported as a stability indicator)\n- Editorial decision (majority vote across 5)\n\n**Cross-model verification**: In calibration mode, `ARS_CROSS_MODEL` is **default-on** rather than opt-in. At least one of the 5 runs should use a different model family if available, to avoid single-model blind spots. If no cross-model is configured, emit a warning and run all 5 on the primary model.\n\n### Phase 2: Build the confusion matrix\n\nCompare reviewer's majority-vote decision against the user's ground-truth label.\n\n- `borderline` ground truth papers are excluded from the binary confusion matrix but reported separately (see Phase 3).\n- Map `Accept` and `Minor Revision` reviewer decisions → positive. Map `Major Revision` and `Reject` → negative. This follows Lu 2026 Table 1's binarization.\n\nCompute:\n\n| Metric | Formula | Report with |\n|---|---|---|\n| Balanced accuracy | (TPR + TNR) / 2 | 95% CI via bootstrap (1000 resamples) |\n| FNR (miss rate) | FN / (FN + TP) | Same |\n| FPR (false alarm) | FP / (FP + TN) | Same |\n| AUC | ROC over rubric-score threshold | Same |\n| Calibration error | Mean &#124;rubric_score - ground_truth_severity&#124; | Per-dimension |\n\n### Phase 3: Borderline handling\n\nBorderline papers don't enter the binary matrix but are useful for rubric-score calibration. For each borderline paper, report:\n- The reviewer's rubric score\n- The reviewer's decision\n- Whether the reviewer's decision respects the user's \"this is borderline\" signal (i.e., did it correctly land in Major Revision rather than confidently Accept or Reject?)\n\nA reviewer that confidently Accepts or Rejects borderline papers has a \"confidence miscalibration\" problem even if its binary accuracy looks fine.\n\n### Phase 4: Produce the Calibration Report\n\nOutput docu\n\nArchive v1.0.1: 39 files, 156627 bytes\n\nFiles: agents/devils_advocate_reviewer_agent.md (15011b), agents/domain_reviewer_agent.md (11729b), agents/editorial_synthesizer_agent.md (13139b), agents/eic_agent.md (8438b), agents/field_analyst_agent.md (9429b), agents/methodology_reviewer_agent.md (13196b), agents/perspective_reviewer_agent.md (14358b), LICENSE (19584b), references/calibration_mode_protocol.md (10542b), references/changelog.md (976b), references/editorial_decision_standards.md (9230b), references/guided_mode_protocol.md (1971b), references/integration_guide.md (680b), references/quality_rubrics.md (7679b), references/re_review_mode_protocol.md (4250b), references/review_criteria_framework.md (10139b), references/review_quality_thinking.md (3015b), references/sprint_contract_protocol.md (11621b), references/statistical_reporting_standards.md (22703b), references/top_journals_by_field.md (12874b), shared/artifact_reproducibility_pattern.md (8921b), shared/benchmark_report_pattern.md (8988b), shared/benchmark_report.schema.json (2789b), shared/collaboration_depth_rubric.md (10810b), shared/compliance_checkpoint_protocol.md (7175b), shared/compliance_report.schema.json (8314b), shared/cross_model_verification.md (11173b), shared/ground_truth_isolation_pattern.md (12649b), shared/handoff_schemas.md (46013b), shared/mode_spectrum.md (4047b), shared/prisma_trAIce_protocol.md (10415b), shared/raise_framework.md (7365b), shared/sprint_contract.schema.json (18954b), shared/style_calibration_protocol.md (7085b), SKILL.md (4257b), templates/editorial_decision_template.md (6446b), templates/peer_review_report_template.md (7479b), templates/revision_response_template.md (7020b), _meta.json (142b)\n\nArchive v1.0.0: 38 files, 149905 bytes\n\nFiles: agents/devils_advocate_reviewer_agent.md (15011b), agents/domain_reviewer_agent.md (11729b), agents/editorial_synthesizer_agent.md (13139b), agents/eic_agent.md (8438b), agents/field_analyst_agent.md (9429b), agents/methodology_reviewer_agent.md (13196b), agents/perspective_reviewer_agent.md (14358b), references/calibration_mode_protocol.md (10542b), references/changelog.md (976b), references/editorial_decision_standards.md (9230b), references/guided_mode_protocol.md (1971b), references/integration_guide.md (680b), references/quality_rubrics.md (7679b), references/re_review_mode_protocol.md (4250b), references/review_criteria_framework.md (10139b), references/review_quality_thinking.md (3015b), references/sprint_contract_protocol.md (11621b), references/statistical_reporting_standards.md (22703b), references/top_journals_by_field.md (12874b), shared/artifact_reproducibility_pattern.md (8921b), shared/benchmark_report_pattern.md (8988b), shared/benchmark_report.schema.json (2789b), shared/collaboration_depth_rubric.md (10810b), shared/compliance_checkpoint_protocol.md (7175b), shared/compliance_report.schema.json (8314b), shared/cross_model_verification.md (11173b), shared/ground_truth_isolation_pattern.md (12649b), shared/handoff_schemas.md (46013b), shared/mode_spectrum.md (4047b), shared/prisma_trAIce_protocol.md (10415b), shared/raise_framework.md (7365b), shared/sprint_contract.schema.json (18954b), shared/style_calibration_protocol.md (7085b), SKILL.md (3524b), templates/editorial_decision_template.md (6446b), templates/peer_review_report_template.md (7479b), templates/revision_response_template.md (7020b), _meta.json (142b)","readmeExcerpt":"Skill: Academic Paper Reviewer Owner: andyrenxu7255 Summary: 7-agent paper review system on Hermes Agent. 6 modes (full/re-review/quick/methodology-focus/guided/calibration). 5-panel review with editorial decision, rev... Tags: academic:1.0.4, cc-by-nc:1.0.2, editorial:1.0.4, hermes:1.0.4, latest:1.0.4, manuscript:1.0.4, multi-agent:1.0.4, paper-review:1.0.4, peer-review:1.0.4, research:1.0.4, review:1.0.4 Version hi","codeSnippets":[],"executableExamples":[{"language":"text","snippet":"Review this paper for journal submission"},{"language":"text","snippet":"delegate_task(tasks=[\n    {\"goal\": \"Review manuscript methodology: design appropriateness, validity threats, replicability. Score 1-5.\", \"context\": \"Use agents/methodology_reviewer.md\", \"toolsets\": [\"file\"]},\n    {\"goal\": \"Review evidence: citation quality, source credibility, evidence hierarchy alignment. Score 1-5.\", \"context\": \"Use agents/evidence_reviewer.md\", \"toolsets\": [\"file\"]},\n    {\"goal\": \"Review argument: logical flow, claim-evidence alignment, counter-argument handling. Score 1-5.\", \"context\": \"Use agents/argument_reviewer.md\", \"toolsets\": [\"file\"]},\n    {\"goal\": \"Review domain positioning: literature coverage, theoretical grounding, contribution significance. Score 1-5.\", \"context\": \"Use agents/domain_reviewer.md\", \"toolsets\": [\"file\"]}\n])"},{"language":"text","snippet":"delegate_task(goal=\"Aggregate all 4 reviewer reports. Apply weighted scoring (Method 30%, Evidence 25%, Argument 25%, Domain 20%). Issue editorial decision: Accept/Minor Revision/Major Revision/Reject with justification.\", context=\"Use agents/editor_in_chief.md\", toolsets=[\"file\"])"},{"language":"text","snippet":"delegate_task(goal=\"Convert editorial decision + reviewer reports into structured Revision Roadmap: prioritized action items, estimated effort, dependency mapping.\", context=\"Use agents/revision_coach.md\", toolsets=[\"file\"])"},{"language":"text","snippet":"# Calibration Report for <Reviewer Instance>\nDomain: <domain>\nGold set: n=<N> (accept=<a>, reject=<r>, borderline=<b>)\nRuns per paper: 5 (ensembled)\nCross-model: <yes/no, model families used>\n\n## Summary metrics\n- Balanced accuracy: 0.XX [95% CI: 0.XX - 0.XX]\n- FNR: 0.XX [95% CI ...]\n- FPR: 0.XX [95% CI ...]\n- AUC: 0.XX\n- Ensemble stability: <mean std of rubric scores across runs>\n\n## Comparison to Lu 2026 Table 1 baselines\n| Metric | This reviewer | Lu 2026 Automated Reviewer | Lu 2026 Human |\n|---|---|---|---|\n| Balanced accuracy | X | 0.65 | 0.67-0.73 |\n| FNR | X | 0.17 | 0.52 |\n| FPR | X | 0.50 | 0.17-0.34 |\n\n(Note: Lu 2026 numbers are for ML venues specifically. Compare with caution outside ML.)\n\n## Per-dimension calibration error\n<table of 7 review dimensions with mean absolute calibration error>\n\n## Systematic biases detected\n<natural-language narrative identifying patterns, e.g.\n \"Reviewer tends to over-score originality on cross-disciplinary papers\"\n \"Reviewer under-scores qualitative methodology by ~8 points vs ground truth\"\n>\n\n## Recommendations for session use\n- Treat this reviewer's rubric scores as having calibration error ±X points\n- For accept/reject decisions, the reviewer misses X% of reject cases (FNR)\n- For decisions near the accept/reject boundary, escalate to human judgement"},{"language":"text","snippet":"> **Reviewer Confidence Disclosure (from calibration session <id>):**\n> This reviewer has measured balanced accuracy 0.XX, FNR 0.XX, FPR 0.XX on a\n> gold set of <N> papers in <domain>. Rubric scores below have calibration\n> error ±X points. Treat borderline decisions with human judgement."}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: academic-paper-reviewer\ndescription: \"7-agent paper review system on Hermes Agent. 6 modes (full/re-review/quick/methodology-focus/guided/calibration). 5-panel review with editorial decision, revision roadmap, and calibration metrics. Uses delegate_task for each reviewer. Triggers: review paper, peer review, manuscript review, check revisions, calibrate reviewer, 審稿, 同儕審查, 論文審查.\"\nmetadata:\n  version: \"1.0-hermes-1.0\"\n  last_updated: \"2026-05-16\"\n  status: active\n  adapted_from: \"imbad0202/academic-research-skills\"\n  adapted_for: \"Hermes Agent (deepseek-v4-pro)\"\n  task_type: open-ended\n  license: \"CC BY-NC 4.0\"\n  original_author: \"Cheng-I Wu\"\n  original_license: \"CC BY-NC 4.0\"\n  original_repo: \"https://github.com/Imbad0202/academic-research-skills\"\n  copyright: \"Copyright (c) 2026 Cheng-I Wu\"\n---\n# Academic Paper Reviewer — 7-Agent Review System (Hermes Edition)\n\n📄 **License:** [CC BY-NC 4.0](https://creativecommons.org/licenses/by-nc/4.0/) · Copyright (c) 2026 Cheng-I Wu  \n🔗 **Original:** [Imbad0202/academic-research-skills](https://github.com/Imbad0202/academic-research-skills)  \n🔄 **Adaptation:** Multi-agent review system implemented via `delegate_task` instead of Claude Code's internal agent system. All agent definitions, references, and quality standards preserved unchanged from original. **This adaptation is distributed under the same CC BY-NC 4.0 license.**\n\n## Quick Start\n\n```\nReview this paper for journal submission\n```\n\n## Agent Team\n\n| # | Agent | Role |\n|---|-------|------|\n| 1 | intake_agent | Receive paper, determine review type |\n| 2 | methodology_reviewer | Method rigor assessment |\n| 3 | evidence_reviewer | Evidence sufficiency & citation quality |\n| 4 | argument_reviewer | Logical coherence & argument structure |\n| 5 | domain_reviewer | Domain expertise & literature positioning |\n| 6 | editor_in_chief | Aggregate reviews → editorial decision |\n| 7 | revision_coach | Convert reviews → actionable roadmap |\n\n## Hermes Execution\n\n### Full Mode: 5-Panel Parallel Review\n```\ndelegate_task(tasks=[\n    {\"goal\": \"Review manuscript methodology: design appropriateness, validity threats, replicability. Score 1-5.\", \"context\": \"Use agents/methodology_reviewer.md\", \"toolsets\": [\"file\"]},\n    {\"goal\": \"Review evidence: citation quality, source credibility, evidence hierarchy alignment. Score 1-5.\", \"context\": \"Use agents/evidence_reviewer.md\", \"toolsets\": [\"file\"]},\n    {\"goal\": \"Review argument: logical flow, claim-evidence alignment, counter-argument handling. Score 1-5.\", \"context\": \"Use agents/argument_reviewer.md\", \"toolsets\": [\"file\"]},\n    {\"goal\": \"Review domain positioning: literature coverage, theoretical grounding, contribution significance. Score 1-5.\", \"context\": \"Use agents/domain_reviewer.md\", \"toolsets\": [\"file\"]}\n])\n```\n\n### Editorial Decision\n```\ndelegate_task(goal=\"Aggregate all 4 reviewer reports. Apply weighted scoring (Method 30%, Evidence 25%, Argument 25%, Domain 20%). Issue editorial decision: Accept/Minor R"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn725tchg3gp72a78w22qa07p584f267\",\n  \"slug\": \"academic-paper-reviewer\",\n  \"version\": \"1.0.4\",\n  \"publishedAt\": 1779500956225\n}"},{"path":"references/calibration_mode_protocol.md","content":"# Calibration Mode Protocol\n\n**Status**: v3.2\n**Parent skill**: `academic-paper-reviewer`\n**Mode name**: `calibration`\n**Purpose**: Measure this reviewer's own false-negative rate (FNR), false-positive rate (FPR), and balanced accuracy against a user-supplied gold-standard set, then attach the resulting error profile as a confidence disclosure to subsequent reviews in the same session.\n\n---\n\n## Why this mode exists\n\nA single LLM reviewer produces an absolute 0-100 rubric score, but that score is weakly interpretable without knowing the reviewer's error profile. Two reviewers could give the same paper a 65, yet one might systematically over-score weak methodology papers and the other might systematically under-score cross-disciplinary work. Absolute scores don't reveal this.\n\nLu et al. (2026, Nature 651:914-919) demonstrated in Table 1 that an LLM-based Automated Reviewer can approach human balanced accuracy (0.65 vs human 0.67-0.73 on 500 ICLR 2022 papers) while having a dramatically different error profile: FNR 0.17 vs human 0.52, at the cost of FPR 0.50 vs human 0.17-0.34. Human reviewers miss half of the papers that should be rejected; the Automated Reviewer misses very few but over-rejects more.\n\nTranslation for ARS: **our reviewer has an error profile too, and we do not currently measure it.** Calibration mode closes that gap. It does not try to make the reviewer perfect; it makes the reviewer's imperfections legible.\n\n---\n\n## Inputs\n\n1. **Gold-standard set**: 5-20 papers the user has labelled with known outcomes. Minimum 5; recommended 10-15. Each entry:\n   - Paper file path or text\n   - Ground-truth label: `accept`, `reject`, or `borderline`\n   - Venue context (journal/conference, tier)\n   - Optional: human reviewer scores for comparison\n\n2. **Domain specification**: the user's target field, used to seed `field_analyst_agent`. Calibration for \"machine learning venues\" is not valid for \"qualitative education research\" — error profiles are domain-specific.\n\n3. **Session persistence**: the error profile is cached for the **current session only**. No cross-session caching, no `~/.ars_calibration_cache/` directory. Calibration is explicitly opt-in per the v3.2 design decision: the user decides when to spend tokens on calibration, and a new session starts fresh. If the user wants to reuse a profile across sessions, they re-run calibration or paste a prior Calibration Report as a session prompt.\n\n---\n\n## Process\n\n### Phase 0: Intake\n\n- Verify the set has at least one `accept` and one `reject` (otherwise FNR or FPR is undefined).\n- If all labels are on one side, refuse to proceed and ask the user for at least one counter-example.\n- Warn if n < 10: \"Calibration with fewer than 10 papers produces wide confidence intervals. Results should be treated as directional, not conclusive.\"\n\n### Phase 1: Run `full` mode on each gold paper, with ensembling\n\nFor each paper, run the standard `full` review pipeline **5 times** (ensembling, per Lu 2026 Methods A.1"},{"path":"references/changelog.md","content":"# Changelog\n\n| Version | Date | Changes |\n|---------|------|---------|\n| 1.4 | 2026-03-08 | Quality rubrics reference (0-100 scoring with 5 descriptors per dimension, weighted aggregation formula, decision mapping); Quick Mode Selection Guide; Dimension Scores upgraded from optional 1-5 to required 0-100 with rubric descriptors |\n| 1.3 | 2026-03-05 | DA vs R3 role boundaries with explicit responsibility tables; CRITICAL finding criteria with concrete examples; Consensus classification (CONSENSUS-4/3/SPLIT/DA-CRITICAL); Confidence Score weighting rules; Asian & Regional Journals reference (TSSCI + Asia-Pacific + OA options) |\n| 1.2 | 2026-03 | Added statistical reporting standards reference; enhanced methodology_reviewer_agent with statistical reporting adequacy sub-step |\n| 1.1 | 2026-02 | Added Devil's Advocate Reviewer (7th agent), added re-review mode, expanded review team from 4 to 5 |\n| 1.0 | 2026-02 | Initial version: 6 agents, 4 modes, 3-phase workflow |"},{"path":"references/editorial_decision_standards.md","content":"# Editorial Decision Standards — Criteria for Editorial Decision Making\n\nThis document defines the explicit criteria for Accept / Minor Revision / Major Revision / Reject decisions, for use by `eic_agent` and `editorial_synthesizer_agent`.\n\n---\n\n## 1. Decision Categories\n\n### Accept\n\n**Definition**: The paper can be published without further review.\n\n**Criteria**:\n- Average score across all universal dimensions >= 4.0\n- No dimension scores below 3.0\n- At least 3/4 reviewers recommend Accept or Minor Revision\n- No unresolved major academic issues\n\n**Conditions**:\n- May include minor copyediting suggestions\n- May require final formatting adjustments\n- Does not need to be sent for review again\n\n**Typical scenarios**:\n- Paper has undergone multiple revision rounds, all issues resolved\n- Rare first-pass acceptance (< 5% of submissions at top-tier journals)\n\n---\n\n### Minor Revision\n\n**Definition**: The paper is fundamentally acceptable and can be published after limited modifications; typically does not need to be sent for review again after revision.\n\n**Criteria**:\n- Average score across all universal dimensions >= 3.5\n- No dimension scores below 2.5\n- At least 3/4 reviewers recommend Accept or Minor Revision\n- Issues can be resolved within 2-4 weeks\n- Modifications do not involve restructuring core arguments or methods\n\n**Typical revision items**:\n- Supplementing a small number of references\n- Clarifying certain methodology description details\n- Improving clarity of argumentation\n- Correcting citation format\n- Adding discussion of limitations\n- Adjusting conclusion wording (avoiding overclaiming)\n\n**Response requirements**:\n- Authors must respond to reviewer comments item by item\n- After revision, reviewed by EIC (usually not sent for external review again)\n- Revision deadline: 2-4 weeks\n\n---\n\n### Major Revision\n\n**Definition**: The paper has potential but has significant issues, requiring substantial revision followed by re-review.\n\n**Criteria**:\n- Universal dimension average score between 2.5-3.4\n- Some dimensions may score below 2.5 (but not fatal)\n- At least 2/4 reviewers recommend Major Revision or better\n- Issues are serious but fixable (not fundamental design flaws)\n- Revision requires 6-8 weeks of work\n\n**Typical revision items**:\n- Re-analyzing data (additional analysis or correcting errors)\n- Substantially rewriting literature review (missing key references)\n- Supplementing additional data collection\n- Reorganizing paper structure\n- Correcting significant methodological flaws\n- Strengthening theoretical framework application\n- Adding robustness checks\n\n**Response requirements**:\n- Authors must write a detailed point-by-point response letter\n- After revision, sent for re-review (may go back to original reviewers or new reviewers)\n- Revision deadline: 6-8 weeks\n- Typically a maximum of 2 rounds of Major Revision allowed\n\n---\n\n### Reject\n\n**Definition**: The paper is not suitable for publication in this journal, even with revision.\n\n**Criteria"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"7-agent paper review system on Hermes Agent. 6 modes (full/re-review/quick/methodology-focus/guided/calibration). 5-panel review with editorial decision, rev... Skill: Academic Paper Reviewer Owner: andyrenxu7255 Summary: 7-agent paper review system on Hermes Agent. 6 modes (full/re-review/quick/methodology-focus/guided/calibration). 5-panel review with editorial decision, rev... Tags: academic:1.0.4, cc-by-nc:1.0.2, editorial:1.0.4, hermes:1.0.4, latest:1.0.4, manuscript:1.0.4, multi-agent:1.0.4, paper-review:1.0.4, peer-review:1.0.4, research:1.0.4, review:1.0.4 Version hi","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1756,"uniquenessScore":49,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-09T19:28:59.608Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-09T19:28:59.608Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T08:23:15.215Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}