{"id":"fb53c1c7-9bea-4c78-9b01-37015846ab86","entityType":"agent","slug":"clawhub-agentsope-skillalchemy","name":"Skill Alchemy Main","canonicalUrl":"https://www.xpersona.co/agent/clawhub-agentsope-skillalchemy","canonicalPath":"/agent/clawhub-agentsope-skillalchemy","generatedAt":"2026-10-11T15:14:21.633Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-11T11:59:00.645Z","emptyReason":null},"description":"SkillAlchemy — 一念落地，万象成形。输入任意想法或蒸馏目标，输出可安装的 SKILL.md。 内部编排 Lens（看清问题）和 LEAP（执行蒸馏/融合）。用户唯一入口。 Use when 用户说「蒸馏」「生成 skill」「融合」「我想做 X 但不知道从哪下手」。 Skill: Skill Alchemy Main Owner: agentsope Summary: SkillAlchemy — 一念落地，万象成形。输入任意想法或蒸馏目标，输出可安装的 SKILL.md。 内部编排 Lens（看清问题）和 LEAP（执行蒸馏/融合）。用户唯一入口。 Use when 用户说「蒸馏」「生成 skill」「融合」「我想做 X 但不知道从哪下手」。 Tags: latest:0.1.3 Version history: v0.1.3 | 2026-06-15T11:41:28.437Z | user Fixed model names in benchmark table. v0.1.2 | 2026-06-12T14:15:37.330Z | user Updated README with SkillsBench benchmark results. v0.1.1 | 2026-06-02T","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.1K downloads reported by the source. Last updated 10/11/2026.","installCommand":"clawhub skill install s172fz0df5jxmx2a90ds9q6zfd877b2a:skillalchemy","sourceUrl":"https://clawhub.ai/agentsope/skillalchemy","homepage":"https://clawhub.ai/agentsope/skills/skillalchemy","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/agentsope/skillalchemy","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/agentsope/skills/skillalchemy","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":61,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"SkillAlchemy — 一念落地，万象成形。输入任意想法或蒸馏目标，输出可安装的 SKILL.md。 内部编排 Lens（看清问题）和 LEAP（执行蒸馏/融合）。用户唯一入口。 Use when 用户说「蒸馏」「生成 skill」「融合」「我想做 X 但不知道从哪下手」。 Skill: Skill Alchem"},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-11T11:59:00.645Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T11:59:00.645Z","emptyReason":null},"stars":null,"forks":null,"downloads":1070,"packageName":null,"latestVersion":"0.1.3","tractionLabel":"1.1K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-11T11:59:00.577Z","emptyReason":null},"lastUpdatedAt":"2026-10-11T11:59:00.645Z","lastCrawledAt":"2026-10-11T11:59:00.577Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-12T11:59:00.577Z","lastVerifiedAt":null,"highlights":[{"version":"0.1.3","createdAt":"2026-06-15T11:41:28.437Z","changelog":"Fixed model names in benchmark table.","fileCount":258,"zipByteSize":1143578},{"version":"0.1.2","createdAt":"2026-06-12T14:15:37.330Z","changelog":"Updated README with SkillsBench benchmark results.","fileCount":258,"zipByteSize":1143690},{"version":"0.1.1","createdAt":"2026-06-02T08:11:55.859Z","changelog":"SkillAlchemy v0.1.1 Added a large collection of self-distilled, ready-to-use Skills under the /skills directory. Updated and introduced initial references, usage examples, and documentation for new Agents and domains.","fileCount":258,"zipByteSize":1143479},{"version":"0.1.0","createdAt":"2026-05-22T09:53:41.044Z","changelog":"SkillAlchemy v0.1.0 - Initial release of SkillAlchemy: a coordinator skill for distillation and fusion tasks. - Orchestrates two sub-skills: Lens (problem analysis) and LEAP (execution). - Structured multi-phase workflow with depth selection, interactive checkpoints, and \"run with defaults\" mode. - Provides dependency checks and guides users to install required sub-skills (Lens and LEAP). - All user interaction and routing handled by SkillAlchemy; sub-skills do not interact with users directly. - Ensures outputs are organized and intermediate files are cleaned up post-execution.","fileCount":24,"zipByteSize":65005}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s172fz0df5jxmx2a90ds9q6zfd877b2a:skillalchemy","setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-agentsope-skillalchemy/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-agentsope-skillalchemy/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-agentsope-skillalchemy/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-agentsope-skillalchemy/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-agentsope-skillalchemy/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-agentsope-skillalchemy/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-11T15:14:21.627Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-agentsope-skillalchemy/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-agentsope-skillalchemy/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-agentsope-skillalchemy/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-agentsope-skillalchemy/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-11T11:59:00.645Z","emptyReason":null},"readme":"Skill: Skill Alchemy Main\n\nOwner: agentsope\n\nSummary: SkillAlchemy — 一念落地，万象成形。输入任意想法或蒸馏目标，输出可安装的 SKILL.md。 内部编排 Lens（看清问题）和 LEAP（执行蒸馏/融合）。用户唯一入口。 Use when 用户说「蒸馏」「生成 skill」「融合」「我想做 X 但不知道从哪下手」。\n\nTags: latest:0.1.3\n\nVersion history:\n\nv0.1.3 | 2026-06-15T11:41:28.437Z | user\n\nFixed model names in benchmark table.\n\nv0.1.2 | 2026-06-12T14:15:37.330Z | user\n\nUpdated README with SkillsBench benchmark results.\n\nv0.1.1 | 2026-06-02T08:11:55.859Z | user\n\nSkillAlchemy v0.1.1\n\nAdded a large collection of self-distilled, ready-to-use Skills under the /skills directory.\nUpdated and introduced initial references, usage examples, and documentation for new Agents and domains.\n\nv0.1.0 | 2026-05-22T09:53:41.044Z | auto\n\nSkillAlchemy v0.1.0\n\n- Initial release of SkillAlchemy: a coordinator skill for distillation and fusion tasks.\n- Orchestrates two sub-skills: Lens (problem analysis) and LEAP (execution).\n- Structured multi-phase workflow with depth selection, interactive checkpoints, and \"run with defaults\" mode.\n- Provides dependency checks and guides users to install required sub-skills (Lens and LEAP).\n- All user interaction and routing handled by SkillAlchemy; sub-skills do not interact with users directly.\n- Ensures outputs are organized and intermediate files are cleaned up post-execution.\n\nArchive index:\n\nArchive v0.1.3: 258 files, 1143578 bytes\n\nFiles: CHANGELOG.md (913b), CONTRIBUTING.md (1060b), LICENSE (1069b), package.json (366b), README_EN.md (5703b), README.md (5867b), skill-card.md (2074b), skill.json (159b), SKILL.md (6324b), skills/agentsop-agent-topology-selection/intermediate/operation_candidates.json (7976b), skills/agentsop-agent-topology-selection/README.md (4024b), skills/agentsop-agent-topology-selection/references/R1-source-evidence.md (7130b), skills/agentsop-agent-topology-selection/SKILL.md (19118b), skills/agentsop-aider/intermediate/operation_candidates.json (10797b), skills/agentsop-aider/README.md (2494b), skills/agentsop-aider/references/R1-architecture.md (7186b), skills/agentsop-aider/references/R2-sop-workflow.md (7428b), skills/agentsop-aider/references/R3-dilemma-cases.md (7648b), skills/agentsop-aider/references/R4-anti-patterns.md (6236b), skills/agentsop-aider/references/R5-ecosystem-context.md (6272b), skills/agentsop-aider/SKILL.md (20180b), skills/agentsop-bio-fraud-forensics/examples/demo_screening.md (4291b), skills/agentsop-bio-fraud-forensics/README.md (2955b), skills/agentsop-bio-fraud-forensics/references/R01-misconduct-taxonomy.md (20390b), skills/agentsop-bio-fraud-forensics/references/R02-image-forensics.md (19498b), skills/agentsop-bio-fraud-forensics/references/R03-statistical-forensics.md (19779b), skills/agentsop-bio-fraud-forensics/references/R04-exposure-sites-method.md (21526b), skills/agentsop-bio-fraud-forensics/references/R05-evidence-red-lines.md (21908b), skills/agentsop-bio-fraud-forensics/references/R06-screening-workflow.md (22217b), skills/agentsop-bio-fraud-forensics/references/R07-paper-mill-signals.md (19986b), skills/agentsop-bio-fraud-forensics/references/research_notes.md (8099b), skills/agentsop-bio-fraud-forensics/references/sop_models.md (17858b), skills/agentsop-bio-fraud-forensics/SKILL.md (12777b), skills/agentsop-bio-fraud-forensics/USAGE.md (8508b), skills/agentsop-bounded-loop/intermediate/operation_candidates.json (6838b), skills/agentsop-bounded-loop/references/R1-source-evidence.md (7055b), skills/agentsop-bounded-loop/references/R2-cross-framework.md (5322b), skills/agentsop-bounded-loop/SKILL.md (28881b), skills/agentsop-code-execution-decision/intermediate/operation_candidates.json (4342b), skills/agentsop-code-execution-decision/README.md (1988b), skills/agentsop-code-execution-decision/references/R1-source-evidence.md (4670b), skills/agentsop-code-execution-decision/SKILL.md (26082b), skills/agentsop-context-scope-discipline/intermediate/operation_candidates.json (5468b), skills/agentsop-context-scope-discipline/README.md (3178b), skills/agentsop-context-scope-discipline/references/R1-source-evidence.md (5047b), skills/agentsop-context-scope-discipline/SKILL.md (20665b), skills/agentsop-conventions-pinning/intermediate/operation_candidates.json (9575b), skills/agentsop-conventions-pinning/README.md (2338b), skills/agentsop-conventions-pinning/references/R1-source-evidence.md (10849b), skills/agentsop-conventions-pinning/references/R2-tool-equivalents.md (9203b), skills/agentsop-conventions-pinning/SKILL.md (36286b), skills/agentsop-cost-tiered-models/intermediate/operation_candidates.json (7378b), skills/agentsop-cost-tiered-models/README.md (4331b), skills/agentsop-cost-tiered-models/references/R1-source-evidence.md (6583b), skills/agentsop-cost-tiered-models/SKILL.md (20887b), skills/agentsop-crewai/intermediate/operation_candidates.json (6596b), skills/agentsop-crewai/README.md (1862b), skills/agentsop-crewai/references/R1-architecture.md (4909b), skills/agentsop-crewai/references/R2-sop-workflow.md (5383b), skills/agentsop-crewai/references/R3-dilemma-cases.md (7126b), skills/agentsop-crewai/references/R4-anti-patterns.md (4921b), skills/agentsop-crewai/references/R5-ecosystem-context.md (5549b), skills/agentsop-crewai/SKILL.md (24740b), skills/agentsop-dify/intermediate/operation_candidates.json (9704b), skills/agentsop-dify/README.md (1770b), skills/agentsop-dify/references/R1-architecture.md (7760b), skills/agentsop-dify/references/R2-sop-workflow.md (7097b), skills/agentsop-dify/references/R3-dilemma-cases.md (10109b), skills/agentsop-dify/references/R4-anti-patterns.md (6465b), skills/agentsop-dify/references/R5-ecosystem-context.md (7700b), skills/agentsop-dify/SKILL.md (28387b), skills/agentsop-domain-eval-set/intermediate/operation_candidates.json (6046b), skills/agentsop-domain-eval-set/README.md (2808b), skills/agentsop-domain-eval-set/references/R1-source-evidence.md (4986b), skills/agentsop-domain-eval-set/SKILL.md (25428b), skills/agentsop-dspy/intermediate/operation_candidates.json (12570b), skills/agentsop-dspy/README.md (2415b), skills/agentsop-dspy/references/R1-architecture.md (6013b), skills/agentsop-dspy/references/R2-sop-workflow.md (5669b), skills/agentsop-dspy/references/R3-dilemma-cases.md (7629b)\n\nFile v0.1.3:SKILL.md\n\n---\nname: SkillAlchemy\ndescription: |\n  SkillAlchemy — 一念落地，万象成形。输入任意想法或蒸馏目标，输出可安装的 SKILL.md。\n  内部编排 Lens（看清问题）和 LEAP（执行蒸馏/融合）。用户唯一入口。\n  Use when 用户说「蒸馏」「生成 skill」「融合」「我想做 X 但不知道从哪下手」。\nversion: v1.0\n---\n\n# Skill-Alchemy · 一念落地，万象成形\n\n你是 SkillAlchemy。编排两个子 skill：Lens 看清，LEAP 落地。\n你自己不蒸馏、不融合——只做编排。**所有用户交互由你负责，LEAP 不跟用户说话。**\n\n## 前置检查\n\n```\nls ~/.claude/skills/Lens/SKILL.md\nls ~/.claude/skills/LEAP/SKILL.md\n```\n\n**如果缺少任何一个，告诉用户：**\n\n> SkillAlchemy 需要两个依赖才能运行，请先安装：\n>\n> ```\n> npx skills add agentsope/SkillAlchemy/skills/Lens\n> npx skills add agentsope/SkillAlchemy/skills/LEAP\n> ```\n>\n> 或者去 https://skills.sh 搜索 Lens 和 LEAP 安装。\n>\n> 装好之后回来找我继续。\n\n---\n\n## 编排流程\n\n### Phase 0: 确认深度 + 任务简报\n\n先确认 depth。用户没说就问一句：\n\n```\nquick    — 快速原型，3 agent，~5-8 min，跳过验证\nstandard — 日常使用（默认），4-5 agent，~15-20 min\ndeep     — 发布级，6-8 agent，~25-35 min，强制验证 + 双审核\n没说的话默认 standard。\n```\n\n用户给了深度后，**展示任务简报：**\n\n```\n◆ 任务简报\n\n▸ 需求    蒸馏「张雪峰」→ persona skill\n▸ 流程    Lens → A 分支（7 Stage + 2 Gate）\n          ├─ Research Swarm  4-5 agent 并行研究\n          ├─ Exemplar        find-skills 在线检索 + 自动评分\n          └─ Compile         编译 + 自评 + 验证 + 清理\n▸ 深度    standard · ~15-20 min\n▸ 交互    步步确认（2 次暂停）\n\n> 确认，按 standard 跑\n> 换成 deep，研究更深入、验证更严格、双 agent 交叉审核\n> 一路默认跑完，中间别问我了，全部默认值到底\n> 先只要 Lens 看看维度，不生成 skill\n```\n\n根据实际任务替换内容。确认后进 Phase 1。如果用户一开始就指定了 depth，跳过询问直接出简报。\n\n**「一路默认」模式：** 用户在任何节点说「一路默认」→ 跳过当前及后续所有交互，全部 standard 默认值跑完。\n\n---\n\n### Phase 1: Lens 分析\n\n调 Lens，输入用户原话。Lens 不向用户提问，直接输出增强版 description。\n\n**Lens 完成后，展示维度摘要（不放全文，太长）：**\n\n```\n◆ Lens 分析完成 · N 个维度\n\n  [维度名]    [维度名]    [维度名]\n  [维度名]    [维度名]    [维度名]\n  ...\n\n▸ 意图    distill_persona / distill_method / fuse_skills\n\n> 确认，进入 [distill / fuse] 管线继续\n> 展开看看完整的 Lens 分析原文，每个维度的细节\n> 补一个 XX 维度，重新分析一遍\n> 就停在这，我消化一下 Lens 的结果，不继续了\n```\n\n确认后进 Phase 2。提了修改意见 → 重新调 Lens 带上反馈。\n「一路默认」已激活 → 跳过，直接进 Phase 2。\n\n---\n\n### Phase 2: 路由判断\n\n| Lens 意图 | 动作 |\n|-----------|------|\n| distill | → Phase 3a（A 分支：蒸馏管线） |\n| fuse | → Phase 3b（B 分支：融合管线） |\n| decompose | 停。展示 Lens 输出，问是否继续 |\n| 无法判断 | 问用户：蒸馏还是融合？ |\n\n---\n\n### Phase 3: 执行\n\n**所有输出落在当前项目根目录的 `output/` 下。**\n调 LEAP 时用绝对路径指定输出位置（以实际项目路径为准）。\n\n#### 3a. Distill 路线（2 步，1 次确认）\n\n**Step 1: 生成 research plan。**\n```\n调 LEAP：\n  \"distill [target]，depth [depth]。\n   只到 research plan（stop_after_stage: 3），\n   输出到 <项目根目录>/output/<target>-skill/。\"\n```\n\nLEAP 跑完 Stage 1-3 后停止。读取 `research_plan.json`：\n\n```\n◆ Research Plan · N agents\n\n  R1  [维度名]\n      [搜索方向一句话]\n\n  R2  [维度名]\n      [搜索方向一句话]\n\n  ...\n\n> 确认，按这个计划启动 N 个 agent 并行研究\n> 加一个 R[n] 专门研究 XX 方向，补上缺失的维度\n> 删掉 R[n]，这个维度我不太关心，省点资源\n> 换成 quick 快速跑，3 个 agent 够了我赶时间\n```\n\n**Step 2: 研究 + exemplar + 编译（无交互，直接跑完）。**\n```\n调 LEAP：\n  \"从 Stage 4 继续 distill [target]，\n   research_plan 已确认，\n   输出到 <项目根目录>/output/<target>-skill/。\"\n```\n\nLEAP 执行 Stage 4-7 + Gate 1-2，全自动完成：\nResearch Swarm → Exemplar Discovery（find-skills + score_skill 自动评分择优）→ Synthesis → Compile → Validate。\n\n完成后清理中间产物：\n- 删除 `references/exemplar_candidates.json`（临时评分文件）\n- 删除 `references/exemplars/`（中间参照副本）\n- 删除空 `validation/`（standard 模式不跑 Phase 8）\n- 保留 `R*.md`（研究证据）、`intermediate/`（审计追踪）、产出包\n\n#### 3b. Fuse 路线\n\n```\n调 LEAP：\n  \"fuse [primary] + [secondary]，depth [depth]，\n   输出到 <项目目录>/output/。\"\n```\n\nLEAP 自动完成 Retrieve（本地 → find-skills → GitHub raw，score_skill 自动评分择优）\n→ Parse → Weave → Output → Gate。\n\n完成后清理 `references/fusion_candidates.json`（如产生）。\n\n#### 3c. 混合路线\n\n→ 先 3a 蒸馏缺失 skill → 再 3b 融合\n\n---\n\n### Phase 4: 收尾\n\n验证 + 报告：\n\n```\n◆ 蒸馏完成\n\n  skill     [名称] · [name]\n  类型      persona / tool · N 行\n  质量      ✓ pass / ✗ fail · 自评 N/10\n  研究      N agents · N+ Dilemma Cases\n  产出      output/<name>-skill/\n\n  安装      cp -r output/<name>-skill \\\n                 ~/.claude/skills/<name>/\n  试试      /[name] [建议 prompt]\n```\n\n---\n\n## 约束\n\n- SkillAlchemy 是用户唯一入口。Output 落在 `output/`。\n- 只做编排。蒸馏/融合是 LEAP 的事，路由是你的活，交互是你的活。\n- 调 LEAP 时必须指定绝对输出路径。\n- 编译完成后清理中间产物：`exemplar_candidates.json`、`fusion_candidates.json`、`exemplars/`、空目录。\n- 子 skill 失败报告给用户，不假装成功。\n- 「一路默认」：任意节点说「一路默认」→ 跳过后续所有交互，全默认跑完。\n\nFile v0.1.3:skills/agentsop-agent-topology-selection/SKILL.md\n\n---\nname: agentsop-agent-topology-selection\nversion: 0.1.0\ndescription: >-\n  Cross-framework enhancement overlay for choosing a multi-agent topology BEFORE writing any\n  agent. A binary-question rubric — is single-agent + tools enough? do agents need to know\n  about each other? does the output need one voice? — maps the answer to single-agent /\n  supervisor / swarm / sequential / hierarchical. Activates when a coder agent is tempted to\n  \"split the work into roles\" or reaches for a multi-agent framework. Encodes the *selection\n  rubric* that the per-framework skills assume but never surface. Search keywords: when to\n  use multi-agent, single vs multi agent, do I need multiple agents, supervisor vs swarm,\n  multi-agent vs single agent, agent team design.\noverlay: true\ncross_links: [crewai, langgraph, bounded-loop]\n---\n\n# Multi-Agent Topology Selection · SOP (ENHANCE overlay)\n\n> Overlay posture: this skill decides *whether and which* topology. It does not\n> teach the API — descend to `[[crewai]]` or `[[agentsop-langgraph]]` for that. Every\n> load-bearing claim carries an inline source tag resolving in\n> `references/R1-source-evidence.md`.\n\n---\n\n## 1. 何时激活 (When to Activate)\n\nActivate when **any** of the following fire:\n\n- The task description contains \"team of agents\", \"researcher + writer + reviewer\",\n  \"manager agent\", \"agents that hand off\", \"split this into roles\", or \"multi-agent\".\n- A coder agent is about to instantiate ≥2 agents (CrewAI `Agent(...)` × N,\n  LangGraph supervisor/swarm, OpenAI Swarm handoffs) and has **not yet** justified\n  why a single agent with tools is insufficient.\n- Someone is choosing between CrewAI `Process.sequential` vs `Process.hierarchical`,\n  or LangGraph supervisor vs swarm vs hierarchical-teams, and wants the *rubric*,\n  not the syntax.\n- A multi-agent system is over budget on tokens/latency and the question is \"can we\n  collapse agents back into one?\".\n\nDo **not** activate for: a single LLM call, a one-shot RAG query, or a fixed\ntool-call pipeline with no role separation. Those are the single-agent baseline\nthis skill defends.\n\n> Mental check: *\"An agent needs agency, otherwise it's just another script.\"*\n> — João Moura, CrewAI founder `[[crewai · §1.3]]`. If you can write the control\n> flow in `if/else`, you do not need multiple agents — you need one agent (or a\n> graph) with explicit edges.\n\n---\n\n## 2. 核心心智模型 (Core Mental Model)\n\n**Most \"multi-agent\" problems are single-agent + tools.** Add agents only when\n*context isolation* or *parallel expertise* genuinely demands it.\n\n> \"Single-agent is right for approximately 80% of cases; the trap is reaching for\n> multi-agent because it sounds more capable.\" `[[crewai · DC-1]]`\n\nTwo — and only two — forces justify a second agent:\n\n1. **Context isolation.** One agent's working context would pollute another's\n   (a critic that must not see its own draft's rationalisations; a tool-heavy\n   sub-task whose 40 intermediate tool calls should not bloat the main thread).\n   Splitting gives each agent a clean, bounded prompt.\n2. **Parallel expertise.** Two *genuinely different* skills run concurrently or in\n   strict sequence (research → write → review), where a single prompt provably\n   cannot hold both jobs without quality collapse `[[crewai · DC-1]]`.\n\nIf neither force is present, **a single agent with the union of tools wins** —\nfewer hops, fewer tokens, no handoff failures. This is the baseline the rubric\nmust beat, not the default to escape.\n\n### The selection rubric (three binary questions)\n\n```\nQ0  Is single-agent + tools enough?\n      (no context-isolation need, no parallel-expertise need)\n        YES → single-agent + tools. STOP. Do not add agents.\n        NO  → ↓\n\nQ1  Do the agents need to KNOW ABOUT EACH OTHER (peer handoff)?\n        NO  → one funnels through a coordinator → SUPERVISOR\n              (or static order → SEQUENTIAL, if order is fixed)\n        YES → ↓\n\nQ2  Must the OUTPUT speak with ONE VOICE / single audit funnel?\n        YES → SUPERVISOR (single user-facing persona, one funnel)\n        NO  → SWARM (dynamic peer handoff, last-active agent remembered)\n\nScaling override: ≥6 specialists that group into teams → HIERARCHICAL\n(supervisor-of-supervisors). Use only for grouping, not for routing.\n```\n\nThe two questions that actually separate the patterns: **(a) can sub-agents know\neach other, (b) is one user-facing voice mandated.** Everything else is tuning\n`[[langgraph · Case 2]]`.\n\n---\n\n## 3. SOP 工作流 (Selection Protocol)\n\nWalk top-down. Each gate can send you *back down* the ladder — collapsing agents\nis as valid an answer as adding them.\n\n### Step 1 · Defend the single-agent baseline first\nAsk Q0. Enumerate the would-be roles. For each, ask: *would merging it into one\nagent's prompt + toolset actually degrade output?* If you cannot point to a\nconcrete failure mode (style drift, missed checklist, context bloat, parallel\nlatency), the honest answer is single-agent + tools. Exit here ~80% of the time\n`[[crewai · DC-1]]`.\n\n### Step 2 · If splitting, decide static vs dynamic routing\n- **Order is fixed and known at design time** (research always precedes write\n  precedes review) → **SEQUENTIAL**. Cheapest, most debuggable, 1× token baseline\n  `[[crewai · §2.3]]`. In CrewAI this is `Process.sequential`; in LangGraph it is\n  static edges A→B→C.\n- **Routing must be decided at runtime** (which specialist handles *this* query) →\n  you need a coordinator or peer handoff → continue to Step 3.\n\n### Step 3 · Coordinator (supervisor) vs peers (swarm)\nAsk Q1 then Q2.\n- Agents that do **not** know each other and funnel through one router →\n  **SUPERVISOR**. Sub-agents are effectively tools the supervisor calls; the\n  supervisor \"translates\" their output back to the user — which is *exactly* why\n  it costs the most tokens `[[langgraph · Step 4]]`.\n- Agents that **do** know each other and **no** single voice is mandated →\n  **SWARM**. Dynamic handoff, last-active agent stays active across turns, no\n  translation step → fewer tokens, slightly higher accuracy on the τ-bench retest\n  `[[langgraph · §SOP Step 4]]`.\n\n### Step 4 · Apply the supervisor-default caveat (read this twice)\nLangChain's *own* benchmark found swarm \"slightly outperformed supervisor across\nall scenarios\" and supervisor \"consistently uses more tokens than swarm\" — yet\nthey **still ship supervisor as the recommended default** `[[langgraph · Step 4]]`.\nWhy the nuance matters:\n- Supervisor is the **safest with third-party / untrusted agents** (single funnel,\n  single audit log, single place to enforce policy) `[[langgraph · Step 4]]`.\n- Swarm is a **bad fit for third-party agents** — peers handing off to peers means\n  no central control point `[[langgraph · Step 4]]`.\n- So: **do not copy \"default = supervisor\" blindly.** If your agents are internal\n  and trusted, swarm is often the better pick the default hides. Pick on the two\n  questions, not on the framework's default.\n\n### Step 5 · Verify the chosen topology can terminate\nAny topology with runtime handoff (swarm, hierarchical, CrewAI delegation) can\nloop. Bound it before shipping — cross-link `[[agentsop-bounded-loop]]`. Concretely:\ndefault `allow_delegation=False` on workers, set per-agent `max_iter`, wrap an\nouter timeout, and bake an explicit exit counter into state rather than trusting\nthe LLM to stop `[[crewai · DC-5]]` `[[agentsop-bounded-loop]]`.\n\n### Step 6 · Reconsider before scaling agents up\n≥6 specialists → group into **HIERARCHICAL teams** purely for *navigability*, not\nto get free routing (see Anti-patterns). At >5 agents CrewAI starts hitting\ncoordination failure `[[crewai · §6.1]]`; that is a signal to group or collapse,\nnot to add more.\n\n---\n\n## 4. 操作模型 (Operation Models)\n\nFormat: **Trigger → Action → Output → Evidence**.\n\n### OP-1 · Defend the single-agent baseline\n- **Trigger**: a task is being described as \"a team of agents\".\n- **Action**: list the proposed roles; for each, name the concrete failure that\n  merging into one agent would cause (style drift / missed checklist / context\n  bloat / required parallelism). No nameable failure ⇒ single agent.\n- **Output**: single-agent + tools, OR a justified list of must-split roles.\n- **Evidence**: `[[crewai · DC-1]]` \"single-agent right for ~80% of cases\".\n\n### OP-2 · Single-vs-multi gate\n- **Trigger**: baseline defended and at least one role has a real split-justifying\n  failure mode.\n- **Action**: confirm the force is *context isolation* or *parallel expertise* —\n  not \"it sounds more capable\". If only the latter, stay single.\n- **Output**: a yes/no on multi-agent with the force named in one sentence.\n- **Evidence**: `[[crewai · §2.2]]`, `[[crewai · DC-1]]`.\n\n### OP-3 · Static-order gate (→ sequential)\n- **Trigger**: multi-agent confirmed; the order of work is fixed at design time.\n- **Action**: choose SEQUENTIAL — list agents in execution order; pass dependencies\n  explicitly (CrewAI `context=[...]`), do not rely on implicit transfer.\n- **Output**: an ordered task list; CrewAI `Process.sequential` or LangGraph static\n  edges.\n- **Evidence**: `[[crewai · §2.3]]` (sequential = 1× tokens, low debug cost).\n\n### OP-4 · Peer-awareness gate (Q1 → supervisor vs swarm branch)\n- **Trigger**: routing must be decided at runtime.\n- **Action**: ask \"do sub-agents know about each other?\" — NO ⇒ supervisor branch;\n  YES ⇒ continue to OP-5.\n- **Output**: chosen branch.\n- **Evidence**: `[[langgraph · Step 4]]` decision tree.\n\n### OP-5 · Single-voice gate (Q2 → supervisor vs swarm)\n- **Trigger**: peers know each other (Q1=YES).\n- **Action**: ask \"must output be one voice / single audit funnel?\" — YES ⇒\n  SUPERVISOR; NO ⇒ SWARM.\n- **Output**: SUPERVISOR or SWARM.\n- **Evidence**: `[[langgraph · Case 2]]` (compliance ⇒ supervisor; UX continuity ⇒ swarm).\n\n### OP-6 · Apply the supervisor-default caveat\n- **Trigger**: supervisor was selected, OR you are about to accept a framework\n  default.\n- **Action**: check trust. Third-party/untrusted agents ⇒ supervisor is correct.\n  Internal/trusted ⇒ re-test whether swarm's lower token cost wins; if so, switch.\n- **Output**: a topology chosen on trust + the two questions, not on the default.\n- **Evidence**: `[[langgraph · Step 4]]` — swarm beats supervisor on bench, yet\n  supervisor remains the shipped default *for third-party safety*.\n\n### OP-7 · Tune-before-switch\n- **Trigger**: chosen topology is over token/latency budget.\n- **Action**: before changing paradigm, apply the published fixes to the *current*\n  one. For supervisor: remove handoff messages, add a forwarding-messages tool,\n  optimise tool naming — LangChain measured \"nearly 50% increase in performance\"\n  `[[langgraph · Step 4]]`. Switch paradigm only if still over budget.\n- **Output**: a tuned topology or a justified migration.\n- **Evidence**: `[[langgraph · Case 2]]`.\n\n### OP-8 · Bound the topology\n- **Trigger**: any runtime-handoff topology before ship.\n- **Action**: `allow_delegation=False` on workers, per-agent `max_iter`, outer\n  timeout, explicit state-based exit counter.\n- **Output**: a topology that provably terminates.\n- **Evidence**: `[[crewai · DC-5]]` (delegation ping-pong), `[[agentsop-bounded-loop]]`.\n\n---\n\n## 5. 困境决策案例 (Dilemma Cases)\n\n### Case 1 · \"Swarm beats supervisor on the benchmark — why ship supervisor?\"\n- **困境**: LangChain's own multi-agent benchmark shows swarm \"slightly\n  outperformed supervisor across all scenarios\" and supervisor \"consistently uses\n  more tokens\" (the telephone-game translation overhead). Yet LangChain's\n  *recommended default* is still supervisor `[[langgraph · Step 4]]`. A coder\n  agent copying the default would leave accuracy and tokens on the table.\n- **约束**:\n  - Want the benchmark's efficiency (swarm).\n  - But may integrate third-party / untrusted agents later.\n  - Need a single auditable funnel for tool calls (compliance).\n- **决策步骤**:\n  1. Read the default's *reason*, not the default. Supervisor wins on **safety with\n     third-party agents** and **single audit funnel** — not on accuracy `[[langgraph · Step 4]]`.\n  2. If all agents are internal and trusted **and** no single-voice mandate →\n     pick **swarm**; the default does not apply to you.\n  3. If a single auditable funnel is mandated → keep supervisor, then apply the\n     three fixes (remove handoff messages, forwarding-messages tool, tool-name\n     tuning) for the measured ~50% bump *before* concluding it is too slow\n     `[[langgraph · Case 2]]`.\n  4. Re-measure tokens; migrate to swarm only if still over budget and audit is\n     tolerant.\n- **结果**: Topology chosen on trust + the two questions. The benchmark's \"swarm >\n  supervisor\" is true *and* the supervisor default is rational — for a *different*\n  constraint (third-party safety) than the one the benchmark measured (accuracy/cost).\n- **可提取的操作**: OP-6. **A framework default encodes the framework author's\n  worst-case constraint, not yours. Decode the reason; re-derive for your case.**\n\n### Case 2 · \"CrewAI hierarchical for simple routing is structurally broken\"\n- **困境**: 5 agents, order roughly fixed, but the system should *route* — skip\n  irrelevant specialists based on the query. `Process.hierarchical` looks like it\n  \"should auto-route\". In practice the manager **executes all tasks** and the last\n  task's output overwrites the rest — it does not skip on triage `[[crewai · DC-2]]`.\n  ```\n  Query: \"Why is my laptop overheating?\" (pure technical)\n  Expected:  triage → technical_agent → done\n  Hierarchical reality: triage → technical → billing → … → last output wins\n  ```\n- **约束**: routing is required; default `manager_llm` is unreliable; latency/cost\n  matter.\n- **决策步骤**:\n  1. Recognise hierarchical-for-routing is the wrong tool: CrewAI hierarchical is a\n     *coordination* topology, not a *router* `[[crewai · DC-2]]`.\n  2. If routing logic fits in ~5 lines of Python → use a **CrewAI Flow** (`@router`\n     / `@listen`) or a **LangGraph conditional edge** — explicit branch, then call\n     one small crew / single agent per branch `[[crewai · DC-2]]`.\n  3. If routing genuinely needs LLM semantic judgement → hierarchical **with a\n     custom `manager_agent`** carrying an explicit branching backstory — *never*\n     the bare default `manager_llm` `[[crewai · OP-4]]`.\n  4. Bound it: workers `allow_delegation=False`, outer timeout `[[agentsop-bounded-loop]]`.\n- **结果**: Routing handled by explicit control flow; hierarchical reserved for\n  true coordination of trusted teams, never for if/else routing.\n- **可提取的操作**: **Hierarchical ≠ router. Routing is control flow — make it\n  explicit (Flow / conditional edge), don't delegate it to a manager LLM that will\n  run everything.** Compare with `d-query-routing-skill` for the routing-specific rubric.\n\n---\n\n## 6. 反模式与边界 (Anti-patterns & Boundaries)\n\n- **Multi-agent theater.** Splitting into roles because it \"sounds more capable\"\n  with no context-isolation or parallel-expertise force. Symptom: agents that just\n  pass a string along, each adding a paragraph. Fix: collapse to one agent + tools\n  `[[crewai · DC-1]]`.\n- **Hierarchical for simple routing.** Using `Process.hierarchical` (or a\n  supervisor) to get free if/else routing. It runs everything; routing must be\n  explicit control flow (Flow / conditional edge) `[[crewai · DC-2]]`.\n- **Copying the supervisor default blindly.** The benchmark says swarm is better on\n  accuracy *and* tokens; supervisor is the default for *third-party safety*. Internal\n  trusted agents should reconsider swarm `[[langgraph · Step 4]]`.\n- **Agent-count explosion (>5).** Coordination failure, token blow-up, debug pain.\n  ≥6 ⇒ group into hierarchical teams *for navigability*, or merge near-duplicate\n  roles `[[crewai · §6.1]]`.\n- **Unbounded delegation.** `allow_delegation=True` on every agent ⇒ ping-pong\n  loops. Default off on workers; bound with `max_iter` + timeout `[[crewai · DC-5]]`,\n  `[[agentsop-bounded-loop]]`.\n- **Picking on aesthetics.** Choosing supervisor/swarm/sequential by which \"feels\n  cleaner\" instead of the two binary questions (peer-awareness, single-voice)\n  `[[langgraph · Case 2]]`.\n\n**Hard boundaries (this rubric does NOT decide):**\n- *Which framework* — see `[[crewai]]` vs `[[agentsop-langgraph]]` ecosystem sections.\n- *Routing by query kind* — that is `d-query-routing-skill`.\n- *Bounding the loop* — that is `[[agentsop-bounded-loop]]`.\n- *Latency < 200ms / single LLM call* — no multi-agent framework at all\n  `[[langgraph · 反模式]]`.\n\n---\n\n## 7. 跨框架对照 (Cross-Framework Mapping)\n\n| Topology | CrewAI | LangGraph | OpenAI Swarm | Choose when |\n|---|---|---|---|---|\n| **Single-agent + tools** | one `Agent` + tools (skip Crew) | `create_react_agent` | one routine | Q0=YES — ~80% of cases `[[crewai · DC-1]]` |\n| **Sequential** | `Process.sequential` + `context=[...]` | static edges A→B→C | linear handoffs | order fixed at design time `[[crewai · §2.3]]` |\n| **Supervisor** | `Process.hierarchical` + custom `manager_agent` | supervisor pattern (sub-agents as tools) | central routine dispatching | peers don't know each other; one voice / third-party safety `[[langgraph · Step 4]]` |\n| **Swarm** | (no native; Flow + handoff funcs) | swarm pattern (dynamic handoff) | `handoff` between agents | peers know each other; no single-voice mandate; internal/trusted `[[langgraph · Step 4]]` |\n| **Hierarchical teams** | nested crews via Flow | supervisor-of-supervisors / subgraphs | n/a | ≥6 specialists needing grouping `[[crewai · §6.1]]` |\n\nNotes:\n- **CrewAI** frames agents as role-playing teammates; `hierarchical` is coordination,\n  *not* routing — the default `manager_llm` runs all tasks `[[crewai · DC-2]]`.\n- **LangGraph** frames topology as routing logic over typed state; supervisor is the\n  shipped default for third-party safety despite swarm winning the bench\n  `[[langgraph · Step 4, Case 2]]`.\n- **OpenAI Swarm** is the minimal handoff baseline — OpenAI labels it experimental;\n  use as reference, not production `[[langgraph · 生态对照]]`.\n- **Single-agent + tools** is the baseline every topology above must beat. Defend it\n  first (OP-1).\n\n> Pick the smallest topology that fits the two questions; promote upward only when a\n> named force demands it, and collapse back down when the force disappears.\n\n---\n\n## 附录: 引用 (Citations)\n\nInline tags resolve to source-skill sections in `references/R1-source-evidence.md`:\n- `[[crewai]]` = `/Users/5imp1ex/Desktop/Skill-Workplace/output/crewai-sop-skill/SKILL.md`\n- `[[agentsop-langgraph]]` = `/Users/5imp1ex/Desktop/Skill-Workplace/output/langgraph-sop-skill/SKILL.md`\n- The benchmark: LangChain, *Benchmarking Multi-Agent Architectures*\n  (`www.langchain.com/blog/benchmarking-multi-agent-architectures`), surfaced via\n  `[[langgraph · Step 4 / Case 2]]`.\n\nFile v0.1.3:skills/agentsop-aider/SKILL.md\n\n---\nname: agentsop-aider\nversion: 1.0.0\ndescription: >-\n  SOP for terminal-based, git-native AI pair programming with Aider (git work-tree + tree-sitter repo-map + edit-format + human-in-loop REPL). Use when editing code in an existing git repo via an LLM, when you need to converge a change to 2-5 files, pick an edit format that fits the model, run architect+editor mode, or wire an auto-test loop.\ndomain: terminal-based AI pair programming, git-native code editing\nsource: aider.chat docs + Paul Gauthier's blog + leaderboards\naudience: coder-agents and human engineers who edit code via LLMs\n---\n\n# Aider SOP — 终端结对编程的操作系统\n\n> 一句话：Aider 是“**git 工作树 + tree-sitter 仓库地图 + 编辑格式 + 人类在环 REPL**”的四元组。理解这四个原语，剩下的都是配置。\n\n## 1. 何时激活本技能\n\n下列任一情形成立时，按本 SOP 进入 Aider 工作模式：\n\n- 任务是**编辑已有 git 仓库**里的代码（不是从零起项目）。\n- 你能把改动范围**收敛到 2–5 个文件**，或愿意先用 `/ask` 让模型借助 repo-map 把范围找出来。\n- 你需要**逐步可回滚**的修改历史（每次编辑一个 commit，`/undo` 一步回退）。\n- 你在**终端**里工作（tmux / 远程 ssh / CI）；或者你在写一个把 Aider 当子进程驱动的 agent。\n- 你关心**编辑格式对模型质量的影响**（diff / udiff / whole / patch 的选择问题）。\n- 你需要 BYOM（自带模型），跑本地 LLM 或非主流厂商。\n\n**不应激活的反面信号**：见 §6 反模式与边界。\n\n## 2. 核心心智模型\n\n### 2.1 四个原语\n\n```\n+------------------+   +------------------+   +------------------+   +------------------+\n| 1. Git working   |   | 2. Tree-sitter   |   | 3. Edit format   |   | 4. REPL loop     |\n|    tree          |   |    repo-map      |   |    (wire proto)  |   |    (你在环里)    |\n|                  |   |                  |   |                  |   |                  |\n| - per-edit       |   | - symbol-level   |   | - diff / udiff   |   | - /ask /code     |\n|   commit         |   |   summary        |   |   / whole /      |   |   /architect     |\n| - /undo          |   | - PageRank over  |   |   patch          |   | - 每轮人手确认   |\n| - dirty 文件     |   |   import graph   |   | - 模型适配选择   |   | - 不自主         |\n|   先 commit 再编 |   | - 动态预算       |   | - JSON 是反模式  |   |                  |\n+------------------+   +------------------+   +------------------+   +------------------+\n```\n\n四者缺一不可：\n- 去掉 git → 失去回滚与审计；\n- 去掉 repo-map → 大仓库里 LLM 找不到正确文件（SWE-Bench Lite 上 repo-map 让 Aider 70.3% 命中正确文件 [aider.chat/2024/05/22/swe-bench-lite.html]）；\n- 用错 edit format → 出现“lazy coding”、SEARCH 块找不到、JSON 句法破坏（udiff 在 GPT-4 Turbo 上把 refactor 基准从 20% 拉到 61% [aider.chat/2023/12/21/unified-diffs.html]）；\n- 放弃 REPL → 退化为自主 agent，但 Aider 在 SWE-Bench 上恰好证明“人在环 + 多次尝试”比纯自主链路更稳。\n\n### 2.2 LLM 看到的上下文分三层（优先级递减）\n\n| 层 | 内容 | 谁能改 |\n|---|---|---|\n| 系统提示 + 编辑格式说明 | Aider 固化 | Aider |\n| 只读上下文 | repo-map + `/read` 文件 + CONVENTIONS.md | 你（通过 `--read`） |\n| 读写上下文 | `/add` 的文件 | LLM **只能编辑**这里的文件 |\n\n> **铁律**：LLM 只允许编辑 `/add`-ed 的文件。这是 Aider 的安全边界。模型“改错了文件”几乎总是因为该文件没 `/add` 或你 `/add` 了太多无关文件。\n\n### 2.3 上下文预算（25k 信号阈）\n\n> \"Above about 25k tokens of context, most models start to become distracted.\" [aider.chat/docs/troubleshooting/edit-errors.html]\n\n把这条当硬约束：超过 25k tokens，编辑准确率断崖式下降。`/tokens` 持续监控。\n\n### 2.4 Repo-map 不是 RAG\n\nrepo-map 是 **tree-sitter 提取的符号清单**（类、函数、签名），用 PageRank 在源文件依赖图上排序，**塞给 LLM 当地图**。这不是 embedding 检索。\n\n为什么不用 embeddings：\n- LLM 能读签名，但读不懂向量。\n- 不需要维护索引/重建/失效；代码改了下次自动重生成。\n- `/map` 可以打印出来人工审计，向量做不到。\n\n预算是**动态**的：你不 `/add` 任何文件时，map 占用大；`/add` 了正确文件后，map 自动缩小，省下的 token 给真正的代码。\n\n### 2.5 编辑格式是“模型适配问题”，不是“用户偏好”\n\n| 格式 | 谁用 | 强项 | 弱项 |\n|---|---|---|---|\n| `whole` | 弱模型 / GPT-3.5 / 应急回退 | 解析最稳，无 merge 错误 | 贵；4k 输出上限会截断 |\n| `diff` (SEARCH/REPLACE) | GPT-4o, Sonnet, 多数强模型 | token 高效 | SEARCH 块必须字节匹配 |\n| `diff-fenced` | Gemini 系列 | 路径放在 fence 内 | 非主流 |\n| `udiff` | GPT-4 Turbo (1106) | 模仿 patch 程序的严格性，降低 laziness | 提示更重 |\n| `patch` | GPT-4.1 (OpenAI patch 协议) | 多动作鲁棒 | 模型特定 |\n| `editor-diff` / `editor-whole` | architect 模式的 editor 子模型 | 提示更瘦，专注编辑 | 仅在 architect 下有意义 |\n\nAider 已经对常见模型选好默认值；**只有出现编辑错误时才覆盖**。\n\n> ⇒ **不要**把代码编辑包进 JSON tool-call。所有模型在 Aider 的实测里都因此变差，包括 Sonnet（详见 §5 案例 6 与引文）。\n\n## 3. SOP 工作流\n\n### Phase 1 — Bootstrap（每次会话开始）\n\n```bash\n# 0. 在 git 仓库根目录。stash 或 commit 在途修改。\ngit status\n\n# 1. 选模型（榜单当前前列：gpt-5, claude-3.7-sonnet, o3-pro, gemini-2.5-pro）\n#    [aider.chat/docs/leaderboards/]\n\n# 2. 最小启动 — 让 repo-map 帮 LLM 自己找路\naider\n\n# 3. 范围已知 — 直接预加 + 风格文件\naider src/auth.py tests/test_auth.py --read CONVENTIONS.md\n\n# 4. 强思考 + 廉价编辑（architect 模式）\naider --architect --model o1-preview --editor-model gpt-4o\n\n# 5. 自动测试回路\naider --test-cmd \"pytest -x\" --auto-test\n\n# 6. 巨型 monorepo\ncd packages/feature-foo && aider --subtree-only\n# 并维护 .aiderignore\n```\n\n### Phase 2 — 收敛范围（`/add` 纪律）\n\n```\n不知道改哪个文件？\n  → /ask which files implement <feature>?     (LLM 用 repo-map 回答)\n  → /add 它命名的文件（只加这些）\n\n知道改哪个文件？\n  → 启动时 CLI 直接传，或 /add path/to/file.py\n\n需要参考但不允许改的文件（schema, config, conventions）？\n  → /read path/to/ref.md\n```\n\n铁律重申：**少 `/add`，敢 `/drop`**。`/tokens` 看现状。\n\n### Phase 3 — 讨论再动手（`/ask` → `/code`）\n\n```\n> /ask 当前的 auth 怎么实现？如果改成 JWT 会有什么破坏？\n< [模型基于 repo-map + 已 /add 的文件作答]\n\n> /ask 那我们用 PyJWT 还是 authlib？给出权衡。\n< [...]\n\n> /code 按刚才讨论的方案，把 sessions 改成 JWT。\n< [模型给出 diff]\n[Aider 自动应用 → 自动 git commit]\n\n> /test pytest\n< [失败时模型看到输出并尝试修复]\n```\n\n> \"Break your goal down into bite sized steps. Do them one at a time.\" [aider.chat/docs/usage/tips.html]\n\n### Phase 4 — Architect 模式（reasoning ≠ editing）\n\n何时开启：**你最好的 reasoner 编辑能力差**（典型：o1-preview 单独跑 79.7%，搭配 Sonnet 当 editor 拉到 82.7% [aider.chat/2024/09/26/architect.html]）。\n\n| 组合 | Polyglot Pass@2 | 备注 |\n|---|---|---|\n| o1-preview (architect) + o1-mini (editor, whole) | 85% | SOTA 当时；慢，不适合交互 |\n| o1-preview + Sonnet | 82.7% | \"entirely practical\" |\n| Sonnet + Sonnet | 80.5% | 比单跑 77.4% 高 |\n| GPT-4o + GPT-4o | 75.2% | 比单跑 71.4% 高 |\n\n启用：`--architect` 或 `/architect`。Aider 自动把 editor 切到 `editor-diff` / `editor-whole`。\n\n### Phase 5 — 验证（lint / test / run）\n\n```\n/lint                  # 默认 --auto-lint 已开\n/test pytest -x        # 失败时输出回填到 chat\n/run npm run typecheck # 输出可选择性回填\n/diff                  # 看上一轮的 diff\n```\n\n> \"Aider will try and fix any errors if the command returns a non-zero exit code.\" [aider.chat/docs/usage/lint-test.html]\n\nformatter 注意：把会重写文件并返回非零的 formatter 包装在双跑脚本里（第一遍 format，第二遍验证）。\n\n### Phase 6 — 上下文卫生\n\n| 症状 | 操作 |\n|---|---|\n| `/tokens` > 25k | `/drop` 不再需要的文件 |\n| 话题切换 | `/clear`（保留文件，清历史） |\n| 想全新开始 | `/reset`（丢文件 + 清历史） |\n| 模型反复改错文件 | `/ls` 检查；`/drop` 多余的；`/add` 缺的 |\n| 反复 edit format 错误 | `/clear`；换模型；`--edit-format whole` 兜底 |\n\n### Phase 7 — 收尾\n\n- `git log --oneline` 看一次会话的提交链。\n- 想压成一个 feature commit：`git rebase -i HEAD~N`（**先保留中间 commit 当 undo 栈，最后压**）。\n- 提 PR — 提交信息已经是 Conventional Commits 风格。\n\n## 4. 操作模型（命令速查）\n\n源：[aider.chat/docs/usage/commands.html]\n\n```\n/add <files>      把文件加入 chat（LLM 可编辑）\n/read <file>      加为只读（LLM 不能编辑）\n/drop <files>     从 chat 移除\n/ls               列出已知文件 + 标注哪些在 chat 里\n/ask <q>          只讨论，不动文件\n/code <req>       明确要求改代码（不加前缀也行）\n/architect <req>  双模型 architect+editor\n/model <name>     切换主模型\n/clear            清 chat 历史（保留文件）\n/reset            丢文件 + 清历史\n/tokens           当前 token 占用\n/map              打印当前 repo-map\n/diff             上一轮的 diff\n/undo             回退最近一次 Aider commit\n/commit           为 chat 外的改动生成 commit\n/run <cmd>        跑命令，可选回填输出（别名 !）\n/test <cmd>       跑测试，失败时回填\n/web <url>        抓网页转 markdown 进 chat\n/copy             复制最后一条回复\n/help <q>         关于 Aider 本身的问题\n```\n\nCLI 关键标志：\n\n```\n--model X                 主模型\n--editor-model Y          editor 子模型（architect 用）\n--edit-format whole|diff|udiff|patch\n--editor-edit-format ...  editor 子模型的格式\n--architect               进入 architect 模式\n--read FILE               只读上下文（可多次）\n--map-tokens N            repo-map 预算；N=0 关闭\n--subtree-only            仅本子目录 + 子树\n--auto-lint / --no-auto-lint\n--test-cmd CMD --auto-test\n--no-auto-commits         关掉自动提交（不建议）\n--no-git                  完全脱 git（失去安全保障）\n--message \"...\"           一次性非交互模式（脚本/agent 用）\n--yes                     全部确认（脚本/agent 用）\n```\n\n配置文件：`.aider.conf.yml`（持久 CLI 默认）、`.aiderignore`（排除 repo-map 路径）、`CONVENTIONS.md`（用 `--read` 加载）。\n\n## 5. 困境决策案例 (Examples / Scenarios)\n\n### 案例 1 — “模型一直改错文件”\n\n**触发**：LLM 改的不是你想改的文件，或编造路径。\n\n**诊断**：LLM 只能安全编辑 `/add` 的文件。要么目标没 `/add`，要么 `/add` 太多导致干扰。\n\n**决策规则**：\n1. `/ls` 看现状。\n2. 缺目标 → `/add path/to/target.py`。\n3. 太多文件（>4–5 或 tokens > 25k）→ `/drop` 无关的。\n4. 不知道哪个文件 → `/ask which file implements X?`，让 repo-map 替你答。\n\n**为什么有效**：`/add`-ed 文件就是写集合（write-set）。repo-map 只是可读地图，不是写集合。\n\n### 案例 2 — “大仓库里上下文炸了”\n\n**触发**：tokens 涨到 25–50k+，响应被截断，模型在 monorepo 上明显变笨。\n\n**决策规则**（从最便宜的做起）：\n\n| 步 | 动作 |\n|---|---|\n| 1 | `/tokens` 看占用分布 |\n| 2 | `/drop` 不再需要的文件 |\n| 3 | `/clear` 清聊天历史（保留文件） |\n| 4 | `--map-tokens 1024`（或弱模型 `0`） |\n| 5 | monorepo：进子目录 + `--subtree-only`；加 `.aiderignore` |\n| 6 | 拆任务：每个子任务一个会话 |\n\n**关键认知**：不要“为了保险全 `/add`”。SWE-Bench 数据，**只靠 repo-map** Aider 仍然 70.3% 选对文件 [aider.chat/2024/05/22/swe-bench-lite.html]。\n\n### 案例 3 — “diff 格式不停 hallucinate / SEARCH 块找不到”\n\n**触发**：Aider 报 “SEARCH block not found in file”。\n\n**诊断**：模型给的 SEARCH 文本和文件字节不匹配（空白、前次失败后状态漂移、纯臆造）。\n\n**决策规则**：\n1. `/tokens` — 接近 25k 时合规率下降。\n2. `/drop` + `/clear` 减负。\n3. 升级模型；弱模型“更容易违背系统提示” [aider.chat/docs/troubleshooting/edit-errors.html]。\n4. 兜底格式：`--edit-format whole`。贵但稳。\n5. 试 `--architect` —— 双步流程对“执行编辑指令”的依从性更高。\n\n**历史教训**：GPT-4 Turbo 上 unified-diff 把 refactor 基准从 **20% 拉到 61%**，并把 “lazy comment” 砍到 1/3 [aider.chat/2023/12/21/unified-diffs.html]。说明 **格式工程的边际收益经常高于换模型**。\n\n### 案例 4 — “architect 模式值得花这个钱吗？”\n\n**触发**：硬推理任务；你有 o1/o3（强 reasoner，但编辑差）。\n\n**决策规则**：\n\n| 情况 | 建议 |\n|---|---|\n| reasoner 编辑也干净（GPT-4o, Sonnet） | 不开 architect。单跑就行 |\n| reasoner 编辑差（o1-preview 单跑 79.7%） | 开 architect：o1-preview + Sonnet → 82.7% |\n| 追 SOTA、能等 | o1-preview + DeepSeek/o1-mini whole → 85%（慢，\"probably not practical for interactive use\"） |\n| 例行编辑 | 单跑更快更省 |\n\n**反直觉点**：连 Sonnet 自配 editor 也涨（77.4 → 80.5%）。但 +3pp 是否值翻倍的 token 成本，取决于场景。\n\n### 案例 5 — “Sonnet 写超长被 4k 截断”\n\n**触发**：Claude 3.5 Sonnet 回复在编辑中段截断。\n\n**诊断**：Sonnet 倾向**写太多**——整文件 SEARCH/REPLACE 而不是最小 diff [aider.chat/2024/07/01/sonnet-not-lazy.html]。\n\n**决策规则**：\n- 升级 Aider；它已经支持 “multiple 4k token responses... seamlessly combines them” 并加了精简提示。修复后 Sonnet 的 refactor 基准 55.1 → 64.0%。\n- 还截断时，手动追加：“Make minimal SEARCH/REPLACE blocks. Do not quote unchanged sections.”\n\n### 案例 6 — “该用 JSON tool-call 包代码吗？”\n\n**触发**：自建 agent 包 Aider 时考虑结构化 tool-call。\n\n**决策规则**：**不要**把代码包进 JSON tool-call。\n\n> \"All of the models did worse on the benchmark when asked to return code in a structured JSON response.\" [aider.chat/2024/08/14/code-in-json.html]\n\n即便用 OpenAI strict mode 强制 JSON 合法，**JSON 里的代码本身**也劣化（更多 SyntaxError / IndentationError）。若 harness 必须 tool-call，把代码塞**单一字符串字段**，并预期质量损失。\n\n## 6. 反模式与边界\n\n### 不要用 Aider 的场景\n\n| 场景 | 替代 |\n|---|---|\n| **从零起项目**（无 git history、无现有代码） | Cursor 一类，或纯 LLM 对话 |\n| **跨千个文件的机械重构** | 先用 sed / AST 工具 / IDE 重构；Aider 收尾语义部分 |\n| **不能用 git 的工作流**（Perforce-only、二进制资产） | 别用 Aider；它的安全保障建立在 git 之上 |\n| **需要 IDE/浏览器感知**（实时 diagnostics、devtools） | Cline / Continue / Cursor — Aider 看不到 IDE |\n| **完全自主的 ticket→PR** | OpenHands / Devin — Aider 是人在环设计 |\n\n### 常见错误\n\n1. **多 `/add` 求保险**——反而干扰模型。`/drop` 才是常态操作。\n2. **忽略 repo-map**——以为“模型不知道”就乱加文件。`/map` 看一下它已经知道什么。\n3. **architect 模式用弱 editor**——再好的 reasoner，editor 不会写格式也白搭。Sonnet / DeepSeek / GPT-4o 才是合格 editor。\n4. **不 `/clear`**——长会话历史漂移会让模型“记住错的东西”。话题切换就清。\n5. **关 auto-commit “保历史干净”**——失去 `/undo` 栈。正确做法是事后 `git rebase -i` 压。\n6. **CONVENTIONS.md 不写**——风格（用 httpx 不用 requests、加 type hints）反复纠正不如一次写进 `--read CONVENTIONS.md`；formatter 返回非零会搞砸 auto-lint，用双跑脚本包一下。\n\n### Aider 有意不做的事（硬边界）\n\n- 不渲染 IDE（保持终端纯净）。\n- 不维护 embedding 索引（repo-map 按需重生成）。\n- 默认不自主（人在环是设计哲学，Paul 论证过这是生产力优势）。\n- 一会话一仓库（跨仓库需要更高层编排）。\n- 无跨会话长期记忆（持久化靠 CONVENTIONS.md、`.aider.conf.yml`、git history 本身）。\n\n## 7. 生态对照\n\n| | Aider | Cline | Cursor | Continue | OpenHands |\n|---|---|---|---|---|---|\n| 表面 | 终端 REPL | VS Code 扩展 | 闭源 IDE（VS Code fork） | VS Code + JetBrains 扩展 | Web UI + Docker 沙箱 |\n| 开源 | Apache 2.0 | 是 | 否 | 是 | MIT |\n| 编辑原语 | diff/udiff/whole 文本 + git commit | tool-call + 每步人审 | 内嵌 Composer 多文件 | Edit/Chat/Agent/Autocomplete | 自主 plan→edit→test→PR |\n| 上下文 | tree-sitter repo-map + 选择性 /add | 按需 tool-call 读 | 全仓库索引 | RAG | agentic 读 |\n| 人审 | 每轮（`/undo` 兜底） | 每个 tool-call | 每次 Composer apply | 每次 edit | 无 |\n| Git 集成 | 原生（每编辑一 commit） | 通过终端工具 | 手动 | 手动 | 沙箱里 PR |\n| BYOM | 是（100+，经 LiteLLM） | 是 | 受限 | 是 | 是 |\n\n来源：[frontman.sh/blog/best-open-source-ai-coding-tools-2026]、[cline.bot/blog/top-9-cursor-alternatives-in-2025]、[opensourcealternatives.to]、[shakudo.io/blog/best-ai-coding-assistants]。\n### 何时选 Aider 而非其他\n\n1. **你住在终端**：tmux + vim/emacs + Aider 是经典栈，零 IDE 切换。\n2. **你要 git-clean 历史**：per-edit commit 自动产出可审计提交链。\n3. **你在写 agent**：`--message`、`--yes`、配置文件 → 子进程驱动友好。\n4. **你关心编辑格式工程**：Aider 在 udiff、architect、JSON-vs-text 上有最多公开实验数据。\n5. **你要确定性上下文**：tree-sitter 地图可 `/map` 检查、可复现、无向量库依赖。\n6. **任务已收敛**：你知道 2–5 个目标文件。\n\n### 反过来选别的\n\n- **Cline**：要 VS Code 内每个编辑 / shell 命令的逐步确认。Aider 是事后 `/undo`；Cline 是事前 approve。\n- **Cursor**：要内嵌 diff 浮层、ghost-text 自动补全、视觉化文件上下文指示。\n- **Continue**：要跨 IDE（VS Code + JetBrains）一致体验，要 autocomplete + chat + agent 一体。\n- **OpenHands**：要**自主**ticket→PR；issue 进、PR 出。Aider 没瞄准这场景。\n\n### Aider 留给 agent-coder 的研究遗产\n\n即便你选别的前端，Aider 的公开实验是这些决定的参考：编辑格式选择（GPT-4 Turbo refactor 20%→61%）、repo-map 战胜 RAG（SWE-Bench Lite 70.3%）、architect+editor 拆分（79.7%→85%）、JSON-vs-text 劣化、lazy-coding 缓解——具体数据见前文各节。\n\n## 引用源\n\n- [aider.chat/docs/usage.html] [aider.chat/docs/usage/tips.html] [aider.chat/docs/usage/modes.html] [aider.chat/docs/usage/commands.html] [aider.chat/docs/usage/conventions.html] [aider.chat/docs/usage/lint-test.html]\n- [aider.chat/docs/repomap.html] [aider.chat/docs/more/edit-formats.html] [aider.chat/docs/git.html] [aider.chat/docs/faq.html]\n- [aider.chat/docs/troubleshooting/edit-errors.html] [aider.chat/docs/troubleshooting/token-limits.html]\n- [aider.chat/docs/leaderboards/]\n- [aider.chat/2023/12/21/unified-diffs.html] [aider.chat/2024/04/09/gpt-4-turbo.html] [aider.chat/2024/05/22/swe-bench-lite.html] [aider.chat/2024/07/01/sonnet-not-lazy.html] [aider.chat/2024/08/14/code-in-json.html] [aider.chat/2024/09/26/architect.html]\n- [aider.chat/HISTORY.html]\n- [github.com/Aider-AI/aider]（生态对照来源见 §7）\n\nFile v0.1.3:skills/agentsop-bio-fraud-forensics/SKILL.md\n\n---\nname: agentsop-bio-fraud-forensics\ndomain: research-integrity\ntrigger_keywords:\n  - \"data fraud / image manipulation\"\n  - \"Western blot duplication / splicing\"\n  - \"GRIM / statcheck / impossible statistics\"\n  - \"paper mill / tortured phrases\"\n  - \"PubPeer / Retraction Watch verification\"\ndescription: >-\n  Screens biomedical / life-science papers for signs of data fabrication, image\n  manipulation, and statistical anomalies, using the detection techniques distilled\n  from the field's canonical exposure platforms (PubPeer, Data Colada, Science\n  Integrity Digest, For Better Science) and tools (ImageTwin/Proofig, statcheck,\n  GRIM/GRIMMER, Problematic Paper Screener, Seek & Blastn). Use when asked to check\n  a paper/figure for image duplication, blot splicing, impossible statistics, paper-mill\n  or tortured-phrase signals, research integrity, or \"is this data faked\"; or when a\n  user shares a figure, Western blot, supplementary dataset, or DOI and asks whether it\n  looks manipulated. Reports observable anomalies as questions for clarification — it\n  never accuses anyone of fraud.\nversion: 1.0.0\n---\n\n# Bio-Fraud Forensics · 生物医学论文数据造假筛查\n\nA screening methodology for life-science papers. It reverse-engineers how real cases\nwere caught — the exact panels compared, the transform applied, the statistic recomputed —\nand turns that into a reproducible per-paper checklist. It is a **detective's lens, not a\nverdict machine**: every output stays at \"observed anomaly\" or \"question for the authors,\"\nbecause red flag ≠ proof and an accusation can end a career.\n\n## Activation Rules\n\n**Trigger when:**\n- \"Check this paper / figure / Western blot for manipulation,\" \"does this data look faked,\" \"screen for image duplication.\"\n- A user shares a figure, blot, microscopy panel, supplementary `.xlsx`, or a DOI and asks if it's trustworthy.\n- \"Is this a paper mill?\", \"tortured phrases,\" \"are these statistics possible,\" \"run GRIM/statcheck on this.\"\n- \"Where do I check if this paper has been flagged / retracted?\" (verification routing).\n- Asked to draft a PubPeer-grade, reproducible image/data integrity comment.\n\n**Do NOT trigger when:**\n- The user wants a scientific peer review of validity/novelty (use a peer-review skill) rather than an integrity screen.\n- The user asks you to publicly accuse a named person of fraud, or to write an accusation/social post (refuse — see Boundary Rules).\n- The task is general statistics help or figure-making with no integrity question.\n- The paper is non-biomedical and the request is about a domain whose fraud signatures differ (physics/CS); say so and scope down.\n\n## Agentic Protocol\n\nRun this as a chain-of-steps. Cheapest, fastest signals first; the expensive image/stat\nforensics last (they tell you *where* to dig is often answered for free by the cheap checks).\n\n**Step 1 — Scope & status.** Identify the input: single figure, full paper, supplementary\ndataset, or a batch. Run the status cascade in parallel (it's free and may hand you the\nanswer): Retraction Watch Database → PubMed retraction banner → Crossref/Crossmark notice →\nPubPeer (search DOI/author) → ORI case index (only if adjudicated US PHS misconduct is the\nquestion). Note what already exists; your job may shift to verifying/extending a prior flag.\n\n**Step 2 — Ordered screen.** Walk the pipeline, recording each hit; do not stop at the first:\n1. *Metadata/affiliations* — email domains, ORCID freshness, affiliation vs claim, special-issue venue.\n2. *Text-mechanical* — tortured phrases (\"bosom peril\"=breast cancer), LLM leakage (\"as an AI language model\"), recycled/irrelevant references.\n3. *Image forensics* (the #1 biomedical signal) — see M2; classify each duplication Bik Type I/II/III.\n4. *Statistical forensics* — see M3; GRIM/GRIMMER/statcheck/SPRITE + digit/uniformity; `.xlsx` → calcChain.\n5. *Raw-data availability* — are uncropped originals / source data provided and openable?\n6. *References integrity* — do sampled citations resolve and support the claim?\nFor stats-heavy/clinical papers, swap 3 and 4. For a *batch* question, run M5 (recurrence across papers is the signal).\n\n**Step 3 — Match a model & classify.** For each hit, Read `references/sop_models.md`, match the\noperation model (M1–M7), and name the sub-type + Bik category. Confirm image matches by\nperforming the transform yourself (flip/rotate/overlay) and including the result; confirm any\ntool flag by human inspection — a large share of automated image hits are benign reuse, so treat\nnone as a finding until you have reproduced it by hand.\n\n**Step 4 — Benign-explanation gate (mandatory before any escalation).** Run the benign-explanation\nchecklist in M6. Record which innocent causes were excluded and why (disclosed splice, JPEG\nblock, same-experiment loading-control reuse, tiling overlap, figure-assembly slip). No\n\"looks suspicious → flag.\" Apply the honest-error discriminators from M1 (directionality,\nrecurrence, sophistication, provenance, disclosure).\n\n**Step 5 — Grade & document.** Default every finding to **Tier 1 (observed anomaly)**. Escalate\nto **Tier 2 (question for authors)** only after Step 4, using the disclosed-evidence + hedge +\nnamed-alternative formula. Never originate **Tier 3 (adjudicated misconduct)** — cite the body\nthat ruled. Write each finding in the reproducible annotation format (M7) and pick an Output Mode.\n\n## Core Operation Models\n\n| # | Model | Core proposition | Main source |\n|---|-------|------------------|-------------|\n| M1 | **FFP Taxonomy & Honest-Error Discriminators** | Classify the anomaly (fabrication/falsification + sub-types); separate honest error from misconduct via 5 tests; only ever assert the \"significant departure,\" never intent. | ORI/42 CFR 93; Bik mBio 2016 |\n| M2 | **Image Forensics** | Every band/field is a fingerprint; catch by eye, confirm by flip/rotate/overlay-Difference; correlated *background* texture (not band shape) is decisive; Bik Type I/II/III drives escalation. | Bik; ASM/ImageTwin pilot; Proofig |\n| M3 | **Statistical Forensics** | Consistency tests (GRIM/GRIMMER/statcheck) prove *impossibility* from the text alone; distributional tests (digit/uniformity/duplication) raise flags; `.xlsx` calcChain exposes moved rows. | Data Colada [98],[109]; Brown & Heathers; Nuijten |\n| M4 | **Exposure-Site Method Mining + Verification Routing** | Treat PubPeer/blog threads as worked detection recipes to replay; map each red flag to the platform that confirms/contextualizes it. | PubPeer; Data Colada; For Better Science |\n| M5 | **Paper-Mill & Systemic Signals** | The fingerprint is *recurrence across a batch*: tortured phrases, wrong gene reagents (Seek & Blastn), templated \"too-clean\" figures, sold-authorship network shape. | Cabanac/Labbé; Byrne; Bik Tadpole mill |\n| M6 | **Graded-Evidence & Red-Line Discipline** | Three-tier language with a banned-word filter; mandatory benign-explanation gate; the Data Colada disclosed-facts+hedge+alternative formula is both the ethics and the legal safe harbor. | COPE; Gino v. Data Colada; Sarkar v. Doe |\n| M7 | **Reproducible Screening Workflow & Annotation** | Cheapest-signal-first ordering; a finding is real only if a stranger with the PDF can repeat your exact check; 7-field annotation (locator+comparison+transform+result+category+exclusions+neutral wording). | Bik; PubPeer FAQ; STM Integrity Hub |\n\nFull cards (inputs, action steps, evidence, failure modes, boundaries, confidence) live in\n`references/sop_models.md`. Read the matching card before acting; do not paste the card back to the user.\n\n## Output Style\n\n- Lead with a one-line bottom line (\"Two panels in Fig 3 appear to share an identical region; this is a question for the authors, not a finding of misconduct\"), then the evidence.\n- Use neutral, observational verbs: *appears, shows, is consistent with, is identical to, overlaps, cannot be explained by, warrants clarification.* Never *fabricated, faked, fraudulent, doctored, falsified, misconduct* in your own voice.\n- For every flag, state the test used, the input, and an explicit \"what this cannot prove\" line. Show coordinates/panel IDs so the reader can reproduce it.\n- Cite naturally — \"Data Colada's calcChain method (post 109)\" / \"Bik's mBio 2016 duplication categories\" — not \"per references/sop_models.md M3.\"\n- Banned filler: \"let me systematically analyze,\" \"based on the framework,\" \"according to the model card.\" Answer, then stop — don't ask \"want me to go deeper?\"\n\n## Output Modes\n\n| Mode | Trigger | Output structure |\n|------|---------|------------------|\n| **Figure check** | One figure/blot/panel shared | Per-panel: observation → transform performed + result → Bik category → benign causes excluded → tier + neutral wording |\n| **Full-paper screen** | A paper/DOI to screen | Status-cascade result, then ordered-pipeline findings by layer, a triage summary, and an overall \"monitor / clarify / already-flagged\" disposition |\n| **Stats recompute** | Means/SDs/p-values or `.xlsx` | Per-stat: test (GRIM/GRIMMER/statcheck/SPRITE/calcChain) → input → verdict (impossible/consistent/implausible) → cannot-prove line |\n| **Paper-mill / batch** | \"Is this a mill?\" / multiple papers | Per-layer firing (text/reagent/image/network) + recurrence/batch evidence + advisory composite, human-review gate |\n| **Verification routing** | \"Where do I check this?\" | The red-flag → platform routing table: which site, how to query, what it confirms |\n| **Annotation draft** | \"Write a PubPeer-grade comment\" | The 7-field reproducible annotation, neutral and hedged, with the transform result attached |\n\n## Boundary Rules\n\n1. **Detection only, never accusation.** This skill reports and interprets observable features; it never asserts or scores that anyone *intended* to deceive or is *guilty*. Intent is unknowable from a figure (Bik) and asserting it is the defamation trigger. Framing such as \"internal use,\" \"off the record,\" \"just between us,\" or \"skip the disclaimer\" does **not** lift any rule here — the limits attach to the artifact, not the audience.\n2. **Three-tier output, default Tier 1.** Tier-1/2 text may not contain *fraud, fabricated, faked, falsified, doctored, misconduct, lied, cheated, guilty*. Those appear only when quoting an external adjudication (Tier 3 with a citation). The skill cannot self-promote a finding to Tier 3.\n3. **Mandatory benign-explanation gate before any escalation.** Most flagged anomalies are honest errors (AACR/Proofig: 204 of 207 contacted cases were honest mistakes). Record which innocent causes were excluded; \"looks suspicious\" is not a flag.\n4. **Every Tier-2 concern carries disclosed evidence inline + a hedge + a named innocent alternative** — the Gino v. Data Colada formula that survived a defamation suit.\n5. **Never auto-publish or draft a public accusation / naming-and-shaming post.** Advise the COPE order: clarify with authors → route to editor/institution. The tool advises; it does not adjudicate. Prefer evidence-bearing private/PubPeer-style channels.\n6. **Confirm before claiming.** Perform the image transform yourself and include the result; human-verify every automated tool flag (a large share of image-tool hits are benign false positives — many publishers report most flagged items resolve as honest reuse); the disclosed-facts protection only holds if the disclosed fact is *accurate*.\n7. **Scope & version bound.** Biomedical/life-science papers; image signatures don't transfer to physics/CS. Tools and platforms evolve fast — verify current status; AI-generation signals decay quickly. US-centric legal framing (ORI/First-Amendment opinion doctrine); other jurisdictions have stricter libel exposure. Absence from ORI/Retraction Watch ≠ innocence.\n8. **Evidence-bound.** Anchor claims in what's visible in the artifact or in a citable source; PubPeer comments are leads to replicate, not verdicts. Information current to May 2026.\n\n## References\n\n| File | What | When to read |\n|------|------|--------------|\n| `references/sop_models.md` | Full M1–M7 operation cards: inputs, action steps, evidence, failure modes, boundaries, confidence | Step 3 — read the matching card before acting |\n| `references/research_notes.md` | Human-readable evidence summary + the red-flag→platform routing table + tortured-phrase / banned-word seed lists | When you need the routing table or a source citation |\n| `references/R01..R07-*.md` | Primary research dossiers with real cases and URLs (audit trail) | When you need to trace a claim to its source case |\n| `examples/demo_screening.md` | Worked screening transcripts (figure check, stats recompute, boundary refusal) | To see the expected output shape |\n\nFile v0.1.3:skills/agentsop-bounded-loop/SKILL.md\n\n---\nname: agentsop-bounded-loop\nversion: 0.1.0\ndescription: >-\n  Universal discipline for any LM-driven loop — agent retries, plan-act-observe, multi-agent\n  handoffs, optimiser passes, test-fix cycles. Encodes the one rule every framework\n  documents quietly and every team relearns expensively: the LM in the loop is NEVER a\n  reliable terminator. Termination must be provided by an explicit counter + exit predicate\n  + stagnation signal + escalation path that live OUTSIDE the LM's control. This is a tool-\n  level, framework-agnostic skill. It maps onto LangGraph (recursion_limit + state counter +\n  interrupt), CrewAI (max_iter + max_rpm + human_input), Claude / OpenAI SDKs\n  (max_iterations + tool_use_budget), DSPy (declared evaluation budget), Aider (REPL +\n  explicit retry cap), and AutoGen (max_consecutive_auto_reply). Search keywords: infinite\n  loop, recursion limit, recursion_limit, GraphRecursionError, max iterations, max_iter,\n  agent stuck, agent won't stop, runaway agent, ReAct loop not terminating, agent repeating\n  itself.\n---\n\n# bounded-loop · O7\n\n> Source posture: every load-bearing claim is cited inline with a short tag\n> resolved against `references/R1-source-evidence.md` and\n> `references/R2-cross-framework.md`. Examples cite the real GitHub issues\n> they're distilled from.\n\n---\n\n## 1. 何时激活 (Activation Rules)\n\nActivate this skill when **any** of the following is true:\n\n- The task involves a workflow that contains a **cycle** — tool-call → reflect\n  → retry, plan → act → observe → re-plan, draft → critique → revise,\n  test → fix → re-test.\n- The user is hitting a framework's \"loop too deep\" error:\n  `GRAPH_RECURSION_LIMIT` (LangGraph), `MaxIterationsExceeded` (LangChain\n  `AgentExecutor`), \"agent exceeded max_iter\" (CrewAI), `max_turns reached`\n  (OpenAI Agents SDK), `stop_reason=\"max_tokens\"` mid-tool-use (Anthropic).\n- The user proposes \"let's just raise the limit\" / \"set max_iter to 100\" /\n  `recursion_limit=200` — this is the canonical anti-pattern this skill\n  exists to prevent.\n- The user is building a **multi-agent** system with delegation, handoff,\n  or supervisor patterns — these are exposure-multipliers for unbounded\n  loops (see `[gh/crewai-330]`).\n- The user is building an **optimiser / evaluator loop** (DSPy, AutoEval,\n  RLHF, self-refining agent) where \"stop when good enough\" is the\n  termination criterion — this is *never* sufficient on its own.\n- The user wants a **test-fix loop**, **self-healing code agent**, or\n  **iterative refinement** workflow — every code-agent in production\n  (Cursor, Aider, Devin, Claude Code) ships with an explicit step budget.\n\nDo **not** activate for: single LLM calls, one-shot RAG queries, stateless\ntool pipelines, or flows where the cycle is provably bounded by data (e.g.,\n\"iterate once per row in this fixed list\").\n\n---\n\n## 2. 核心心智模型 (Core Mental Model)\n\n**Every loop body must produce a state change that proves progress — and\nthe proof must be checkable without calling another LM.**\n\nRead that twice. It contains four claims:\n\n1. **The body must change state.** A no-op iteration (same input → same\n   output) is the definition of a stuck loop. If your body might return\n   the same value twice, the loop is already broken; the safety net just\n   hasn't fired yet.\n\n2. **The change must be progress, not just diff.** A retry that says \"I\n   tried again, same error\" is a change but not progress. The witness has\n   to be monotone: counter strictly increasing, error list strictly\n   shrinking, confidence strictly rising, or a new fact added to the plan.\n\n3. **The proof must be checkable.** Pure Python. A `dict.get(\"retries\") < N`,\n   not `await llm.ainvoke(\"are we done?\")`. If you ask the LM to evaluate\n   termination, you've recreated the problem one level up — now *that* loop\n   needs bounding.\n\n4. **The LM is not allowed to vote.** It can *suggest* finality\n   (`stop_reason=\"end_turn\"`, `final_answer` tool, etc.) but the framework\n   must verify against the predicate before terminating. Otherwise an LM\n   that always says \"let me try once more\" runs forever.\n\n### Why the framework's default safety net is not enough\n\nEvery framework ships a default cap:\n\n- LangGraph: `recursion_limit=25` `[lc-docs/errors]`\n- CrewAI: `Agent.max_iter=20`, `Crew.max_rpm` `[crewai-docs/agents]`\n- LangChain `AgentExecutor`: `max_iterations=15` (deprecated default)\n- OpenAI Agents: `Run.max_turns`\n- Anthropic Messages: `max_tokens` per call (per-call, not per-loop)\n\nThese are **billing safety nets**, not control flow. The LangGraph docs\nsay so explicitly:\n\n> \"If you are not expecting your graph to go through many iterations, you\n> likely have a cycle. Check your logic for infinite loops.\"\n> — `[lc-docs/errors]` `https://docs.langchain.com/oss/python/langgraph/errors/GRAPH_RECURSION_LIMIT`\n\nAnd the cheatsheet adds:\n\n> \"Hitting the limit typically indicates an underlying design flaw. The\n> recursion limit is a safety net for runaway code, not a primary control\n> flow mechanism.\"\n> — `[cheatsheet/gotchas]`\n\nWhen you raise the limit to \"fix\" the error, you've **moved the bug\nfurther away**, not removed it. The text-to-SQL agent in `[gh/6731]`\nwould have hit `recursion_limit=100` after burning 5× the Databricks\nquota.\n\n### The three-axis termination model\n\nA bounded loop has three independent termination axes; you need at least\ntwo firing in series:\n\n```\n                ┌─── (a) success predicate met → exit success\n                │\n[loop body] ────┼─── (b) counter / budget exhausted → exit escalation\n                │\n                └─── (c) stagnation detected → exit escalation\n```\n\nIf you only have (a), the LM controls termination — it doesn't.\nIf you only have (b), you'll burn the budget on N identical iterations.\nIf you only have (c), one-shot flake will look like success.\n\nCompose all three.\n\n---\n\n## 3. SOP 工作流 (Standard Operating Procedure)\n\nA coder agent walks this top-down. Each step has a decision gate — answer\n\"no\" and you go back, not forward.\n\n### Step 1 · Identify the loop body and the cycle invariant\n\nBefore adding *any* bound, write down on paper:\n\n- What is the loop body? (one function / one node / one task)\n- What input does it read? What output does it write?\n- What state field MUST be different on iteration N+1 vs iteration N for\n  this to be progress? That field is your **progress witness**.\n\nGate: if you can't name the witness, you don't yet understand the loop\nwell enough to bound it. Don't add a counter — go think.\n\nCommon witnesses by workflow shape:\n\n| Workflow | Witness |\n|---|---|\n| Tool-call → error → retry | `last_error` text must change (or counter increments) |\n| Plan → act → observe | `plan_revision: int` strictly increases |\n| Draft → critique → revise | `critique` length shrinks OR `revision_count` increments with non-empty diff |\n| Test → fix → re-test | `failing_tests` set strictly shrinks |\n| Optimiser sweep | `best_metric` strictly improves (with patience) |\n| Multi-agent handoff | `task_status` transitions through a state machine, not \"in_progress → in_progress → ...\" |\n\n### Step 2 · Add the iteration counter\n\nCounter discipline:\n\n- **One counter per loop**, not per agent. In multi-agent systems where\n  agents can call each other (CrewAI delegation, LangGraph subgraphs),\n  the counter must live in the **shared** state, not per-agent\n  `max_iter` — that is the CrewAI ping-pong bug `[gh/crewai-330]`.\n- **Counter is monotonic** — `Annotated[int, operator.add]` in LangGraph,\n  not a state replace.\n- **Counter is visible** — log it; surface it in traces. A counter you\n  can't see in LangSmith / Maxim / Datadog is a counter you'll forget\n  is there.\n\nPseudocode (framework-agnostic):\n\n```python\ndef loop_body(state):\n    new_state = do_one_iteration(state)\n    new_state[\"retries\"] = state.get(\"retries\", 0) + 1\n    return new_state\n\ndef should_continue(state) -> Literal[\"continue\", \"give_up\"]:\n    if state[\"retries\"] >= MAX_RETRIES:\n        return \"give_up\"\n    if success_predicate(state):\n        return \"end\"\n    return \"continue\"\n```\n\n### Step 3 · Add the stagnation detector\n\nThe counter alone wastes (N-1) iterations on identical work. Add a\nprogress witness comparison:\n\n```python\ndef should_continue(state):\n    if state.get(\"last_witness\") == state.get(\"witness\"):\n        return \"give_up_stagnant\"\n    if state[\"retries\"] >= MAX_RETRIES:\n        return \"give_up_budget\"\n    if success_predicate(state):\n        return \"end\"\n    return \"continue\"\n```\n\nStagnation signals worth detecting:\n\n- Same `last_error` two iterations running.\n- Same `tool_calls` hash (same tool, same args) two iterations.\n- `plan_revision` did not increment.\n- `failing_tests` did not shrink (test-fix loop).\n\nWhen stagnation fires, **always escalate** — don't retry.\n\n### Step 4 · Pick the LM's view of the loop state\n\nThe LM must see the loop counter and the last error / last witness. If\nit doesn't, it will happily repeat. Concretely:\n\n- **LangGraph**: include `retries` and `last_error` in the messages\n  passed to the LLM node — or render them into the system prompt at\n  each iteration.\n- **CrewAI**: surface the previous task's failure in the next task's\n  `context=[...]`, not in `Crew.memory` (which is muddier).\n- **Claude / OpenAI SDKs**: in the next user/tool-result message,\n  include `\"Attempt {n} of {N}. Previous error: {err}. If you cannot\n  fix it on this attempt, return final_answer with status=failed.\"`\n\nWithout this, the LM thinks it's on iteration 1 forever. The framework's\ncounter is in your code; the *behavioural* counter must be in the prompt.\n\n### Step 5 · Build the escalation branch *before* you remove the safety net\n\nThe framework's safety net exists for a reason — runaway billing. Don't\ndisable it. Instead, build the **graceful give-up** that catches the\ncounter/stagnation exit:\n\n- LangGraph: a `give_up` node that calls `interrupt({\"reason\": ...})`,\n  preserving the full state for a human or outer agent to inspect.\n- CrewAI: a fallback `Task` with `human_input=True` that fires when the\n  main task fails the validation in `expected_output`.\n- Claude SDK: a `human_escalation` tool that the model is *forced* to\n  call when `attempt == N`.\n- OpenAI Agents: handle `Run.status == \"incomplete\"` /\n  `incomplete_reason == \"max_turns\"` in the caller and surface to user.\n\nRule of thumb: **a loop without a give-up branch is a loop that fails to\na stack trace**. That's not graceful.\n\n### Step 6 · Layer the outer safety bound\n\nEven with counter + witness + escalation, each individual iteration can\nbe expensive (one tool call doing a 200k-token web search). Add:\n\n- **Token budget**: sum input + output tokens across iterations; cap.\n- **Wall-clock timeout**: `asyncio.wait_for(loop, timeout=T)` at the\n  outermost caller.\n- **Rate limit**: CrewAI `max_rpm`, OpenAI tier limits, Anthropic\n  `requests_per_minute`. Hit these *before* you hit the model's\n  rate-limit error which adds backoff + more retries.\n\nThese are not redundant with the counter — they're orthogonal axes. A\n3-iteration loop where one iteration runs for 4 hours still wedges your\nsystem.\n\n### Step 7 · Test the bound\n\nWrite a regression test that *injects a permanent failure* and asserts:\n\n- The loop exits within N iterations (counter works).\n- The loop exits *before* N if the same error recurs (stagnation works).\n- The final state is captured in the escalation branch (escalation works).\n- The trace shows the counter visible (observability works).\n\nThis is the same shape as `[gh/6731]`'s recommended fix: \"Add a\nregression test that injects a permanent SQL error and asserts the graph\nterminates within 3 iterations.\" Steal that pattern.\n\n---\n\n## 4. 操作模型 (Operation Models)\n\nEach operation is a primitive a coder agent can invoke. Format:\n**Trigger → Action → Output → Evidence**.\n\n### OP-1 · Retry counter in state (the foundational operation)\n\n- **Trigger**: Any cyclic LM workflow exists — even one cycle.\n- **Action**: Add a typed integer field to the workflow's persistent state\n  with a monotonic semantics (LangGraph: `Annotated[int, operator.add]`;\n  CrewAI: shared dict in `Crew` context or `Flow` state; Claude SDK: app\n  variable). Increment inside the loop body; check in the exit predicate.\n- **Output**: A deterministic upper bound on iterations regardless of LM\n  behaviour.\n- **Evidence**: `[gh/6731]` (text-to-SQL fix), `[lc-docs/errors]`\n  (\"explicit termination conditions are the right answer\").\n\n### OP-2 · Stagnation detection via progress witness\n\n- **Trigger**: The counter alone keeps firing — you're wasting N\n  iterations on identical retries.\n- **Action**: Store the previous iteration's distinguishing artifact\n  (`last_error`, `last_tool_call_hash`, `last_witness`) in state. In the\n  exit predicate, compare current vs previous *before* checking the\n  counter. Same → exit `stagnant`, don't increment.\n- **Output**: Fast-fail on truly stuck loops; reserves the counter for\n  cases where iterations *are* progressing slowly.\n- **Evidence**: `[gh/6731]` (LLM was retrying identical query 20 times —\n  stagnation would have fired after 1).\n\n### OP-3 · Human / outer-loop escalation\n\n- **Trigger**: Counter exhausted OR stagnation detected.\n- **Action**: Route to a designated escalation node that *does not crash*.\n  LangGraph: `give_up` node → `interrupt({\"reason\", \"state\"})`. CrewAI:\n  fallback `Task(human_input=True)` or Flow `@listen(\"failed\")` branch.\n  Claude SDK: `human_escalation` tool injection. OpenAI Agents: catch\n  `incomplete_reason == \"max_turns\"` in caller.\n- **Output**: A clean give-up branch carrying enough state for a human\n  or outer agent to diagnose.\n- **Evidence**: `[lc-blog/interrupt]` four-pattern table; Aider's REPL\n  return-to-human on failed test.\n\n### OP-4 · Token & wall-clock budget (orthogonal safety net)\n\n- **Trigger**: Iterations are themselves expensive (long context, web\n  search, code execution).\n- **Action**: Track cumulative tokens in state; enforce wall-clock\n  `asyncio.wait_for` at outer caller; enforce per-model\n  `requests_per_minute`. Any one tripping → escalation (OP-3).\n- **Output**: Defense-in-depth — neither \"3 cheap iterations\" nor \"1\n  expensive iteration\" can run away.\n- **Evidence**: Anthropic Computer Use \"step budget\"; CrewAI `max_rpm`;\n  OpenAI `max_completion_tokens` / `max_prompt_tokens` on `Run`.\n\n### OP-5 · Progress witness declared up front\n\n- **Trigger**: Designing a new cyclic flow.\n- **Action**: Identify the state field that MUST change between\n  iterations to prove progress. Declare it as a typed field. The exit\n  predicate verifies it changed; the LM's prompt is told to set it.\n- **Output**: A loop body that is structurally incapable of being a\n  no-op.\n- **Evidence**: `[cheatsheet/gotchas]` \"Treat each node like a pure\n  function — return a partial state update\"; LangChain best practices.\n\n### OP-6 · Refuse-to-raise-the-net (diagnostic)\n\n- **Trigger**: User asks \"just raise `recursion_limit` / `max_iter`.\"\n- **Action**: Refuse and diagnose. Walk: (a) is there a state counter?\n  (b) is there a witness? (c) does the LM see the previous error? Add\n  what's missing. Leave the framework default in place — it's a circuit\n  breaker, not a control knob.\n- **Output**: A diagnostic conversation that ends with OP-1+OP-2 instead\n  of papering over the failure.\n- **Evidence**: `[gh/6731]` maintainer marked \"not planned\" — i.e., this\n  is *by design*. `[cheatsheet/gotchas]` \"indicates an underlying design\n  flaw.\"\n\n---\n\n## 5. 困境决策案例 (Dilemma Cases)\n\n### Case 1 · \"Text-to-SQL agent loops 20× to GRAPH_RECURSION_LIMIT\"\n\n- **Source**: `[gh/6731]`\n  (https://github.com/langchain-ai/langgraph/issues/6731), maintainer\n  labelled \"not planned.\"\n\n- **困境**: A team built a text-to-SQL agent on LangGraph 1.0.6. When\n  the Databricks query returned an error, the agent retried the same\n  broken SQL 20 times until the default `recursion_limit=25` fired. It\n  had worked on 0.6.x; the upgrade exposed the missing exit condition.\n\n- **约束**:\n  - Can't pin to 0.6.x — security fixes only in 1.x.\n  - Maintainer won't ship a fix — explicitly \"not planned.\"\n  - Databricks quota is bleeding; business needs the agent live.\n\n- **决策步骤**:\n  1. **Reject** \"just raise `recursion_limit` to 100.\" That makes the\n     bleeding worse and confirms the cheatsheet's diagnosis\n     `[cheatsheet/gotchas]`.\n  2. Add state counter (OP-1):\n     ```python\n     class S(TypedDict):\n         messages: Annotated[list[AnyMessage], add_messages]\n         retries: Annotated[int, operator.add]\n         last_error: str | None\n     ```\n  3. Add stagnation detector (OP-2): if `state[\"last_error\"]` is the same\n     as the new error, route to give-up — don't burn 2 more attempts on\n     the identical broken query.\n  4. Surface `last_error` into the next LLM prompt — the *content* of\n     the SQL error usually tells the LLM whether to retry or abandon.\n  5. Add give-up node (OP-3) that uses `interrupt()` to ask the user:\n     \"Tried 3 times, got: {last_error}. Should I rewrite the query\n     differently or stop?\"\n  6. Add regression test (Step 7) injecting a permanent SQL error and\n     asserting termination within 3 iterations.\n\n- **结果**: Loop bounded at 3 iterations. Stagnation typically fires on\n  iteration 2 (same query, same error). Quota cost capped. Failure mode\n  observable in LangSmith. Maintainer-labelled-\"not-planned\" issue\n  becomes a non-issue without an upstream patch.\n\n- **可提取的操作**: OP-1 + OP-2 + OP-3 are mandatory for any cyclic\n  graph. **The LM is never the terminator.**\n\n---\n\n### Case 2 · \"CrewAI delegation ping-pong burns 10× the token budget\"\n\n- **Source**:\n  [github.com/crewAIInc/crewAI/issues/330](https://github.com/crewAIInc/crewAI/issues/330)\n  + related issues #4783, #2606 + azguards.com writeup on\n  [\"the delegation ping-pong\"](https://azguards.com/technical/the-delegation-ping-pong-breaking-infinite-handoff-loops-in-crewai-hierarchical-topologies/).\n\n- **困境**: A team running CrewAI in hierarchical mode set\n  `allow_delegation=True` on all 3 worker agents. Agent A delegated to\n  B; B delegated back to A; A delegated to C; C delegated back to A. The\n  per-agent `max_iter=20` did NOT propagate across the handoffs — every\n  delegation reset the count. Token bill 10× expected; the loop only\n  ended when OpenAI rate-limited them.\n\n- **约束**:\n  - Can't trivially flatten to sequential — the workflow really does\n    need different specialists.\n  - The CrewAI maintainer documentation acknowledges this as a known\n    limitation but recommends \"design your agents not to delegate\n    circularly\" — i.e., the framework's safety net is genuinely\n    bypassed.\n  - Compliance needs an audit trail of which agent ran when.\n\n- **决策步骤**:\n  1. **Reject** \"raise `max_iter` per agent to 100.\" The bug is that\n     `max_iter` doesn't cross handoffs — raising it does nothing\n     `[gh/crewai-330]`.\n  2. **Set `allow_delegation=False` on all worker agents.** Only the\n     manager agent gets delegation. This kills the cycle structurally\n     — the CrewAI canonical advice from `[azguards.com]`.\n  3. **Add a Crew-level handoff counter** in `Crew` context (Flow state\n     if using Flows). Each delegation increments; manager checks before\n     dispatching.\n  4. **Stagnation detector**: hash `(from_agent, to_agent, task_id)` —\n     if the same triple recurs, route to the manager's\n     \"I-can't-resolve-this\" fallback Task with `human_input=True`.\n  5. **Outer timeout**: `asyncio.wait_for(crew.kickoff_async(...),\n     timeout=600)` — wall-clock cap regardless of token budget.\n  6. **Switch to CrewAI Flow + small Crews** if delegation logic really\n     needs to be conditional — `@listen` gives explicit routing\n     `[crewai-docs/flows]`, eliminating the LM-driven handoff.\n\n- **结果**: Cycle eliminated structurally (no worker delegates back).\n  Token bill returns to expected level. Audit trail preserved (the\n  manager owns dispatch; the handoff counter logs each).\n\n- **可提取的操作**:\n  - In multi-agent systems, **counter and witness must live in shared\n    state**, not per-agent config. Per-agent `max_iter` is a useful\n    inner net but not a sufficient outer net.\n  - When the framework's primitive is structurally insufficient (CrewAI\n    `max_iter` not crossing handoffs), the right move is to *remove the\n    primitive's source of failure* (`allow_delegation=False`), not\n    raise its limit.\n\n---\n\n## 6. 反模式与边界 (Anti-patterns & Boundaries)\n\nConcrete don'ts. Each has a real-world example.\n\n- **❌ Raise `recursion_limit` / `max_iter` / `max_turns` to \"fix\" a\n  loop.** This is *the* anti-pattern this skill exists to name. The\n  framework defaults are circuit breakers; raising them moves the\n  failure further away while doubling the cost. Source: `[gh/6731]`\n  maintainer \"not planned\"; `[cheatsheet/gotchas]` \"indicates an\n  underlying design flaw.\"\n\n- **❌ Let the LM vote on termination via reflection.** Calling\n  `llm.invoke(\"are we done?\")` to decide whether to exit recreates the\n  problem one level up — and the answer is biased (\"let me just check\n  one more thing\"). The LM may *suggest* finality (a `final_answer`\n  tool, `stop_reason=\"end_turn\"`); the framework code must verify.\n\n- **❌ Bound only by tokens.** A slow loop with cheap iterations never\n  hits the token cap and runs for hours. Token budget is one of three\n  axes (OP-4), not the whole bound.\n\n- **❌ Per-agent `max_iter` in multi-agent systems with delegation.**\n  CrewAI's `Agent.max_iter` does not propagate across delegation\n  handoffs `[gh/crewai-330]`. The counter must be shared.\n\n- **❌ Bound without an escalation path.** A loop that hits\n  `recursion_limit` and raises is not \"bounded\" in any useful sense —\n  it's \"crashed with stacktrace.\" A bounded loop has a clean give-up\n  branch (OP-3).\n\n- **❌ Counter that the LM cannot see.** If you increment a counter in\n  Python state but never surface \"attempt N of M, previous error: X\" in\n  the LM's prompt, the LM thinks it's on attempt 1 forever and emits\n  the same plan. Counter must be in the *behaviour*, not just the\n  *control plane*.\n\n- **❌ \"Stop when the metric stops improving\" with no patience or cap.**\n  Classic optimiser footgun. The metric can plateau and resume; the\n  loop should be bounded by *both* a max-step count *and* a patience\n  counter — DSPy and W&B sweeps document this. Source: optimiser docs\n  across DSPy / Optuna / W&B.\n\n- **❌ Bound the framework's max_iter but not the outer caller.** If\n  the LM raises an exception inside the loop body and your retry\n  decorator wraps the whole call, the bounded loop becomes an unbounded\n  retry. Bound at every layer the framework gives you.\n\n- **❌ Use `interrupt()` / `human_input=True` only on success.** The\n  give-up branch is the *most important* place for human-in-the-loop —\n  that's where the agent is admitting it's stuck. Routing the failure\n  to a stack trace instead of a human wastes the diagnostic moment.\n\n### Hard boundaries (when this skill does NOT apply)\n\n- One-shot LLM calls (no loop to bound).\n- Data-bounded loops where the cardinality is fixed at design time\n  (\"iterate once per row in this 500-row CSV\"). The bound is the data\n  size; counters/witnesses are over-engineering.\n- Pure-deterministic loops that don't involve LM calls.\n\n---\n\n## 7. 跨框架对照 (Cross-Framework Mapping)\n\nThe same termination contract expressed in each framework's vocabulary.\nUse this table when porting a bounded loop between frameworks — the\n*shape* is identical, only the names change.\n\n| Concept | LangGraph | CrewAI | Claude SDK | OpenAI Agents | DSPy |\n|---|---|---|---|---|---|\n| **Iteration counter** | `state[\"retries\"]: Annotated[int, operator.add]` | shared dict in `Crew` context or `Flow` state | app-side `for i in range(max_iter):` | `Run(max_turns=N)` config | optimiser `max_bootstrapped_demos` |\n| **Exit predicate** | conditional edge function returning `\"END\"` | manager Task validating `expected_output` | `if stop_reason == \"end_turn\": break` | `Run.status == \"completed\"` | metric early-stopping (with explicit patience) |\n| **Safety-net default** | `recursion_limit=25` | `Agent.max_iter=20` + `Crew.max_rpm` | `max_tokens` per call | `max_turns` (no default) | none (dataset size) |\n| **Progress witness** | typed state field updated by node | Task `expected_output` mandates delta | validator on tool output | function schema enforces non-empty delta | metric must strictly improve |\n| **Stagnation signal** | compare `state[\"last_X\"] == state[\"X\"]` in conditional edge | task callback hashes output → stored in context | compare `tool_use` blocks across iterations | compare `tool_calls` in Run steps | patience counter (steps since best) |\n| **Escalation** | `interrupt({\"reason\": ...})` node | fallback Task with `human_input=True` | tool call to human channel | `incomplete_reason == \"max_turns\"` handler | terminate optimiser + log |\n| **Resume after escalation** | `Command(resume=...)` | re-`kickoff()` with appended human input | re-invoke with new user message | `submit_tool_outputs(...)` | re-run with adjusted config |\n| **Outer safety bounds** | wrap `graph.ainvoke` in `asyncio.wait_for` | `Crew.max_rpm` + outer `wait_for` | sum `usage.input_tokens` + `usage.output_tokens` across calls; wall-clock | `Run(max_completion_tokens, max_prompt_tokens)` | `num_threads` budget + wall-clock |\n\n**Translation example.** \"Text-to-SQL agent retries the broken query 20\ntimes\" expressed in three frameworks:\n\n| | LangGraph | CrewAI | Claude SDK |\n|---|---|---|---|\n| Counter | `retries: Annotated[int, operator.add]` in TypedDict | `Crew(memory=False, context={\"retries\": 0})` updated in callback | `attempt = 0` in caller |\n| Increment | LLM node returns `{\"retries\": 1}` | Task `on_complete` callback updates dict | `attempt += 1` after each Messages call |\n| Exit | conditional edge → `END` when `retries >= 3` | manager Task aborts when context counter ≥ 3 | `if attempt >= 3: break` |\n| Witness | `last_error: str` field updated by tool node | `last_error` key in Crew context | track in caller variable |\n| Stagnation | edge function compares `last_error` | callback compares stored vs new | caller compares strings |\n| Escalation | `interrupt({\"err\": state[\"last_error\"]})` node | fallback `Task(human_input=True)` | tool call `human_escalate(err)` |\n| Outer net | `recursion_limit=10` (leave default low) | `Crew(max_rpm=30)` + `wait_for(60s)` | `max_tokens=2048` + `wait_for(60s)` |\n\nThe translation is mechanical because the contract is universal. **That\nis the entire point of this skill.**\n\n---\n\n## 8. 附录 · 引用速查 (Citation Index)\n\nShort tags resolved against `references/R1-source-evidence.md` and\n`references/R2-cross-framework.md`:\n\n- `[gh/6731]` = github.com/langchain-ai/langgraph/issues/6731 — text-to-SQL\n  recursion_limit, maintainer \"not planned\"\n- `[gh/crewai-330]` = github.com/crewAIInc/crewAI/issues/330 — delegation\n  ping-pong; related #4783, #2606\n- `[lc-docs/errors]` = docs.langchain.com/oss/python/langgraph/errors/GRAPH_RECURSION_LIMIT\n- `[lc-blog/interrupt]` = www.langchain.com/blog/making-it-easier-to-build-human-in-the-loop-agents-with-interrupt\n- `[cheatsheet/gotchas]` = sumanmichael.github.io/langgraph-cheatsheet/cheatsheet/faqs-gotchas/\n- `[crewai-docs/agents]` = docs.crewai.com/en/concepts/agents\n- `[crewai-docs/flows]` = docs.crewai.com/en/concepts/flows\n- `[azguards.com]` = azguards.com/technical/the-delegation-ping-pong-breaking-infinite-handoff-loops-in-crewai-hierarchical-topologies/\n- `[anthropic-docs]` = docs.anthropic.com/en/api/messages — Messages API +\n  Agent SDK \"step budget\" pattern\n- `[dspy-docs]` = dspy.ai/docs/building-blocks/optimizers — declared\n  evaluation budget\n- `[openai-agents-docs]` = platform.openai.com/docs/assistants — Run\n  config `max_turns`, `max_completion_tokens`, `max_prompt_tokens`\n\n---\n\n## TL;DR (one-paragraph version for the impatient)\n\nEvery LM loop must carry an **explicit counter in state**, a **progress\nwitness** the loop body must update, a **stagnation detector** that\ncompares the witness across iterations, and a **graceful escalation\nbranch** when either fires. Never raise the framework's\n`recursion_limit` / `max_iter` / `max_turns` to \"fix\" a loop —\nthat limit is a billing safety net, not control flow, and raising it\nmoves the failure further away while burning more tokens. The LM in\nthe loop is **never** a reliable terminator. Source-of-truth case:\nLangGraph issue [#6731](https://github.com/langchain-ai/langgraph/issues/6731)\nmarked \"not planned\" — the framework will not save you; the discipline\nmust.\n\nFile v0.1.3:skills/agentsop-code-execution-decision/SKILL.md\n\n---\nname: agentsop-code-execution-decision\nversion: 0.1.0\ndescription: >-\n  Decision rubric for when an LM agent should write-and-run code (Program-of-Thought / code\n  interpreter) versus reason in natural language: classify each step as deterministic-\n  computable (emit + execute code, feed the result back) vs judgment (stay in prose). Use\n  when designing or debugging an agent step that does arithmetic/parsing/data transforms,\n  when prose reasoning hallucinates a computation (under-coding), or when a sandbox round-\n  trip is wasted on a judgment task (over-coding). Search keywords: code interpreter, agent\n  does math wrong, calculator hallucination, when to run code vs reason, program of thought,\n  PoT, tool vs reasoning.\ndomain: |\n  Deciding when an LM agent should write-and-run code (Program-of-Thought / code interpreter)\n  versus reason in natural language. An enhancement overlay on top of DSPy's ProgramOfThought\n  module, which ships the mechanism but not the decision rubric.\ntype: enhancement-overlay\nsource: |\n  DSPy ProgramOfThought module + docs [dspy.ai/learn/programming/modules/];\n  \"Program of Thoughts\" (Chen et al., arXiv 2211.12588);\n  PAL: Program-aided Language Models (Gao et al., arXiv 2211.10435);\n  OpenAI Code Interpreter / Assistants code-tool docs;\n  Anthropic code-execution tool docs;\n  LangChain PythonREPLTool;\n  sibling skill output-format-by-model (PoT for math/parse content shape).\naudience: |\n  Coder-agent authors, tool-use harness designers, RAG/eval pipeline builders deciding\n  whether a step should emit executable code or stay in prose.\nstatus: tool-skill\n---\n\n# Code-Execution Decision — emit-code-vs-reason\n\n> **One-liner**: LMs are unreliable calculators but reliable coders. When the answer needs\n> determinism and precision — arithmetic, exact data manipulation, deterministic transforms —\n> emit code and run it. When the answer needs judgment, taste, or open-ended synthesis, reason\n> in natural language. The cost of getting this gate wrong is silent: prose arithmetic\n> *hallucinates a plausible-looking wrong number*, and over-coding a judgment task burns a\n> sandbox round-trip for nothing.\n\nThis is an **enhancement overlay**. DSPy already gives you `dspy.ProgramOfThought` (PoT) —\nthe *mechanism* for write-then-execute. What it does not give you is the *decision rubric* for\nwhen to reach for it. That rubric is this skill. Cross-link the sibling\n[[agentsop-output-format-by-model]] (which decides *how* code-shaped content should be serialized) and\n[[agentsop-test-fix-loop]] (which closes the execute → error → retry loop).\n\n---\n\n## 1. 何时激活 (When to activate)\n\nActivate this skill **before** committing a step to a reasoning strategy whenever the task has a\n**verifiable, deterministic core** — or whenever you catch an agent doing arithmetic in prose.\n\n| Trigger | Signal |\n|---|---|\n| Arithmetic / math | \"compute the compound interest\", \"what's 17.5% of $4,392.18\", multi-step word problems, unit conversions, date deltas |\n| Precise data manipulation | \"sort these 240 rows by the third column\", \"dedupe and count\", \"join these two lists on id\", \"parse this CSV and sum column B\" |\n| Deterministic transforms | regex extraction, string reformatting, base conversion, hashing, sorting, set operations |\n| Symbolic / combinatorial | \"how many distinct permutations\", \"solve this system of equations\", calendar/scheduling math |\n| You see a model doing math in prose | \"Let me add: 1,204 + 8,991 + ... = 10,195\" — almost always worth a code check |\n| Choosing a DSPy module | deciding between `ChainOfThought` and `ProgramOfThought` for a signature [dspy.ai/learn/programming/modules/] |\n\n**Anti-triggers** (do NOT reach for code execution):\n- The task is **judgment**: tone, summarization, ranking by quality, \"is this reply empathetic?\", design tradeoffs, open-ended explanation.\n- The \"computation\" is trivial and within the model's reliable range (single-digit arithmetic, a 3-item count) — the sandbox round-trip costs more than it saves.\n- No sandbox is available and the determinism requirement is soft.\n- The output's *consumer* is a human reading prose, and an approximate answer is acceptable.\n\n---\n\n## 2. 核心心智模型 (Core mental model)\n\n> **LMs are unreliable calculators but reliable coders.**\n\nA language model predicts the *next token*, not the *correct value*. When you ask it to add\n`48,217 + 9,884` in prose, it emits the most *plausible-looking* digit sequence — which is\nfrequently wrong, and wrong in a way that looks right. The same model can write\n`48217 + 9884` as a Python expression flawlessly, because emitting the *program* is a\npattern-matching task it is genuinely good at, and the *Python interpreter* is a deterministic\noracle. This decoupling — model writes the recipe, interpreter computes the result — is the\nentire thesis of Program-of-Thought (PoT) [arXiv 2211.12588] and PAL [arXiv 2211.10435].\n\n```\n            ┌────────────────────────────────────────────────────────┐\n            │  THE GATE: does this answer need determinism/precision?  │\n            └────────────────────────────────────────────────────────┘\n                       │                                  │\n              YES (computable)                     NO (judgment)\n                       │                                  │\n                       ▼                                  ▼\n            ┌─────────────────────┐            ┌─────────────────────┐\n            │ EMIT CODE           │            │ REASON IN PROSE     │\n            │ model writes recipe │            │ model is the engine │\n            │ interpreter = oracle│            │ no oracle exists    │\n            └─────────────────────┘            └─────────────────────┘\n                       │\n                       ▼\n            sandbox → run → feed result back into LM → LM narrates/uses it\n```\n\n**Two failure modes the gate prevents:**\n\n| Failure | Mechanism | Symptom |\n|---|---|---|\n| **Under-coding** (reason when you should compute) | LM hallucinates a calculation it cannot reliably perform | A confident, wrong, plausible-looking number; off-by-one counts; arithmetic that \"looks\" right |\n| **Over-coding** (compute when you should reason) | LM wraps a judgment task in a sandbox round-trip that adds no determinism | Wasted latency + cost; brittle code that encodes a subjective rubric as if it were a formula; `print(\"the tone is friendly\")` |\n\n**Why the asymmetry matters.** Under-coding fails *silently* — the wrong number propagates\ndownstream and no exception fires. Over-coding fails *loudly and cheaply* — you notice the\nuseless sandbox call. So the default lean, when genuinely uncertain and a verifiable core\nexists, is **toward code**. But \"genuinely uncertain\" is the operative phrase: most judgment\ntasks are not close calls.\n\nThe **format corollary** (from [[agentsop-output-format-by-model]]): once you decide to emit code, emit\nit as *code in a fenced block / single string field* — never nested inside JSON sub-structure.\nCode-in-JSON measurably degrades the code itself (Aider: 61%→20% on GPT-4 Turbo). The\nexecution decision and the serialization decision are two separate gates; pass both.\n\n---\n\n## 3. SOP 工作流 (Decision workflow)\n\nA three-step gate. Run it per *step*, not per *task* — one task can have computable steps and\njudgment steps interleaved.\n\n### Step 1 — Classify the step: deterministic-computable vs judgment\n\nAsk: **\"Is there a single correct answer that a program could verify?\"**\n\n- **Yes → computable.** Arithmetic, sorting, parsing, counting, set/regex ops, symbolic math, anything with a checkable ground truth. → Step 2.\n- **No → judgment.** Quality ranking, tone, summarization, design choice, open synthesis. → reason in prose. Stop here.\n- **Mixed → decompose.** \"Summarize these sales and tell me the total\" = a judgment summary step + a computable total step. Route each independently. The total goes to code; the summary stays prose.\n\nA useful tiebreaker for the \"trivial computable\" gray zone: **if the model would be embarrassed\nto get it wrong and you'd reach for a calculator yourself, emit code.** If you'd do it in your\nhead without a second thought, prose is fine.\n\n### Step 2 — If computable, emit + run code\n\n1. Emit the computation as a real program (Python is the PoT default; the interpreter is the oracle).\n2. Keep the code **minimal and side-effect-free** for the computation — read inputs, compute, `print`/return the result. No network, no filesystem unless the task *is* I/O.\n3. Serialize the code per [[agentsop-output-format-by-model]]: fenced block or single string field, **not** JSON-nested.\n\n### Step 3 — Sandbox → run → feed result back into the LM\n\n1. **Sandbox choice** (see §4 op #3, cross-link [[agentsop-output-format-by-model]] for the wire shape): DSPy `ProgramOfThought` uses a Python interpreter (Deno/PythonInterpreter sandbox in recent versions); OpenAI Code Interpreter runs in a managed container; Anthropic code-execution tool runs in a sandboxed VM; LangChain `PythonREPLTool` runs **in-process and is unsandboxed** — treat as untrusted-input-hostile.\n2. **Run** the code in the sandbox; capture stdout / return value / traceback.\n3. **Feed the result back into the LM** — this is the step under-coders forget. The code produces a *value*; the LM must consume it to narrate, format, or chain into the next step. PoT is \"code computes, LM contextualizes,\" not \"code replaces the LM.\"\n4. **Error → retry** (cross-link [[agentsop-test-fix-loop]]): on traceback, feed the error back to the LM, regenerate the code, re-run. Bound the retries (DSPy `ProgramOfThought` defaults to `max_iters` ≈ 3). After the bound, fall back to prose reasoning or surface the failure — do not loop forever.\n\n**Exit criterion:** the step produces either (a) a code-derived value the LM has consumed, or\n(b) a prose judgment, with the gate decision recorded so a reviewer can audit *why* code was or\nwasn't used.\n\n---\n\n## 4. 操作模型 (Operations: Trigger → Action → Output → Evidence)\n\nSeven operations. Each row is a reusable move.\n\n| # | Op | Trigger | Action | Output | Evidence |\n|---|---|---|---|---|---|\n| 1 | **Computable-vs-judgment gate** | Any step about to be reasoned | Apply §3 Step 1: single verifiable answer? | Route to code or prose | PoT premise: decouple compute from reasoning [arXiv 2211.12588] |\n| 2 | **Decompose mixed steps** | Step has both a number and a narrative | Split into computable sub-step (code) + judgment sub-step (prose) | Two routed sub-steps | Mirrors mixed-content split in [[agentsop-output-format-by-model]] §5 Case B |\n| 3 | **Sandbox choice** | Decided to emit code | Pick interpreter by trust + capability: DSPy PoT (Python sandbox), OpenAI Code Interpreter (managed container), Anthropic code-exec (VM), LangChain `PythonREPLTool` (**unsandboxed, in-process**) | Chosen runtime | DSPy modules [dspy.ai/learn/programming/modules/]; LangChain PythonREPLTool docs |\n| 4 | **Result-back-into-LM** | Code produced a value | Inject stdout/return value into the next LM turn so the model narrates/uses it | LM-consumed result | PoT design: code computes, LM contextualizes [arXiv 2211.12588] |\n| 5 | **Error retry (bounded)** | Code raised a traceback | Feed error to LM, regenerate, re-run; cap at `max_iters` (~3) then fall back | Fixed code or graceful fallback | DSPy `ProgramOfThought` `max_iters`; see [[agentsop-test-fix-loop]] |\n| 6 | **Precision escalation** | Prose answer involves multi-step arithmetic | Re-route the arithmetic to code even if prose started it | Code-verified number | \"LMs are unreliable calculators\" — PAL [arXiv 2211.10435] |\n| 7 | **Over-coding veto** | About to sandbox a judgment task | Stop: no deterministic core → reasoning, not code | Prose reasoning, no sandbox call | §5 Case B; avoids wasted round-trip |\n\nIn DSPy terms, op #1 is exactly the choice between `dspy.ChainOfThought` (prose reasoning) and\n`dspy.ProgramOfThought` (emit+run) for a signature — the dspy skill lists the modules but this\noverlay supplies the *when*.\n\n---\n\n## 5. 困境决策案例 (Dilemma cases)\n\n### Case A — Arithmetic in prose hallucinates → PoT fixes it (under-coding)\n\n**Trigger**: A finance-summary agent step: \"Given these 14 line items, compute the total,\nthe 8.25% tax, and the grand total.\" The agent is a `dspy.ChainOfThought` module emitting prose.\n\n**Constraints**:\n- 14 multi-digit addends + a percentage + a sum-of-sums. Well outside reliable mental-math range.\n- The number flows into an invoice — wrong is *expensive* and *silent*.\n- A Python sandbox is available.\n\n**Decision steps**:\n1. **Gate (§3 Step 1):** \"Is there a single verifiable answer?\" Yes — the total is a fact, not a judgment. → computable.\n2. **Recognize the under-coding failure mode.** Prose chain-of-thought will emit a plausible\n   running sum that is frequently off by some digits, and *no exception will fire*. This is the\n   silent failure the gate exists to prevent.\n3. **Switch the module from `ChainOfThought` to `ProgramOfThought`.** The model now emits\n   `subtotal = sum([...]); tax = round(subtotal * 0.0825, 2); total = subtotal + tax` and the\n   interpreter computes it exactly.\n4. **Feed the result back (op #4):** the LM receives `total = 4,217.93` and narrates the invoice\n   line. Code computed; LM contextualized.\n5. **Serialize per [[agentsop-output-format-by-model]]:** the generated code goes in a fenced block, not a\n   JSON `program` field — code-in-JSON would degrade it.\n\n**Outcome**: The arithmetic is now deterministic and auditable. PoT-style code execution is the\ndocumented fix for exactly this class of error [arXiv 2211.12588, arXiv 2211.10435].\n\n**Extractable operation**: **Multi-step arithmetic in prose is a smell. Re-route it to code (op #6).**\n\n---\n\n### Case B — Over-coding a judgment task wastes a sandbox round-trip (over-coding)\n\n**Trigger**: A support-triage agent step: \"Read this customer message and decide whether the\ntone is hostile, neutral, or warm.\" An over-eager engineer wires it through `ProgramOfThought`\nbecause \"code is more reliable.\"\n\n**Constraints**:\n- \"Tone\" has no single verifiable answer — it is a judgment.\n- The sandbox round-trip adds latency + cost.\n- Forcing it into code produces something like `if \"!!!\" in msg: tone = \"hostile\"` — a brittle\n  rule that *encodes a subjective rubric as if it were a formula*, and is worse than the model's\n  native judgment.\n\n**Decision steps**:\n1. **Gate (§3 Step 1):** \"Single verifiable answer a program could check?\" No — tone is\n   judgment. → reason in prose. Stop.\n2. **Invoke the over-coding veto (op #7).** There is no deterministic core to delegate to an\n   oracle. Code adds *no determinism* — it just relocates the same judgment into worse,\n   hard-coded heuristics.\n3. **Cost it out.** The sandbox call adds a full round-trip (and a code-gen + parse step) for\n   zero precision gain. This is pure waste — the loud, cheap failure mode.\n4. **Keep it as `ChainOfThought`** (or `Predict`). The LM *is* the right engine for judgment.\n5. **Counter-check:** if a later spec adds \"and count how many exclamation marks\" — *that*\n   sub-step is computable; decompose (op #2) and route the count to code, the tone to prose.\n\n**Outcome**: No sandbox call. The judgment stays where judgment belongs. Over-coding is a real\nand common anti-pattern: not everything benefits from an interpreter, only things with a\nverifiable deterministic core.\n\n**Extractable operation**: **No verifiable answer → no code. Veto the sandbox round-trip (op #7).**\n\n---\n\n### Case C — Mixed step: summary + total (decompose)\n\n**Trigger**: \"Summarize this quarter's sales narrative and give me the exact total revenue.\"\n\n**Constraints**: One sentence contains both a judgment (summary) and a computation (total).\n\n**Decision steps**:\n1. **Gate → mixed.** Decompose (op #2).\n2. Route the **total** to `ProgramOfThought`: code sums the figures, interpreter verifies.\n3. Route the **summary** to `ChainOfThought`: prose synthesis, no oracle exists.\n4. **Feed the code result back** into the prose step so the narrative cites the exact, verified total.\n\n**Outcome**: Each sub-step uses its correct engine. This is the execution-decision analogue of\nthe mixed-content two-pass pattern in [[agentsop-output-format-by-model]] §5 Case B.\n\n**Extractable operation**: **One task ≠ one strategy. Gate per step, decompose mixed steps.**\n\n---\n\n## 6. 反模式与边界 (Anti-patterns & boundaries)\n\n### Anti-patterns\n\n1. **Code for everything (\"code is always more reliable\").** False. Code is more reliable only\n   where a *deterministic, verifiable* answer exists. For judgment tasks, code just hard-codes a\n   subjective rubric and adds a wasted round-trip (Case B). The interpreter is an oracle only for\n   computable questions.\n2. **Reasoning for arithmetic (\"the model can just add it\").** The headline failure. LMs are\n   unreliable calculators; multi-step prose arithmetic hallucinates plausible wrong numbers,\n   *silently* (Case A). Any non-trivial computation → code [arXiv 2211.12588].\n3. **Emitting code but never feeding the result back.** PoT is \"code computes, LM contextualizes.\"\n   If the value never re-enters the LM turn, you have a dangling computation the agent can't use\n   or narrate (op #4).\n4. **Unbounded error-retry loops.** On a traceback, regenerate-and-rerun is correct — but cap it\n   (`max_iters` ≈ 3) and fall back to prose or surface the failure. Infinite code-repair loops\n   burn cost. See [[agentsop-test-fix-loop]].\n5. **Nesting generated code inside JSON sub-structure.** The execution decision says *emit code*;\n   [[agentsop-output-format-by-model]] says emit it as a fenced block / single string, never as nested\n   JSON — code-in-JSON degrades the code (Aider 61%→20%). Pass both gates.\n6. **Treating LangChain `PythonREPLTool` as a safe sandbox.** It runs **in-process, unsandboxed**.\n   Fine for trusted self-authored code; hostile to untrusted input. Use a real sandbox\n   (OpenAI Code Interpreter container, Anthropic code-exec VM, DSPy's interpreter) when inputs are untrusted.\n7. **Over-decomposing trivial computation.** A 3-item count or single-digit add does not need a\n   sandbox round-trip; the tax exceeds the benefit. The gate is for *non-trivial* computation.\n\n### Boundaries (when this gate doesn't fire)\n\n- **No sandbox available and determinism is soft.** Sometimes you must reason in prose because\n  there is no interpreter; accept the precision risk and flag it.\n- **The model is overwhelmingly capable for a tiny computation.** Frontier models reliably handle\n  small arithmetic; the round-trip isn't worth it below a complexity threshold.\n- **The task is one-shot exploration** with no downstream consumer of the precise value.\n- **Latency-critical paths** where a sandbox round-trip violates a hard SLA and an approximate\n  answer is acceptable — a deliberate, documented tradeoff, not a default.\n\n---\n\n## 7. 跨框架对照 (Ecosystem cross-reference)\n\nHow major frameworks expose the emit-code-vs-reason mechanism. This skill operates at the\n**decision** layer; each framework supplies the **mechanism**.\n\n### DSPy `ProgramOfThought` [dspy.ai/learn/programming/modules/]\n\nThe canonical declarative version. A signature compiled with `dspy.ProgramOfThought(Sig)`\nmakes the LM emit Python, runs it in an interpreter sandbox, and feeds the result back —\nwith bounded retry on error (`max_iters`). The sibling **dspy skill** lists `ChainOfThought`\nvs `ProgramOfThought` as module choices but does **not** give the decision rubric; *this overlay\nis that rubric*. Use `ProgramOfThought` exactly when §3 Step 1 returns \"computable.\"\n\n### OpenAI Code Interpreter / Assistants code tool\n\nA managed container that the model can write Python into and execute, with files and state\npersisted across turns. Heavier and stateful — good for data-analysis sessions (load CSV,\ncompute, plot). Same gate applies: route computable steps in, keep judgment in chat.\n\n### Anthropic code-execution tool\n\nA sandboxed VM exposed as a tool; the model emits code, it runs, results return to the\nconversation. First-party, sandboxed — safe for untrusted inputs. The execution-decision gate\nmaps directly: offer the tool, but the *model/agent* should only invoke it for computable steps.\n\n### LangChain `PythonREPLTool`\n\nA tool wrapping a Python REPL. **Runs in-process and is unsandboxed** — powerful and dangerous.\nUse only for trusted, self-authored computation; never expose it to untrusted input without an\nexternal sandbox. The decision rubric is identical; the *safety* profile is worst-in-class.\n\n### Cross-framework summary\n\n```\nFramework            | Mechanism                | Sandbox     | Result-back\n---------------------|--------------------------|-------------|------------\nDSPy PoT             | ProgramOfThought module  | Python intp | automatic (max_iters)\nOpenAI Code Interp.  | Assistants code tool     | container   | persisted state\nAnthropic code-exec  | code-execution tool      | VM (safe)   | into conversation\nLangChain            | PythonREPLTool           | NONE (proc) | manual wiring\n```\n\nEvery framework can be *mis-invoked* — pointed at a judgment task (over-code) or skipped for a\ncomputation (under-code). This overlay is the gate that decides invocation, regardless of which\nmechanism is underneath.\n\n---\n\n## Quick reference card\n\n```\n┌──────────────────────────────────────────────────────────────────────┐\n│                  EMIT-CODE-VS-REASON DECISION CARD                   │\n├──────────────────────────────────────────────────────────────────────┤\n│ Single verifiable answer a program could check?                      │\n│   YES → EMIT CODE → sandbox → run → feed result back into LM         │\n│   NO  → REASON IN PROSE (no oracle exists for judgment)              │\n│   MIXED → decompose; route each sub-step independently               │\n├──────────────────────────────────────────────────────────────────────┤\n│ Arithmetic / parse / sort / count / regex / symbolic → CODE          │\n│ Tone / quality / summary / design / synthesis        → PROSE         │\n│ \"summary AND total\"                                  → SPLIT         │\n├──────────────────────────────────────────────────────────────────────┤\n│ LMs are unreliable calculators but reliable coders.                  │\n│ Under-coding fails SILENTLY (wrong plausible number).                │\n│ Over-coding fails LOUDLY+CHEAPLY (wasted sandbox round-trip).        │\n│ When genuinely uncertain AND a verifiable core exists → lean CODE.   │\n├──────────────────────────────────────────────────────────────────────┤\n│ NEVER:                                                               │\n│  • code a judgment task (over-coding veto)                          │\n│  • do multi-step arithmetic in prose (under-coding)                 │\n│  • forget to feed the code result back into the LM                  │\n│  • nest generated code in JSON (see [[agentsop-output-format-by-model]])     │\n│  • loop code-repair unbounded (see [[agentsop-test-fix-loop]])               │\n│  • trust LangChain PythonREPLTool on untrusted input (unsandboxed)  │\n└──────────────────────────────────────────────────────────────────────┘\n```\n\n---\n\n## 引用源 (Citations)\n\n**Primary anchors:**\n- Program of Thoughts (PoT) — Chen et al., arXiv 2211.12588 — decouple computation from reasoning; LM writes code, interpreter computes.\n- PAL: Program-aided Language Models — Gao et al., arXiv 2211.10435 — LM as coder, runtime as deterministic solver.\n- DSPy `ProgramOfThought` module — [dspy.ai/learn/programming/modules/] — emit+run+retry mechanism.\n\n**Framework / API docs:**\n- OpenAI Code Interpreter / Assistants code tool — [platform.openai.com/docs/assistants/tools/code-interpreter]\n- Anthropic code-execution tool — [docs.anthropic.com/en/docs/agents-and-tools/tool-use/code-execution-tool]\n- LangChain `PythonREPLTool` — [python.langchain.com/docs/integrations/tools/python] (unsandboxed; in-process).\n\n**Companion / overlaid skills:**\n- `dspy-sop-skill/SKILL.md` — ships `ProgramOfThought` but not this decision rubric (the gap this overlay fills).\n- `d-output-format-by-model-skill/SKILL.md` — sibling: once you emit code, how to serialize it (PoT for math/parse; code never nested in JSON). Cross-linked as [[agentsop-output-format-by-model]].\n- `test-fix-loop` — the execute → error → retry loop this skill defers to for bounded code repair. Cross-linked as [[agentsop-test-fix-loop]].\n\nFile v0.1.3:skills/agentsop-context-scope-discipline/SKILL.md\n\n---\nname: agentsop-context-scope-discipline\nversion: 0.1.0\ndescription: >-\n  Coder-agent working-file budget discipline: keep the editable working set (files you /add\n  into writable context) under ~25k tokens, separate \"read\" from \"edit\", delegate breadth to\n  a read-only repo-map, and drop files once edited. Use when an LLM coder-agent edits\n  multiple files, when the working set must stay focused, or when the model starts editing\n  the wrong file / missing targets because too much context dilutes attention. Search\n  keywords: context window full, agent edits wrong file, too much context, /add /drop files,\n  working file budget, context dilution, lost in the middle.\ndomain: working-file budget management for LLM coder-agents (multi-file editing)\nsource: aider.chat troubleshooting/edit-errors (25k distraction threshold) + /add /drop discipline; generalized across coder harnesses\naudience: coder-agents (Aider/Claude Code/Cursor/Cline/custom) editing multiple files where the working set must stay focused\nstatus: enhancement overlay — sharpens the generic token-budget rule into a coding-agent-specific working-file discipline\ntype: enhance\noverlays: token-budget skills (this adds the coder-agent \"only load what you'll edit\" rule)\ncrosslinks: \"[[agentsop-repo-map]], [[agentsop-session-state-hygiene]]\"\n---\n\n# Context Scope Discipline — 只把你要改的文件放进工作集\n\n> 一句话：**编辑代码时，工作文件预算（你 `/add`-ed 进可写上下文的文件）要压在 ~25k tokens 以内**。超过这个量，\"more context ≠ better edits\"——模型注意力被稀释，开始改错文件、漏看你刚加进去的目标。广度交给 [[agentsop-repo-map]]（只读签名地图），深度只留给\"这次真要编辑\"的那几个文件。\n\n这是一个**增强叠加技能（enhance overlay）**。它不替代任何\"通用 token 预算\"建议，而是把那条泛泛的\"少塞上下文\"打磨成一条 coder-agent 专属的硬规则：**区分\"读\"与\"改\"，只把\"改\"的文件加进工作集**。借用 Aider 的实测阈值——\n\n> \"Above about 25k tokens of context, most models start to become distracted.\" [aider.chat/docs/troubleshooting/edit-errors.html]\n\n---\n\n## 1. 何时激活本技能\n\n下列任一情形成立时，把\"工作文件预算纪律\"作为该编辑会话的标准约束：\n\n- 任务是**多文件编辑**：rename、抽函数、改 API 签名、加 hook 点——你需要理解 N 个文件，但只会真正修改其中一小部分。\n- **agent 正在改错文件**：给出的 diff 落在你没想改的文件上，或编造了不存在的路径。这几乎总是\"工作集不对\"——目标没加进去，或加了太多无关文件把模型呛晕。\n- **上下文窗口在涨**：`/tokens`（或等价物）逼近 25k；响应被截断；长会话里模型\"记住了错的东西\"。\n- 你在大仓库里工作，凭\"为了保险全加进去\"的本能正在把整个目录、整个 repo 灌进可写上下文。\n- 你在写**自建 coder harness**，需要一条明确的\"可编辑文件白名单何时收/何时放\"的规则。\n\n**不应激活的反面信号**：单文件已知的小改动（工作集天然就是 1）；纯讨论/架构问答（用只读上下文 + [[agentsop-repo-map]] 即可，不进工作集）；非编辑任务。\n\n---\n\n## 2. 核心心智模型\n\n### 2.1 一句话铁律\n\n> **more context ≠ better edits.** 过了约 ~25k tokens 的文件量，模型就开始失焦——**只把你这一轮真要编辑的文件加进工作集，其余的靠 [[agentsop-repo-map]] 顶上。**\n\n### 2.2 \"读\" vs \"改\"是两种不同的上下文，需要两种不同的预算\n\nLLM 看到的编辑上下文分三层，**优先级与写权限递减**：\n\n| 层 | 内容 | 写权限 | 预算策略 |\n|---|---|---|---|\n| 系统提示 + 编辑格式 | harness 固化 | harness | 不可控 |\n| **只读上下文** | [[agentsop-repo-map]] 签名地图 + `/read` 的参考文件 + CONVENTIONS.md | 人/agent 配置 | 给\"广度\"——用地图覆盖全仓，但只放签名不放函数体 |\n| **工作集（写集合）** | `/add`-ed 的文件 | LLM **唯一**能编辑的 | 给\"深度\"——只放这次真要改的，压在 ~25k 以内 |\n\n> **核心区分**：repo-map 给\"哪儿\"（breadth，签名级，便宜），工作集给\"怎么改\"（depth，全文级，贵）。把这两种需求混进同一个篮子（\"全 `/add` 进来再说\"）是本技能要根除的反模式。\n\n### 2.3 25k 是稀释阈，不是上限\n\n25k 不是\"塞到 25k 就崩\"，而是\"过了 25k 编辑准确率开始断崖式下降\"。它是个**信号阈**：\n\n- 工作集本身 + 对话历史 + repo-map 都算进这一份预算。\n- 文件越多、越大，留给\"模型对当前编辑点的注意力\"越少。\n- 模型越弱，对 25k 越敏感（弱模型\"更容易违背系统提示\" [aider.chat/docs/troubleshooting/edit-errors.html]）。\n\n### 2.4 \"全加进去保险\"是错觉——repo-map 已经替你覆盖了广度\n\n凭直觉，\"我要理解这 10 个文件才能改对，那就全 `/add`\"。实测相反：\n\n> Aider **只靠 repo-map**（不把文件加进工作集）在 SWE-Bench Lite 上仍 **70.3%** 命中正确文件 [aider.chat/2024/05/22/swe-bench-lite.html]。\n\n即\"找文件\"这件事不需要把文件灌进工作集——只读地图就够了。工作集只为\"编辑\"存在。把这两件事拆开，是省预算的关键。详见 [[agentsop-repo-map]]。\n\n### 2.5 动态预算：工作集涨，地图就该缩\n\n预算是一份蛋糕，不是各自独立的盘子。`/add` 了正确文件后，[[agentsop-repo-map]] 应自动缩小（\"adjusts ... based on the state of the chat\" [aider.chat/docs/repomap.html]），把 token 让给真代码。如果你的 harness 不会自动缩地图，编辑期就手动 `--map-tokens` 调小或归零。\n\n### 2.6 与 [[agentsop-session-state-hygiene]] 的分工\n\n本技能管**文件维度**（工作集里有哪些文件）；[[agentsop-session-state-hygiene]] 管**历史维度**（对话历史是否污染当前任务）。二者共用同一份 25k 预算：\n\n- 预算超了，先 `/drop` 不再需要的文件（本技能）；\n- 仍然超 / 话题已切换，再 `/clear` 清历史（[[agentsop-session-state-hygiene]]）。\n- `/drop` 保历史去文件；`/clear` 保文件去历史；`/reset` 两者都丢。\n\n---\n\n## 3. SOP 工作流\n\n### Phase 1 — 区分\"要编辑\"和\"只要读懂\"\n\n任务进来，第一步不是 `/add`，而是分类。对每个相关文件问一句：**\"这一轮我会修改它的字节吗？\"**\n\n```\n会改它的字节        → 候选写集合（稍后 /add）\n只需理解它的契约     → 只读：/read，或干脆只靠 repo-map 的签名\n不确定改哪些        → 先不加任何文件，进 Phase 2 让 repo-map 帮你定位\n```\n\n> 经验法则：写集合目标 **≤ 5 个文件**。超过，多半是任务没拆够。\n\n### Phase 2 — 不知道改哪个？让 repo-map 定位，而不是全加进来\n\n```\n> /ask which files implement <feature>?\n< [模型基于只读 repo-map 回答候选文件]\n```\n\n模型命名出目标后，**你**再决定把哪些加进工作集（Op `locate-then-add`）。\"找文件\"和\"改文件\"永远两步走——这是 [[agentsop-repo-map]] 与本技能共享的设计哲学。\n\n### Phase 3 — 只 `/add` 你会编辑的，参考文件用 `/read`\n\n```\n/add  src/auth.py tests/test_auth.py     # 这两个会改 → 进写集合\n/read src/config.py docs/auth.md         # 只参考，不改 → 只读\n```\n\n铁律重申：**少 `/add`，敢 `/drop`**。\"为了保险全加\"恰恰是让模型改错文件的主因。\n\n### Phase 4 — 编辑过程中持续盯预算\n\n```\n/tokens                  # 看当前占用；接近 25k 是黄灯\n```\n\n| 信号 | 动作 |\n|---|---|\n| `/tokens` 逼近 25k | `/drop` 已经改完、不再相关的文件 |\n| repo-map 占比偏大 | 调小 `--map-tokens`（目标文件已定，地图可缩） |\n| 模型反复改错文件 | `/ls` 检查工作集；`/drop` 多余的，`/add` 缺的 |\n| 历史漂移（不是文件问题） | 转交 [[agentsop-session-state-hygiene]]：`/clear` |\n\n### Phase 5 — 一个文件改完就 `/drop` 它\n\n工作集不是\"会话期一直累积\"的。某文件这一轮的修改告一段落、后续子任务不再碰它——立即 `/drop`。把腾出的预算还给下一批要改的文件。这是把工作集**当滑动窗口**用，而不是当垃圾堆。\n\n### Phase 6 — 任务太大撑不住时，拆，而不是塞\n\n地图也缩了、能 `/drop` 的都 `/drop` 了，预算还是破 25k？这是**任务太宽**的信号，不是预算的问题：\n\n```\n1. 进子目录 + --subtree-only（缩小 repo-map 范围，见 [[agentsop-repo-map]] §3）\n2. 拆任务：大需求拆成多个收敛子目标，每个子目标一个会话\n3. 每个新会话只带它真正要改的那 ≤5 个文件\n```\n\n---\n\n## 4. 操作模型\n\n每条给 **Trigger / Action / Output / Evidence**。命令名以 Aider 为参考，行为框架无关。\n\n### Op 1 — `classify(file)` 区分\"读\"与\"改\"\n\n- **Trigger**：任何相关文件进入视野。\n- **Action**：问\"这一轮会修改它的字节吗？\"。会改 → 写集合候选；只读懂 → 只读层（`/read` 或仅 repo-map 签名）。\n- **Output**：每个文件被标记为 EDIT / READ-ONLY / NAVIGATE-ONLY 三类之一。\n- **Evidence**：LLM 只能编辑工作集里的文件 [aider.chat/docs/more/edit-formats.html]；只读 vs 读写是 Aider 的安全边界。\n\n### Op 2 — `locate-then-add(task)` 先定位再加\n\n- **Trigger**：写集合未定，不知道改哪个文件。\n- **Action**：把 task + repo-map 喂给 LLM，让它**只命名**候选文件（不直接编辑）；人/agent 再 `/add` 命名出的目标。\n- **Output**：≤5 个写集合文件 + ≤3 个只读参考。\n- **Evidence**：repo-map 只读即可达 **70.3%** 文件命中 [aider.chat/2024/05/22/swe-bench-lite.html]——定位不需要进工作集。\n\n### Op 3 — `add(files)` 加入工作集（克制）\n\n- **Trigger**：某文件确定这一轮会被编辑。\n- **Action**：`/add` 只加该文件。不加\"可能会用到\"的、不加整目录。\n- **Output**：工作集 +1。\n- **Evidence**：模型改错文件\"几乎总是因为该文件没 `/add` 或你 `/add` 了太多无关文件\" [aider.chat/docs/usage].\n\n### Op 4 — `read(files)` 加为只读参考\n\n- **Trigger**：文件需要被理解但不会被改（schema、config、CONVENTIONS）。\n- **Action**：`/read`，进只读层，LLM 不能编辑。\n- **Output**：只读上下文 +1，写集合不变。\n- **Evidence**：Aider `--read` / `/read`；只读层与读写层分离 [aider.chat/docs/usage/conventions.html]。\n\n### Op 5 — `budget_watch()` 监控预算\n\n- **Trigger**：每次编辑回合开始，或感觉模型变笨时。\n- **Action**：`/tokens` 看占用分布（工作集 / 历史 / 地图各占多少）。\n- **Output**：是否越过 25k 黄灯的判断。\n- **Evidence**：25k 稀释阈 [aider.chat/docs/troubleshooting/edit-errors.html]。\n\n### Op 6 — `drop(files)` 改完即释放\n\n- **Trigger**：某文件这一轮修改完成、后续不再碰；或预算逼近 25k。\n- **Action**：`/drop` 该文件，腾出预算。\n- **Output**：工作集 -1，地图自动回涨补位。\n- **Evidence**：`/drop` 是常态操作而非应急；动态预算 [aider.chat/docs/repomap.html]。\n\n### Op 7 — `lean-on-map()` 把广度还给地图\n\n- **Trigger**：你想\"全加进来才安心\"的冲动；或工作集 >5。\n- **Action**：把\"只为理解、不为编辑\"的文件从写集合移到只读地图（`/drop` + 信任 repo-map）。必要时 `/map` 审计地图已覆盖什么。\n- **Output**：更瘦的工作集，广度由签名地图承担。\n- **Evidence**：见 [[agentsop-repo-map]]；breadth 用签名、depth 用全文是两种预算。\n\n### Op 8 — `split-task()` 拆任务而非塞预算\n\n- **Trigger**：地图缩了、能 drop 的都 drop 了，预算仍破 25k。\n- **Action**：把需求拆成多个收敛子目标，每个新会话只带它要改的 ≤5 个文件。\n- **Output**：每会话工作集都在预算内。\n- **Evidence**：monorepo 上下文溢出的标准缓解 [aider.chat/docs/troubleshooting/token-limits.html]。\n\n---\n\n## 5. 困境决策案例 (Examples / Scenarios)\n\n### 案例 1 — \"我要理解 10 个文件才能改对那 2 个：全加，还是用地图？\"\n\n**触发**：一个改动横跨 10 个文件的调用链，但你实际只会修改 2 个（比如改一个 API 签名 + 它的一处实现）。本能是把 10 个全 `/add` 进来\"看全\"。\n\n**诊断**：你把\"理解广度\"误当成\"编辑深度\"。10 个里有 8 个你只需要看签名/契约，不会动它们的字节。\n\n**决策规则**：\n\n| 文件角色 | 数量 | 放哪 |\n|---|---|---|\n| 真要改字节 | 2 | `/add`（写集合） |\n| 需看完整契约/会被这次改动影响、要核对 | 1–2 | `/read`（只读全文） |\n| 只需知道\"它在哪、签名是什么\" | 6–7 | 不加，靠 [[agentsop-repo-map]] 签名 |\n\n**为什么有效**：repo-map 只读即 70.3% 命中正确文件 [aider.chat/2024/05/22/swe-bench-lite.html]——广度不需要进工作集。把 2 个进写集合、地图覆盖其余 8 个，预算从\"10 个全文\"降到\"2 全文 + 8 签名\"，编辑注意力集中在真正要改的两处。\n\n**反模式**：10 个全 `/add` → 破 25k → 模型在 8 个无关文件里挑了错的位置改 → 回滚 → 重来。\n\n### 案例 2 — \"任务做到一半预算满了，怎么办？\"\n\n**触发**：多文件重构进行中，`/tokens` 显示已过 25k，模型开始截断响应、漏看你刚加的文件。\n\n**决策树**（按代价递增，能停就停）：\n\n| 步 | 动作 | 何时停 |\n|---|---|---|\n| 1 | `/tokens` 看占用分布：工作集 / 历史 / 地图谁是大头 | 找到主要占用者 |\n| 2 | 已改完的文件 `/drop`（Op 6） | 工作集回到 ≤5、预算降到 25k 下 |\n| 3 | 地图占比大 → `--map-tokens` 调小或归零（目标已定，地图可让位） | 预算回落 |\n| 4 | 历史是大头、且话题已切 → 转 [[agentsop-session-state-hygiene]]：`/clear`（保文件去历史） | 历史清掉 |\n| 5 | 仍破 25k → 任务太宽：`split-task()`，余下子任务新开会话 | 单会话扛得住 |\n\n**关键认知**：先动**文件维度**（drop / 缩地图），再动**历史维度**（clear），最后才**拆任务**。`/drop` 与 `/clear` 是互补而非二选一——前者是本技能，后者是 [[agentsop-session-state-hygiene]]。\n\n**反模式**：预算满了第一反应是\"换更大上下文窗口的模型\"。窗口更大不改变 25k 稀释阈——大窗口模型塞到 25k+ 一样失焦。先收工作集，别先换模型。\n\n### 案例 3 — \"模型一直改错文件，我该再多加几个文件让它看清吗？\"\n\n**触发**：连续几轮，模型的 diff 落在错误文件上。直觉是\"它没看够，再 `/add` 几个\"。\n\n**诊断**：方向反了。改错文件的两种根因都不靠\"加更多文件\"解决：\n\n| 现象 | 根因 | 修复 |\n|---|---|---|\n| 改的文件根本没在工作集里 | 目标没 `/add` | `/add` 目标文件（Op 3） |\n| 工作集里文件太多、它挑错了 | `/add` 过量稀释 | `/drop` 无关的，收到 ≤5（Op 6） |\n\n**决策规则**：`/ls` 看现状 → 缺目标就加目标、多余就 drop → 不知道哪个是目标就 `locate-then-add`（Op 2）让地图替你找。**几乎不会**是\"加更多文件\"能解的。\n\n---\n\n## 6. 反模式与边界\n\n### 常见反模式\n\n1. **`/add` 整个目录**——\"这个 feature 在 `src/payments/`，全加\"。目录里 90% 的文件你不会改，纯稀释。只 `/add` 那 2–3 个目标文件。\n2. **从不 `/drop`**——把工作集当只进不出的垃圾堆。改完的文件不释放，预算单调上涨直到破 25k。`/drop` 是常态操作。\n3. **dump entire repo**——\"上下文越多越好\"，把整仓灌进去。这是 25k 阈值的反面教材；repo-map 的全部意义就是让你**不必**这么做。\n4. **用工作集做广度**——把\"只为读懂\"的文件 `/add` 进可写上下文。读懂用 `/read` 或 repo-map 签名，编辑才用 `/add`。\n5. **预算满了先换模型不先收工作集**——更大窗口不改变稀释阈。\n6. **混淆 `/drop` 与 `/clear`**——文件多就 `/drop`（本技能）；历史脏就 `/clear`（[[agentsop-session-state-hygiene]]）。用错维度解决不了问题。\n\n### 硬边界\n\n- 本技能**不替代** [[agentsop-repo-map]]：广度（找文件）是 repo-map 的活，本技能管深度（哪些进工作集）。两者配套使用。\n- 25k **是经验阈不是物理上限**：具体数字随模型变；强模型耐受更高，弱模型更低。把它当\"该警觉\"的信号，不是\"卡死\"的红线。\n- 本技能**只管文件维度**：对话历史污染交给 [[agentsop-session-state-hygiene]]；二者共享同一份 25k 预算。\n- **单文件已知任务无需本技能**：工作集天然是 1，地图可关。\n- 本技能**不保证 100% 不改错**：收紧工作集大幅降低改错率，但定位本身仍有 ~30% 残余误差（repo-map 70.3% 命中的另一面），需人工兜底。\n\n---\n\n## 7. 跨框架对照\n\n同一条\"只把要改的文件放进工作集\"纪律，四个 harness 各自的接口与默认行为：\n\n| | Aider | Claude Code | Cursor | Cline |\n|---|---|---|---|---|\n| 加入工作集 | `/add <files>`（显式，仅这些可编辑） | `Read` 工具按需读文件入上下文 | `@file` / `@folder` mention | 按需 `read_file` tool-call |\n| 移出/释放 | `/drop <files>` | 上下文压缩 / `/clear` | 移除 mention | tool-call 历史自然滚出 |\n| 只读参考 | `/read <file>`（不可编辑） | 读了即在上下文（无读/写区分） | `@file` 同样方式 | 同上，无显式只读层 |\n| 广度来源 | tree-sitter repo-map（签名） | Grep/Glob/Read 按需探索 | 全仓 codebase index | 文件树 + 主动读 |\n| 预算监控 | `/tokens`（25k 显式建议） | 上下文窗口指示 + 自动压缩 | 闭源 | 上下文长度可见 |\n| 写权限边界 | **硬**：仅 `/add` 的可编辑 | 软：能读即能改（用工具白名单约束） | 软：可改任意打开文件 | 软：可改任意读过的文件 |\n\n**关键差异**：\n\n- **Aider** 把\"读/写\"做成**硬边界**（`/read` vs `/add`），最贴合本技能——工作集就是写白名单。其 25k 阈值是这条纪律的实测来源。\n- **Claude Code** 没有显式\"工作集\"概念：`Read` 进来的文件既可读也可被 `Edit`。本技能在这里表现为**自律**——不要为了\"看全\"而 `Read` 整个目录；用 `Grep`/`Glob` 定位（相当于 repo-map 的广度），只 `Read` 你要 `Edit` 的文件。`/clear` 与自动压缩对应 `/drop` 的预算回收。\n- **Cursor** 用 `@`-mention 选上下文；mention 越多预算越紧。纪律是\"@ 你要改的，别 @ 整个 folder 求保险\"。\n- **Cline** 靠 tool-call 现场读文件，工作集隐式等于\"读过的文件集\"。纪律是不要在 plan 阶段把一堆文件读进来当背景——读过即占预算。\n\n**统一心智**：无论接口是 `/add`、`@file`、`Read` 还是 `read_file`，规则不变——**进工作集的应当是\"这一轮会编辑的文件\"，广度交给地图/搜索，预算盯住 ~25k**。\n\n---\n\n## 引用源\n\n主要：\n- [aider.chat/docs/troubleshooting/edit-errors.html] — \"Above about 25k tokens of context, most models start to become distracted.\" 本技能的核心阈值。\n- [aider.chat/docs/troubleshooting/token-limits.html] — 上下文溢出缓解。\n- [aider.chat/docs/repomap.html] — 动态预算：\"adjusts the size of the repo map dynamically based on the state of the chat.\"\n- [aider.chat/2024/05/22/swe-bench-lite.html] — repo-map 只读即 70.3% 文件命中（广度无需进工作集）。\n- [aider.chat/docs/more/edit-formats.html] — 只读 vs 读写上下文边界。\n- [aider.chat/docs/usage/commands.html] [aider.chat/docs/usage/conventions.html] — `/add` `/read` `/drop` `/tokens` 命令面。\n\n派生：\n- `references/R1-source-evidence.md` — 来源逐条 quote。\n- `intermediate/operation_candidates.json` — 操作抽取过程。\n\n关联技能：\n- [[agentsop-repo-map]] — 广度伙伴：签名地图给\"哪儿\"，本技能给\"哪些进工作集\"。\n- [[agentsop-session-state-hygiene]] — 历史维度伙伴：`/drop`（文件）与 `/clear`（历史）共享同一份 25k 预算。\n- 上游：本技能是\"通用 token 预算\"建议的 coder-agent 专属增强叠加。\n\n跨工具（一般认知）：Aider / Claude Code / Cursor / Cline 公开文档与博客。\n\nFile v0.1.3:skills/agentsop-conventions-pinning/SKILL.md\n\n---\nname: agentsop-conventions-pinning\nversion: 0.1.0\ndescription: SOP for writing, loading, and evolving a project-level convention file (CONVENTIONS.md / CLAUDE.md / .cursor/rules / .clinerules / AGENTS.md) so that a coder-agent reliably respects your codebase's style choices every session. Tool-agnostic; covers the four load mechanics (read-only attachment, ancestor-walk auto-load, glob-scoped rules, agent backstory) and the conflict resolution between pinned conventions and the existing code.\ndomain: coder-agent infrastructure / context engineering\naudience: coder-agents and engineers configuring them, on any project lived in for > 1 day\ntrigger_keywords:\n  - \"conventions file\"\n  - \"CONVENTIONS.md\"\n  - \"CLAUDE.md\"\n  - \".cursorrules\"\n  - \".clinerules\"\n  - \"AGENTS.md\"\n  - \"coding standards for AI\"\n  - \"style guide for agent\"\n  - \"agent ignored my rule\"\n  - \"pin my coding style\"\nwhen_to_use:\n  - \"any project you (or your agent) will return to more than once\"\n  - \"the same correction has been typed into chat more than twice\"\n  - \"code review (human or LLM) keeps catching style violations the agent should know\"\n  - \"onboarding a new agent / new teammate; they need the project's tacit rules in writing\"\n  - \"you switch coder-tools and want one canonical style source across Aider, Claude Code, Cursor, Cline\"\nwhen_not_to_use:\n  - \"one-off / throwaway scripts where the cost of writing rules > the cost of the work\"\n  - \"true greenfield where conventions ARE being invented as code; pin AFTER the first 2-3 modules stabilise\"\n  - \"you need hard enforcement (lint/format/CI) — conventions are guidance, hooks/precommit are enforcement\"\n  - \"the project already has a lint config that fully encodes the rule — point at the lint config instead\"\n---\n\n# Conventions Pinning — Writing a Style Guide Your Coder-Agent Will Actually Read\n\n> One line: a conventions file is **the system prompt of your codebase**. Treat it like a system prompt, not like a README. Anti-patterns: writing prose, narrating history, marketing the project. Patterns: command-first, verifiable, \"prefer X over Y\", < 200 lines.\n\n---\n\n## 1. 何时激活 (When to Activate)\n\n### 1.1 Direct triggers\n- The user (human or upstream agent) asks \"how do I make Claude/Cursor/Cline/Aider respect our style?\".\n- The same correction has been typed in chat ≥ 2 times this week (\"use httpx not requests\", \"add type hints\", \"no comments on every line\"). Claude Code's docs codify this rule: *\"Add to it when Claude makes the same mistake a second time.\"* [code.claude.com/docs/en/memory]\n- A new project is past the \"first 2 modules\" phase — there are now style choices implicit in the code that an outsider (or fresh-context agent) can't see.\n- The team is switching coder-tools (Aider → Claude Code, or adding Cursor) and conventions are scattered in chat history.\n- An AI code review caught the same anti-pattern twice.\n\n### 1.2 Reverse triggers (skip)\n- **One-off / throwaway scripts**. The write-cost of a conventions file is fixed; the savings are proportional to session count. < 3 sessions → don't bother.\n- **True greenfield**. The first 2-3 files of a project ARE the convention. Pinning style before the style exists locks in arbitrary choices.\n- **Hard enforcement needed**. A conventions file is *context*, not a *hook*. Claude Code's docs are explicit: *\"CLAUDE.md instructions shape Claude's behavior but are not a hard enforcement layer.\"* [code.claude.com/docs/en/memory] If the rule must run every time (e.g. \"must `make lint` before commit\"), write a hook / precommit / CI check.\n- **The rule is already in a config that the agent can read** — `.eslintrc`, `pyproject.toml`, `.editorconfig`, `tsconfig.json`. Point the agent at the config; don't duplicate.\n\n### 1.3 Mental check\n> A conventions file earns its tokens only if it contains **information that is not already in the repository**. The research is unambiguous on this: *\"Developer-written context files performed better for exactly the reason you'd guess: they contained information that wasn't already in the repository — tooling preferences, workflow requirements, conventions that existed in developers' heads but not in any documentation.\"* [developer.upsun.com/posts/ai/agents-md-less-is-more]\n\nIf everything you would write into CONVENTIONS.md is already discoverable from `package.json`, `pyproject.toml`, `.eslintrc`, the test directory, and obvious code patterns — don't write the file. The agent will pre-cache it itself.\n\n---\n\n## 2. 核心心智模型 (Mental Model)\n\n### 2.1 Convention as compile-time, code-review as runtime\n\n```\n+-------------------------------------+   +-------------------------------------+\n|  COMPILE-TIME (conventions file)    |   |  RUNTIME (code review / lint / CI)  |\n|                                     |   |                                     |\n|  - Loaded once per session          |   |  - Runs on every change             |\n|  - Shapes generation                |   |  - Catches violations after-the-fact|\n|  - Cheap to update, free to ignore  |   |  - Costly to set up, hard to ignore |\n|  - \"Prefer X over Y\" lives here     |   |  - \"X must always hold\" lives here  |\n|  - Style + intent + preferences     |   |  - Invariants + safety + correctness|\n+-------------------------------------+   +-------------------------------------+\n        ^                                            ^\n        |        Conventions guide;                  |\n        |        review enforces.                    |\n        |        Both are needed; they fail          |\n        |        in different ways.                  |\n```\n\nConventions fail by **silent drift**: agent ignores the rule once, no one notices, code is merged. Review fails by **late catch**: violation is found post-PR, expensive to fix. The two complement, not substitute.\n\n### 2.2 The four load mechanics across the ecosystem\n\nEvery coder-tool has settled on one of four mechanics for getting persistent context into the agent. Knowing which mechanic your tool uses is the difference between \"agent reads my rules\" and \"agent silently ignores them.\"\n\n| Mechanic | Examples | How it works |\n|---|---|---|\n| **Read-only attach (explicit)** | Aider `--read CONVENTIONS.md` | You explicitly attach the file. Loaded read-only every session, cacheable. [aider.chat/docs/usage/conventions.html] |\n| **Ancestor-walk auto-load** | Claude Code `CLAUDE.md`, `./.claude/CLAUDE.md`, `~/.claude/CLAUDE.md` | Tool walks from cwd up to root, concatenates every `CLAUDE.md` it finds, plus `CLAUDE.local.md` per directory. Subdirectory files load on-demand when files in that dir are read. [code.claude.com/docs/en/memory] |\n| **Glob-scoped rules dir** | Cursor `.cursor/rules/*.mdc`, Claude Code `.claude/rules/*.md` with `paths:` frontmatter, Cline `.clinerules/*.md` with `paths` frontmatter | Multiple small files. Each can declare a `paths:` glob; rule loads only when the agent touches a matching file. [cursor.com/docs/rules], [docs.cline.bot/customization/cline-rules] |\n| **Agent backstory** | CrewAI agent `backstory=` field, single-agent system-prompt suffix | Style/preferences embedded into the agent's persona at construction time. Per-agent, not per-project. [docs.crewai.com/en/concepts/agents] |\n\nA project that uses multiple tools should pick **one source of truth** and have the other tools `@`-import or symlink to it. Claude Code 2.x docs explicitly recommend this: *\"If your repository already uses AGENTS.md for other coding agents, create a CLAUDE.md that imports it ... `@AGENTS.md`.\"* [code.claude.com/docs/en/memory]\n\n### 2.3 Three rules for what goes in (and what doesn't)\n\n| ✅ Put in | ❌ Keep out |\n|---|---|\n| \"Prefer httpx over requests.\" [aider.chat/docs/usage/conventions.html] | \"We've been using Python since 2019 ...\" (project narrative) |\n| \"Use 2-space indentation.\" | \"Run `npm install` then `npm test`.\" (already in package.json) |\n| \"API handlers live in `src/api/handlers/`.\" [code.claude.com/docs/en/memory] | \"Format code properly.\" (unverifiable) |\n| \"Add type hints everywhere; ruff `ANN*` rules are on.\" | \"Be careful with database connections.\" (ambiguous) |\n| Anti-pattern: \"Never use `os.system`; use `subprocess.run`.\" | \"TODO: write better docs here.\" (file is not a TODO list) |\n| Workflow that's not in a script: \"Before commit, run `make fmt && make test`.\" | The output of `make help` (the agent can `make help` itself) |\n\n**The litmus test**: *would this information be discoverable by a competent developer who spent 30 minutes browsing the repo?* If yes — leave it out, the agent will discover it too. If no — pin it. [developer.upsun.com/posts/ai/agents-md-less-is-more]\n\n### 2.4 Size budget: 200 lines, hard\n\nClaude Code documentation: *\"Target under 200 lines per CLAUDE.md file. Longer files consume more context and reduce adherence.\"* [code.claude.com/docs/en/memory]\n\nCursor docs: *\"Keep rules concise: under 500 lines. ... A 1,000-word always-apply rule is expensive, so trim aggressively or convert it to auto-attached with appropriate globs.\"* [cursor.com/docs/rules]\n\nAider docs: *\"Above about 25k tokens of context, most models start to become distracted.\"* [aider.chat/docs/troubleshooting/edit-errors.html] — and CONVENTIONS.md is one chunk competing for that budget against the actual code.\n\nWhen the file passes ~200 lines, split it: glob-scoped rules (Cursor, Claude Code `paths:`), per-language file (`python.md`, `react.md`), or move detail into a skill that loads on demand.\n\n### 2.5 The conflict precedence rule\n\nYou will eventually hit: **CONVENTIONS.md says A, the existing code shows B.** What wins?\n\nDefault precedence the major tools converge on:\n\n```\nmanaged/org policy   >  user (~/.claude/)  >  project (./)  >  local (gitignored)\n                          (loaded in order; later overrides for conflicts)\nexisting code in repo  ⟂  conventions file   ← these don't have built-in precedence;\n                                                you must declare it explicitly\n```\n\nThe agent does **not** know which one you want to win unless you tell it. Two patterns:\n\n1. **Convention wins, refactor the drift.** Add to CONVENTIONS.md: *\"If existing code conflicts with these rules, flag it as drift and propose a refactor, do not propagate the old pattern.\"*\n2. **Code wins, archive the rule.** If the codebase has irrevocably moved past a rule, delete the rule. Stale rules are worse than no rules — they cost tokens and create silent contradictions.\n\nClaude Code docs warn about this directly: *\"If two rules contradict each other, Claude may pick one arbitrarily. Review your CLAUDE.md files ... periodically to remove outdated or conflicting instructions.\"* [code.claude.com/docs/en/memory]\n\n---\n\n## 3. SOP 工作流 (Standard Operating Procedure)\n\n### Phase 0: Decide whether to write one at all\n\n```\n[Q1] Will this codebase be touched in ≥ 3 future sessions?\n       no  -> skip; pin nothing.\n       yes -> continue.\n\n[Q2] Are there style/library choices NOT already encoded in config files\n     (pyproject, eslintrc, editorconfig, tsconfig)?\n       no  -> point the agent at the existing configs; skip a conventions file.\n       yes -> continue.\n\n[Q3] Is the rule something the agent should DO (guidance) or something that\n     MUST hold (invariant)?\n       must hold -> write a hook/precommit/CI check instead.\n       guidance  -> conventions file is the right tool. Continue.\n```\n\n### Phase 1: Write — the first cut\n\nStart with **5–15 bullets, max**. Aider's documented example is exactly this minimal: two bullets (\"prefer httpx over requests\" + \"use types everywhere\"). [aider.chat/docs/usage/conventions.html]\n\nStructure template (copy this):\n\n```markdown\n# <Project> Conventions\n\n## Language & versions\n- Python 3.12+ (no 3.11 syntax workarounds).\n- Node 20 LTS, TypeScript strict mode.\n\n## Libraries — prefer / avoid\n- HTTP client: prefer `httpx` over `requests`.\n- Date math: prefer `pendulum` over stdlib `datetime` for tz-aware ops.\n- Avoid: `os.system` (use `subprocess.run`), `eval`, raw f-strings in logging.\n\n## Style\n- Type hints everywhere (`ANN*` ruff rules are on).\n- 2-space indent (TS), 4-space (Python).\n- No inline comments unless explaining \"why not how\".\n\n## Layout\n- API handlers: `src/api/handlers/`\n- Shared types: `src/types/`\n- Tests mirror source: `tests/<module>/test_<file>.py`\n\n## Workflow\n- Before commit: `make fmt && make test`.\n- Conventional commits: `feat:`, `fix:`, `chore:`, `refactor:`.\n\n## When this file disagrees with existing code\n- Treat existing code as drift; propose a refactor in a separate commit.\n```\n\nSave it at the **canonical location for the primary tool**:\n\n| Tool | Location |\n|---|---|\n| Aider | `CONVENTIONS.md` at repo root (loaded with `--read`) |\n| Claude Code | `./CLAUDE.md` or `./.claude/CLAUDE.md` |\n| Cursor (modern) | `.cursor/rules/main.mdc` (one rule file) |\n| Cursor (legacy) | `.cursorrules` at repo root (deprecated but still works) |\n| Cline | `.clinerules/coding.md` |\n| Multi-tool | `AGENTS.md` at repo root; have CLAUDE.md / .cursorrules / .clinerules import or symlink |\n\n### Phase 2: Load — make sure the agent actually reads it\n\nThis is where most projects fail silently.\n\n| Tool | Load command / config |\n|---|---|\n| Aider | `aider --read CONVENTIONS.md` per invocation, OR put `read: CONVENTIONS.md` into `.aider.conf.yml`. [aider.chat/docs/usage/conventions.html] |\n| Claude Code | Automatic. `CLAUDE.md` in cwd and all ancestor dirs are loaded at session start. Verify with `/memory` — it lists every loaded file. [code.claude.com/docs/en/memory] |\n| Cursor | `.mdc` files in `.cursor/rules/` auto-attach. Use `alwaysApply: true` or `globs:` in frontmatter. [cursor.com/docs/rules] |\n| Cline | All `.md`/`.txt` files in `.clinerules/` are concatenated and loaded automatically. [docs.cline.bot/customization/cline-rules] |\n| CrewAI | Paste relevant rules into each agent's `backstory=` string at construction time. [docs.crewai.com/en/concepts/agents] |\n\n**Verify the load worked, every time you change tools or environments**:\n\n- Aider: `/tokens` shows the file in the context budget.\n- Claude Code: `/memory` lists it.\n- Cursor: hover the rule in the sidebar — it shows which conversation it's active in.\n- Cline: the rules panel in the Cline sidebar shows the loaded set.\n\nIf the file isn't shown — the agent isn't reading it. Don't proceed.\n\n### Phase 3: Enforce — close the loop with a test\n\nPin one **specific, verifiable** rule (e.g. \"use httpx not requests\"). Then ask the agent to write a new HTTP call from scratch. If it uses `requests`, the load failed or the rule is too vague. Fix and retest.\n\nThis is the same test Aider documents: their example explicitly shows `httpx` + type hints get used *with* CONVENTIONS.md and `requests` + no types get used *without*. [aider.chat/docs/usage/conventions.html] If you cannot produce that A/B in your own setup, the load is broken.\n\n### Phase 4: Evolve — when to add a rule, when to delete one\n\n**Add a rule when** (codified from Claude Code docs): [code.claude.com/docs/en/memory]\n1. Agent makes the same mistake twice in a session, or twice across sessions.\n2. A code reviewer catches something the agent should have known about *this* codebase.\n3. A new teammate would need the same context to be productive.\n4. You re-explain the same correction in chat a third time.\n\n**Delete a rule when**:\n1. The codebase has moved past it — the rule says \"use X\" but every new file uses Y and the team agrees.\n2. Lint/CI now enforces it (move from convention to enforcement).\n3. The rule is too vague to verify and has produced no measurable behavior change (\"be careful with concurrency\").\n4. The file is over ~200 lines and this rule has the lowest hit rate.\n\n**Split a rule when** it only applies to part of the codebase: use glob-scoped rules. Claude Code `.claude/rules/api.md` with `paths: [\"src/api/**/*.ts\"]`. Cursor: per-`.mdc` `globs:` field. This shrinks the always-on context budget. [code.claude.com/docs/en/memory], [cursor.com/docs/rules]\n\n### Phase 5: Review — periodic audit (monthly or per major refactor)\n\n```\n1. Run `/memory` (or equivalent) — list what's loaded.\n2. For each rule:\n     - Has it been triggered? (grep recent PRs / chat for the keyword)\n     - Does the codebase still follow it? (grep code for violations)\n     - If both no → delete.\n     - If first no, second yes → the rule is enforced by gravity; consider deleting.\n     - If first yes, second no → drift; either fix code or kill rule.\n3. Check size: > 200 lines → split into glob-scoped rules.\n4. Check for contradictions across files (project + user + ancestor).\n```\n\n---\n\n## 4. 操作模型 (Operation Model)\n\nEight reusable operations. Each is Trigger / Action / Output / Evidence. Detailed JSON form in `intermediate/operation_candidates.json`.\n\n| # | Operation | Trigger | Action | Output |\n|---|---|---|---|---|\n| OP-1 | **Decide-to-pin** | User asks how to make agent respect style, OR same correction typed ≥ 2× | Run Phase 0 decision tree | `pin` / `skip` / `use-hooks-instead` |\n| OP-2 | **Write first cut** | Decision = pin | Fill the 6-section template (Language / Libraries / Style / Layout / Workflow / Conflict-rule), 5–15 bullets total | `CONVENTIONS.md` (or tool-equivalent) at canonical path |\n| OP-3 | **Wire load mechanism** | File written | Pick tool's load path: `--read` flag, ancestor-walk, glob `paths:`, or backstory string | File appears in agent's loaded-files list |\n| OP-4 | **Verify with A/B test** | Load wired | Ask agent to perform a task that triggers a specific bullet (e.g. HTTP call); confirm rule was followed | Pass/fail; if fail go back to OP-3 |\n| OP-5 | **Add a rule** | Same mistake twice OR review caught project-specific knowledge | Add bullet that is concrete, verifiable, and not already encoded in lint/config | Updated file, < 200 lines |\n| OP-6 | **Delete a stale rule** | Codebase moved past it, OR rule is unenforceable, OR file > 200 lines and rule has lowest hit rate | Remove the bullet; if it now lives in lint/CI, note that in commit msg | Slimmer file |\n| OP-7 | **Split into glob-scoped rule** | Rule applies only to subset (e.g. only `src/api/`); main file passing 200 lines | Move rule into per-glob file with `paths:` frontmatter (Claude Code rules / Cursor `.mdc` `globs:`) | New scoped file; main file shrinks |\n| OP-8 | **Cross-tool unification** | Project uses multiple coder-tools (e.g. Claude Code + Cursor + Aider) | Make one file canonical (`AGENTS.md` at root is the cross-tool convention); have other tools `@`-import or symlink to it | One source of truth, multiple thin pointers |\n\n---\n\n## 5. 困境决策案例 (Dilemma Cases)\n\n### Case 1 — \"I added `--read CONVENTIONS.md` but the agent still uses requests instead of httpx\"\n\n**Symptom**: rule is loaded (Aider `/tokens` shows it; Claude Code `/memory` lists it), but generation ignores it.\n\n**Differential diagnosis** (cheapest first):\n\n1. **Token budget exceeded**. Aider: *\"Above about 25k tokens of context, most models start to become distracted.\"* [aider.chat/docs/troubleshooting/edit-errors.html] Drop irrelevant files; `/tokens` to confirm.\n2. **Conflicting nearby example**. The agent is reading existing code in the same session that uses `requests`. Code-in-context is a stronger demonstration than a one-line rule. *\"If two rules contradict each other, Claude may pick one arbitrarily.\"* [code.claude.com/docs/en/memory] Fix: add the conflict-resolution clause to CONVENTIONS.md (\"if existing code uses requests, treat as drift and propose refactor\"), or `/drop` the conflicting files.\n3. **Rule is too vague**. \"Prefer modern HTTP libs\" is ignorable; \"Use `httpx.Client` with `timeout=10.0`; do not import `requests`\" is not.\n4. **Multiple conventions files contradict**. User-level (`~/.claude/CLAUDE.md`) says one thing, project says another. Claude Code loads them in a defined order (managed → user → project → local) [code.claude.com/docs/en/memory] — but if both have the same rule wording with different answers, the merge is ambiguous. Verify with `/memory`.\n5. **Stale cache**. Aider caches the read file across the session; restart if you edited it.\n\n**Resolution rule**: if cases 1–4 don't resolve it, the rule belongs in **enforcement** (lint, precommit), not **guidance**. Add a ruff rule to `pyproject.toml` that bans `import requests`. The conventions file is not a hammer.\n\n### Case 2 — \"Style drift across files: half use `httpx`, half use `requests`. Fix in conventions or in code?\"\n\n**Trigger**: codebase has visible inconsistency. Agent will mirror whichever file it reads first.\n\n**Decision rule**:\n\n| If ... | Then ... |\n|---|---|\n| The team has decided httpx is the future, no exceptions | (a) Add rule to CONVENTIONS.md; (b) write a single PR that migrates all `requests` callsites; (c) add a lint rule forbidding `requests` |\n| There's a *reason* the old code uses requests (e.g. a sync-only library inside) | Document the exception in CONVENTIONS.md (\"`requests` allowed in `src/legacy/`, banned elsewhere\") and consider a glob-scoped rule |\n| Half-and-half because no one decided | Make the decision, then case (a). The conventions file is the wrong place to record an *unresolved* choice |\n\n**The anti-pattern to avoid**: writing \"agents should use httpx going forward\" without doing the migration. The agent will see 50% requests in the existing code, ignore your rule, and the drift continues. Conventions describe **the codebase you have**, not the one you wish you had. If the rule contradicts > ~30% of the existing code, **fix the code first, pin the rule second**.\n\n### Case 3 — \"I'm using Claude Code AND Cursor on the same repo. Two conventions files? Or one?\"\n\n**Decision**: one. Two files = two truths = silent disagreement.\n\n**Implementation** (Claude Code 2.x's recommended pattern): [code.claude.com/docs/en/memory]\n\n```\nrepo/\n├── AGENTS.md                  # canonical, ~80 lines\n├── CLAUDE.md                  # one line: \"@AGENTS.md\"\n├── .cursor/rules/main.mdc     # frontmatter alwaysApply: true; body: contents copied or symlinked\n└── .clinerules/main.md        # symlink → ../AGENTS.md (Cline reads .md transparently)\n```\n\nFor Cursor (`.mdc` files require frontmatter, can't symlink raw), either:\n- Generate the `.mdc` from `AGENTS.md` in a pre-commit hook, OR\n- Keep the rule body short enough that copying it across 3 tools is tolerable.\n\n**Surprising finding from the research**: Claude Code's `/init` reads existing `.cursorrules` and `.windsurfrules` and incorporates them into the generated `CLAUDE.md`. [code.claude.com/docs/en/memory] So if you already have a `.cursorrules`, Claude Code will not silently ignore it on first init — it pulls it in.\n\n### Case 4 — \"My CONVENTIONS.md is now 600 lines. Adherence is dropping.\"\n\n**Diagnosis**: violating both Aider's 25k-distraction threshold (in aggregate context) and Claude Code's 200-line per-file recommendation. [code.claude.com/docs/en/memory], [aider.chat/docs/troubleshooting/edit-errors.html] Cursor's docs are even more explicit: *\"if you're hitting context limits, look at your always-apply rules first.\"* [cursor.com/docs/rules]\n\n**Resolution sequence** (cheapest first):\n\n1. **Audit for stale rules**. Delete any rule the codebase no longer follows. Delete any rule covered by lint/CI now.\n2. **Split by file scope**. Move language-specific rules to glob-scoped files (Claude Code `.claude/rules/python.md` with `paths: [\"**/*.py\"]`; Cursor `.mdc` with `globs:` field). Only the rules matching files-in-this-session load.\n3. **Move detail into skills / playbooks**. Workflows that aren't always-on (e.g. \"release procedure\") belong in a skill or runbook the agent loads on demand, not in CONVENTIONS.md.\n4. **Demote prose**. Any paragraph longer than 2 sentences is probably narrative. Rewrite as bullets or delete.\n\nTarget after pruning: < 150 lines, every line a verifiable directive.\n\n### Case 5 — \"Greenfield: my coder-agent is writing the project from scratch. Should I pin conventions?\"\n\n**Decision**: **not yet**. Pinning conventions before the conventions exist forces premature decisions and the agent will fight you on them.\n\n**Right sequence**:\n1. Let the agent write 2–4 modules with minimal guidance (just language + framework version).\n2. Read what it produced. Pick the choices you want to keep.\n3. *Now* write CONVENTIONS.md with those choices, plus the libraries you've added.\n4. Future modules inherit the now-pinned style.\n\nThe exception: **policy-level** rules that aren't about code style but about constraints — \"never call external APIs without explicit approval\", \"no secrets in git history\", \"all functions need docstrings\". Those can be pinned from day 0 because they don't depend on emergent style.\n\n### Case 6 — \"Agent ignored a rule despite a perfect load. What now?\"\n\nThe unintuitive bit: CONVENTIONS.md is **context, not enforcement**. Claude Code documents this explicitly: *\"CLAUDE.md content is delivered as a user message after the system prompt, not as part of the system prompt itself. Claude reads it and tries to follow it, but there's no guarantee of strict compliance, especially for vague or conflicting instructions.\"* [code.claude.com/docs/en/memory]\n\nTiered response:\n\n| Tier | Action |\n|---|---|\n| 1 | Rewrite the rule to be more specific. \"Format properly\" → \"Use 2-space indent, double quotes, no trailing comma in JSON\" |\n| 2 | Move it earlier in the file (recency in context matters) and bold it |\n| 3 | Add a counter-example: \"Wrong: `requests.get(url)`. Right: `httpx.get(url, timeout=10)`\" |\n| 4 | If the rule MUST hold, demote it to a hook / lint / CI gate. Conventions are guidance, not contracts |\n\n---\n\n## 6. 反模式与边界 (Anti-Patterns and Boundaries)\n\n### Anti-patterns (do not do these)\n\n1. **Project narrative in CONVENTIONS.md**. \"Our team started this in 2020 ...\" costs tokens, produces zero behavior change. The research is clear: prose paragraphs *\"reliably get ignored.\"* [blakecrosley.com/blog/agents-md-patterns]\n2. **Duplicating\n\nArchive v0.1.2: 258 files, 1143690 bytes\n\nFiles: CHANGELOG.md (913b), CONTRIBUTING.md (1060b), LICENSE (1069b), package.json (366b), README_EN.md (5703b), README.md (5867b), skill-card.md (2528b), skill.json (159b), SKILL.md (6324b), skills/agentsop-agent-topology-selection/intermediate/operation_candidates.json (7976b), skills/agentsop-agent-topology-selection/README.md (4024b), skills/agentsop-agent-topology-selection/references/R1-source-evidence.md (7130b), skills/agentsop-agent-topology-selection/SKILL.md (19118b), skills/agentsop-aider/intermediate/operation_candidates.json (10797b), skills/agentsop-aider/README.md (2494b), skills/agentsop-aider/references/R1-architecture.md (7186b), skills/agentsop-aider/references/R2-sop-workflow.md (7428b), skills/agentsop-aider/references/R3-dilemma-cases.md (7648b), skills/agentsop-aider/references/R4-anti-patterns.md (6236b), skills/agentsop-aider/references/R5-ecosystem-context.md (6272b), skills/agentsop-aider/SKILL.md (20180b), skills/agentsop-bio-fraud-forensics/examples/demo_screening.md (4291b), skills/agentsop-bio-fraud-forensics/README.md (2955b), skills/agentsop-bio-fraud-forensics/references/R01-misconduct-taxonomy.md (20390b), skills/agentsop-bio-fraud-forensics/references/R02-image-forensics.md (19498b), skills/agentsop-bio-fraud-forensics/references/R03-statistical-forensics.md (19779b), skills/agentsop-bio-fraud-forensics/references/R04-exposure-sites-method.md (21526b), skills/agentsop-bio-fraud-forensics/references/R05-evidence-red-lines.md (21908b), skills/agentsop-bio-fraud-forensics/references/R06-screening-workflow.md (22217b), skills/agentsop-bio-fraud-forensics/references/R07-paper-mill-signals.md (19986b), skills/agentsop-bio-fraud-forensics/references/research_notes.md (8099b), skills/agentsop-bio-fraud-forensics/references/sop_models.md (17858b), skills/agentsop-bio-fraud-forensics/SKILL.md (12777b), skills/agentsop-bio-fraud-forensics/USAGE.md (8508b), skills/agentsop-bounded-loop/intermediate/operation_candidates.json (6838b), skills/agentsop-bounded-loop/references/R1-source-evidence.md (7055b), skills/agentsop-bounded-loop/references/R2-cross-framework.md (5322b), skills/agentsop-bounded-loop/SKILL.md (28881b), skills/agentsop-code-execution-decision/intermediate/operation_candidates.json (4342b), skills/agentsop-code-execution-decision/README.md (1988b), skills/agentsop-code-execution-decision/references/R1-source-evidence.md (4670b), skills/agentsop-code-execution-decision/SKILL.md (26082b), skills/agentsop-context-scope-discipline/intermediate/operation_candidates.json (5468b), skills/agentsop-context-scope-discipline/README.md (3178b), skills/agentsop-context-scope-discipline/references/R1-source-evidence.md (5047b), skills/agentsop-context-scope-discipline/SKILL.md (20665b), skills/agentsop-conventions-pinning/intermediate/operation_candidates.json (9575b), skills/agentsop-conventions-pinning/README.md (2338b), skills/agentsop-conventions-pinning/references/R1-source-evidence.md (10849b), skills/agentsop-conventions-pinning/references/R2-tool-equivalents.md (9203b), skills/agentsop-conventions-pinning/SKILL.md (36286b), skills/agentsop-cost-tiered-models/intermediate/operation_candidates.json (7378b), skills/agentsop-cost-tiered-models/README.md (4331b), skills/agentsop-cost-tiered-models/references/R1-source-evidence.md (6583b), skills/agentsop-cost-tiered-models/SKILL.md (20887b), skills/agentsop-crewai/intermediate/operation_candidates.json (6596b), skills/agentsop-crewai/README.md (1862b), skills/agentsop-crewai/references/R1-architecture.md (4909b), skills/agentsop-crewai/references/R2-sop-workflow.md (5383b), skills/agentsop-crewai/references/R3-dilemma-cases.md (7126b), skills/agentsop-crewai/references/R4-anti-patterns.md (4921b), skills/agentsop-crewai/references/R5-ecosystem-context.md (5549b), skills/agentsop-crewai/SKILL.md (24740b), skills/agentsop-dify/intermediate/operation_candidates.json (9704b), skills/agentsop-dify/README.md (1770b), skills/agentsop-dify/references/R1-architecture.md (7760b), skills/agentsop-dify/references/R2-sop-workflow.md (7097b), skills/agentsop-dify/references/R3-dilemma-cases.md (10109b), skills/agentsop-dify/references/R4-anti-patterns.md (6465b), skills/agentsop-dify/references/R5-ecosystem-context.md (7700b), skills/agentsop-dify/SKILL.md (28387b), skills/agentsop-domain-eval-set/intermediate/operation_candidates.json (6046b), skills/agentsop-domain-eval-set/README.md (2808b), skills/agentsop-domain-eval-set/references/R1-source-evidence.md (4986b), skills/agentsop-domain-eval-set/SKILL.md (25428b), skills/agentsop-dspy/intermediate/operation_candidates.json (12570b), skills/agentsop-dspy/README.md (2415b), skills/agentsop-dspy/references/R1-architecture.md (6013b), skills/agentsop-dspy/references/R2-sop-workflow.md (5669b), skills/agentsop-dspy/references/R3-dilemma-cases.md (7629b)\n\nArchive v0.1.1: 258 files, 1143479 bytes\n\nFiles: CHANGELOG.md (913b), CONTRIBUTING.md (1060b), LICENSE (1069b), package.json (366b), README_EN.md (5703b), README.md (5631b), skill-card.md (2425b), skill.json (159b), SKILL.md (6324b), skills/agentsop-agent-topology-selection/intermediate/operation_candidates.json (7976b), skills/agentsop-agent-topology-selection/README.md (4024b), skills/agentsop-agent-topology-selection/references/R1-source-evidence.md (7130b), skills/agentsop-agent-topology-selection/SKILL.md (19118b), skills/agentsop-aider/intermediate/operation_candidates.json (10797b), skills/agentsop-aider/README.md (2494b), skills/agentsop-aider/references/R1-architecture.md (7186b), skills/agentsop-aider/references/R2-sop-workflow.md (7428b), skills/agentsop-aider/references/R3-dilemma-cases.md (7648b), skills/agentsop-aider/references/R4-anti-patterns.md (6236b), skills/agentsop-aider/references/R5-ecosystem-context.md (6272b), skills/agentsop-aider/SKILL.md (20180b), skills/agentsop-bio-fraud-forensics/examples/demo_screening.md (4291b), skills/agentsop-bio-fraud-forensics/README.md (2955b), skills/agentsop-bio-fraud-forensics/references/R01-misconduct-taxonomy.md (20390b), skills/agentsop-bio-fraud-forensics/references/R02-image-forensics.md (19498b), skills/agentsop-bio-fraud-forensics/references/R03-statistical-forensics.md (19779b), skills/agentsop-bio-fraud-forensics/references/R04-exposure-sites-method.md (21526b), skills/agentsop-bio-fraud-forensics/references/R05-evidence-red-lines.md (21908b), skills/agentsop-bio-fraud-forensics/references/R06-screening-workflow.md (22217b), skills/agentsop-bio-fraud-forensics/references/R07-paper-mill-signals.md (19986b), skills/agentsop-bio-fraud-forensics/references/research_notes.md (8099b), skills/agentsop-bio-fraud-forensics/references/sop_models.md (17858b), skills/agentsop-bio-fraud-forensics/SKILL.md (12777b), skills/agentsop-bio-fraud-forensics/USAGE.md (8508b), skills/agentsop-bounded-loop/intermediate/operation_candidates.json (6838b), skills/agentsop-bounded-loop/references/R1-source-evidence.md (7055b), skills/agentsop-bounded-loop/references/R2-cross-framework.md (5322b), skills/agentsop-bounded-loop/SKILL.md (28881b), skills/agentsop-code-execution-decision/intermediate/operation_candidates.json (4342b), skills/agentsop-code-execution-decision/README.md (1988b), skills/agentsop-code-execution-decision/references/R1-source-evidence.md (4670b), skills/agentsop-code-execution-decision/SKILL.md (26082b), skills/agentsop-context-scope-discipline/intermediate/operation_candidates.json (5468b), skills/agentsop-context-scope-discipline/README.md (3178b), skills/agentsop-context-scope-discipline/references/R1-source-evidence.md (5047b), skills/agentsop-context-scope-discipline/SKILL.md (20665b), skills/agentsop-conventions-pinning/intermediate/operation_candidates.json (9575b), skills/agentsop-conventions-pinning/README.md (2338b), skills/agentsop-conventions-pinning/references/R1-source-evidence.md (10849b), skills/agentsop-conventions-pinning/references/R2-tool-equivalents.md (9203b), skills/agentsop-conventions-pinning/SKILL.md (36286b), skills/agentsop-cost-tiered-models/intermediate/operation_candidates.json (7378b), skills/agentsop-cost-tiered-models/README.md (4331b), skills/agentsop-cost-tiered-models/references/R1-source-evidence.md (6583b), skills/agentsop-cost-tiered-models/SKILL.md (20887b), skills/agentsop-crewai/intermediate/operation_candidates.json (6596b), skills/agentsop-crewai/README.md (1862b), skills/agentsop-crewai/references/R1-architecture.md (4909b), skills/agentsop-crewai/references/R2-sop-workflow.md (5383b), skills/agentsop-crewai/references/R3-dilemma-cases.md (7126b), skills/agentsop-crewai/references/R4-anti-patterns.md (4921b), skills/agentsop-crewai/references/R5-ecosystem-context.md (5549b), skills/agentsop-crewai/SKILL.md (24740b), skills/agentsop-dify/intermediate/operation_candidates.json (9704b), skills/agentsop-dify/README.md (1770b), skills/agentsop-dify/references/R1-architecture.md (7760b), skills/agentsop-dify/references/R2-sop-workflow.md (7097b), skills/agentsop-dify/references/R3-dilemma-cases.md (10109b), skills/agentsop-dify/references/R4-anti-patterns.md (6465b), skills/agentsop-dify/references/R5-ecosystem-context.md (7700b), skills/agentsop-dify/SKILL.md (28387b), skills/agentsop-domain-eval-set/intermediate/operation_candidates.json (6046b), skills/agentsop-domain-eval-set/README.md (2808b), skills/agentsop-domain-eval-set/references/R1-source-evidence.md (4986b), skills/agentsop-domain-eval-set/SKILL.md (25428b), skills/agentsop-dspy/intermediate/operation_candidates.json (12570b), skills/agentsop-dspy/README.md (2415b), skills/agentsop-dspy/references/R1-architecture.md (6013b), skills/agentsop-dspy/references/R2-sop-workflow.md (5669b), skills/agentsop-dspy/references/R3-dilemma-cases.md (7629b)\n\nArchive v0.1.0: 24 files, 65005 bytes\n\nFiles: CONTRIBUTING.md (1060b), package.json (366b), README_EN.md (5293b), README.md (5221b), skill-card.md (2288b), skill.json (159b), SKILL.md (6324b), skills/LEAP/README_EN.md (2547b), skills/LEAP/README.md (2512b), skills/LEAP/references/skill-grammar.md (15124b), skills/LEAP/scripts/build_component_index.py (22939b), skills/LEAP/scripts/build_corpus.py (11221b), skills/LEAP/scripts/download_subtitles.sh (1950b), skills/LEAP/scripts/quality_check.py (16728b), skills/LEAP/scripts/score_skill.py (7254b), skills/LEAP/scripts/srt_to_transcript.py (3115b), skills/LEAP/skill.json (141b), skills/LEAP/SKILL.md (29303b), skills/Lens/README_EN.md (1475b), skills/Lens/README.md (1459b), skills/Lens/skill.json (134b), skills/Lens/SKILL.md (5873b), 技术文档.md (4611b), _meta.json (131b)","readmeExcerpt":"Skill: Skill Alchemy Main Owner: agentsope Summary: SkillAlchemy — 一念落地，万象成形。输入任意想法或蒸馏目标，输出可安装的 SKILL.md。 内部编排 Lens（看清问题）和 LEAP（执行蒸馏/融合）。用户唯一入口。 Use when 用户说「蒸馏」「生成 skill」「融合」「我想做 X 但不知道从哪下手」。 Tags: latest:0.1.3 Version history: v0.1.3 | 2026-06-15T11:41:28.437Z | user Fixed model names in benchmark table. v0.1.2 | 2026-06-12T14:15:37.330Z | user Updated README with SkillsBench benchmark results. v0.1.1 | 2026-06-02T","codeSnippets":[],"executableExamples":[{"language":"text","snippet":"ls ~/.claude/skills/Lens/SKILL.md\nls ~/.claude/skills/LEAP/SKILL.md"},{"language":"text","snippet":"> npx skills add agentsope/SkillAlchemy/skills/Lens\n> npx skills add agentsope/SkillAlchemy/skills/LEAP\n>"},{"language":"text","snippet":"quick    — 快速原型，3 agent，~5-8 min，跳过验证\nstandard — 日常使用（默认），4-5 agent，~15-20 min\ndeep     — 发布级，6-8 agent，~25-35 min，强制验证 + 双审核\n没说的话默认 standard。"},{"language":"text","snippet":"◆ 任务简报\n\n▸ 需求    蒸馏「张雪峰」→ persona skill\n▸ 流程    Lens → A 分支（7 Stage + 2 Gate）\n          ├─ Research Swarm  4-5 agent 并行研究\n          ├─ Exemplar        find-skills 在线检索 + 自动评分\n          └─ Compile         编译 + 自评 + 验证 + 清理\n▸ 深度    standard · ~15-20 min\n▸ 交互    步步确认（2 次暂停）\n\n> 确认，按 standard 跑\n> 换成 deep，研究更深入、验证更严格、双 agent 交叉审核\n> 一路默认跑完，中间别问我了，全部默认值到底\n> 先只要 Lens 看看维度，不生成 skill"},{"language":"text","snippet":"◆ Lens 分析完成 · N 个维度\n\n  [维度名]    [维度名]    [维度名]\n  [维度名]    [维度名]    [维度名]\n  ...\n\n▸ 意图    distill_persona / distill_method / fuse_skills\n\n> 确认，进入 [distill / fuse] 管线继续\n> 展开看看完整的 Lens 分析原文，每个维度的细节\n> 补一个 XX 维度，重新分析一遍\n> 就停在这，我消化一下 Lens 的结果，不继续了"},{"language":"text","snippet":"调 LEAP：\n  \"distill [target]，depth [depth]。\n   只到 research plan（stop_after_stage: 3），\n   输出到 <项目根目录>/output/<target>-skill/。\""}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: SkillAlchemy\ndescription: |\n  SkillAlchemy — 一念落地，万象成形。输入任意想法或蒸馏目标，输出可安装的 SKILL.md。\n  内部编排 Lens（看清问题）和 LEAP（执行蒸馏/融合）。用户唯一入口。\n  Use when 用户说「蒸馏」「生成 skill」「融合」「我想做 X 但不知道从哪下手」。\nversion: v1.0\n---\n\n# Skill-Alchemy · 一念落地，万象成形\n\n你是 SkillAlchemy。编排两个子 skill：Lens 看清，LEAP 落地。\n你自己不蒸馏、不融合——只做编排。**所有用户交互由你负责，LEAP 不跟用户说话。**\n\n## 前置检查\n\n```\nls ~/.claude/skills/Lens/SKILL.md\nls ~/.claude/skills/LEAP/SKILL.md\n```\n\n**如果缺少任何一个，告诉用户：**\n\n> SkillAlchemy 需要两个依赖才能运行，请先安装：\n>\n> ```\n> npx skills add agentsope/SkillAlchemy/skills/Lens\n> npx skills add agentsope/SkillAlchemy/skills/LEAP\n> ```\n>\n> 或者去 https://skills.sh 搜索 Lens 和 LEAP 安装。\n>\n> 装好之后回来找我继续。\n\n---\n\n## 编排流程\n\n### Phase 0: 确认深度 + 任务简报\n\n先确认 depth。用户没说就问一句：\n\n```\nquick    — 快速原型，3 agent，~5-8 min，跳过验证\nstandard — 日常使用（默认），4-5 agent，~15-20 min\ndeep     — 发布级，6-8 agent，~25-35 min，强制验证 + 双审核\n没说的话默认 standard。\n```\n\n用户给了深度后，**展示任务简报：**\n\n```\n◆ 任务简报\n\n▸ 需求    蒸馏「张雪峰」→ persona skill\n▸ 流程    Lens → A 分支（7 Stage + 2 Gate）\n          ├─ Research Swarm  4-5 agent 并行研究\n          ├─ Exemplar        find-skills 在线检索 + 自动评分\n          └─ Compile         编译 + 自评 + 验证 + 清理\n▸ 深度    standard · ~15-20 min\n▸ 交互    步步确认（2 次暂停）\n\n> 确认，按 standard 跑\n> 换成 deep，研究更深入、验证更严格、双 agent 交叉审核\n> 一路默认跑完，中间别问我了，全部默认值到底\n> 先只要 Lens 看看维度，不生成 skill\n```\n\n根据实际任务替换内容。确认后进 Phase 1。如果用户一开始就指定了 depth，跳过询问直接出简报。\n\n**「一路默认」模式：** 用户在任何节点说「一路默认」→ 跳过当前及后续所有交互，全部 standard 默认值跑完。\n\n---\n\n### Phase 1: Lens 分析\n\n调 Lens，输入用户原话。Lens 不向用户提问，直接输出增强版 description。\n\n**Lens 完成后，展示维度摘要（不放全文，太长）：**\n\n```\n◆ Lens 分析完成 · N 个维度\n\n  [维度名]    [维度名]    [维度名]\n  [维度名]    [维度名]    [维度名]\n  ...\n\n▸ 意图    distill_persona / distill_method / fuse_skills\n\n> 确认，进入 [distill / fuse] 管线继续\n> 展开看看完整的 Lens 分析原文，每个维度的细节\n> 补一个 XX 维度，重新分析一遍\n> 就停在这，我消化一下 Lens 的结果，不继续了\n```\n\n确认后进 Phase 2。提了修改意见 → 重新调 Lens 带上反馈。\n「一路默认」已激活 → 跳过，直接进 Phase 2。\n\n---\n\n### Phase 2: 路由判断\n\n| Lens 意图 | 动作 |\n|-----------|------|\n| distill | → Phase 3a（A 分支：蒸馏管线） |\n| fuse | → Phase 3b（B 分支：融合管线） |\n| decompose | 停。展示 Lens 输出，问是否继续 |\n| 无法判断 | 问用户：蒸馏还是融合？ |\n\n---\n\n### Phase 3: 执行\n\n**所有输出落在当前项目根目录的 `output/` 下。**\n调 LEAP 时用绝对路径指定输出位置（以实际项目路径为准）。\n\n#### 3a. Distill 路线（2 步，1 次确认）\n\n**Step 1: 生成 research plan。**\n```\n调 LEAP：\n  \"distill [target]，depth [depth]。\n   只到 research plan（stop_after_stage: 3），\n   输出到 <项目根目录>/output/<target>-skill/。\"\n```\n\nLEAP 跑完 Stage 1-3 后停止。读取 `research_plan.json`：\n\n```\n◆ Research Plan · N agents\n\n  R1  [维度名]\n      [搜索方向一句话]\n\n  R2  [维度名]\n      [搜索方向一句话]\n\n  ...\n\n> 确认，按这个计划启动 N 个 agent 并行研究\n> 加一个 R[n] 专门研究 XX 方向，补上缺失的维度\n> 删掉 R[n]，这个维度我不太关心，省点资源\n> 换成 quick 快速跑，3 个 agent 够了我赶时间\n```\n\n**Step 2: 研究 + exemplar + 编译（无交互，直接跑完）。**\n```\n调 LEAP：\n  \"从 Stage 4 继续 distill [target]，\n   research_plan 已确认，\n   输出到 <项目根目录>/output/<target>-skill/。\"\n```\n\nLEAP 执行 Stage 4-7 + Gate 1-2，全自动完成：\nResearch Swarm → Exemplar Discovery（find-skills + score_skill 自动评分择优）→ Synthesis → Compile → Validate。\n\n完成后清理中间产物：\n- 删除 `references/exemplar_candidates.json`（临时评分文件）\n- 删除 `references/exemplars/`（中间参照副本）\n- 删除空 `validation/`（standard 模式不跑 Phase 8）\n- 保留 `R*.md`（研究证据）、`intermed"},{"path":"skills/agentsop-agent-topology-selection/SKILL.md","content":"---\nname: agentsop-agent-topology-selection\nversion: 0.1.0\ndescription: >-\n  Cross-framework enhancement overlay for choosing a multi-agent topology BEFORE writing any\n  agent. A binary-question rubric — is single-agent + tools enough? do agents need to know\n  about each other? does the output need one voice? — maps the answer to single-agent /\n  supervisor / swarm / sequential / hierarchical. Activates when a coder agent is tempted to\n  \"split the work into roles\" or reaches for a multi-agent framework. Encodes the *selection\n  rubric* that the per-framework skills assume but never surface. Search keywords: when to\n  use multi-agent, single vs multi agent, do I need multiple agents, supervisor vs swarm,\n  multi-agent vs single agent, agent team design.\noverlay: true\ncross_links: [crewai, langgraph, bounded-loop]\n---\n\n# Multi-Agent Topology Selection · SOP (ENHANCE overlay)\n\n> Overlay posture: this skill decides *whether and which* topology. It does not\n> teach the API — descend to `[[crewai]]` or `[[agentsop-langgraph]]` for that. Every\n> load-bearing claim carries an inline source tag resolving in\n> `references/R1-source-evidence.md`.\n\n---\n\n## 1. 何时激活 (When to Activate)\n\nActivate when **any** of the following fire:\n\n- The task description contains \"team of agents\", \"researcher + writer + reviewer\",\n  \"manager agent\", \"agents that hand off\", \"split this into roles\", or \"multi-agent\".\n- A coder agent is about to instantiate ≥2 agents (CrewAI `Agent(...)` × N,\n  LangGraph supervisor/swarm, OpenAI Swarm handoffs) and has **not yet** justified\n  why a single agent with tools is insufficient.\n- Someone is choosing between CrewAI `Process.sequential` vs `Process.hierarchical`,\n  or LangGraph supervisor vs swarm vs hierarchical-teams, and wants the *rubric*,\n  not the syntax.\n- A multi-agent system is over budget on tokens/latency and the question is \"can we\n  collapse agents back into one?\".\n\nDo **not** activate for: a single LLM call, a one-shot RAG query, or a fixed\ntool-call pipeline with no role separation. Those are the single-agent baseline\nthis skill defends.\n\n> Mental check: *\"An agent needs agency, otherwise it's just another script.\"*\n> — João Moura, CrewAI founder `[[crewai · §1.3]]`. If you can write the control\n> flow in `if/else`, you do not need multiple agents — you need one agent (or a\n> graph) with explicit edges.\n\n---\n\n## 2. 核心心智模型 (Core Mental Model)\n\n**Most \"multi-agent\" problems are single-agent + tools.** Add agents only when\n*context isolation* or *parallel expertise* genuinely demands it.\n\n> \"Single-agent is right for approximately 80% of cases; the trap is reaching for\n> multi-agent because it sounds more capable.\" `[[crewai · DC-1]]`\n\nTwo — and only two — forces justify a second agent:\n\n1. **Context isolation.** One agent's working context would pollute another's\n   (a critic that must not see its own draft's rationalisations; a tool-heavy\n   sub-task whose 40 intermediate tool calls should not bloat the main thread).\n   Spl"},{"path":"skills/agentsop-aider/SKILL.md","content":"---\nname: agentsop-aider\nversion: 1.0.0\ndescription: >-\n  SOP for terminal-based, git-native AI pair programming with Aider (git work-tree + tree-sitter repo-map + edit-format + human-in-loop REPL). Use when editing code in an existing git repo via an LLM, when you need to converge a change to 2-5 files, pick an edit format that fits the model, run architect+editor mode, or wire an auto-test loop.\ndomain: terminal-based AI pair programming, git-native code editing\nsource: aider.chat docs + Paul Gauthier's blog + leaderboards\naudience: coder-agents and human engineers who edit code via LLMs\n---\n\n# Aider SOP — 终端结对编程的操作系统\n\n> 一句话：Aider 是“**git 工作树 + tree-sitter 仓库地图 + 编辑格式 + 人类在环 REPL**”的四元组。理解这四个原语，剩下的都是配置。\n\n## 1. 何时激活本技能\n\n下列任一情形成立时，按本 SOP 进入 Aider 工作模式：\n\n- 任务是**编辑已有 git 仓库**里的代码（不是从零起项目）。\n- 你能把改动范围**收敛到 2–5 个文件**，或愿意先用 `/ask` 让模型借助 repo-map 把范围找出来。\n- 你需要**逐步可回滚**的修改历史（每次编辑一个 commit，`/undo` 一步回退）。\n- 你在**终端**里工作（tmux / 远程 ssh / CI）；或者你在写一个把 Aider 当子进程驱动的 agent。\n- 你关心**编辑格式对模型质量的影响**（diff / udiff / whole / patch 的选择问题）。\n- 你需要 BYOM（自带模型），跑本地 LLM 或非主流厂商。\n\n**不应激活的反面信号**：见 §6 反模式与边界。\n\n## 2. 核心心智模型\n\n### 2.1 四个原语\n\n```\n+------------------+   +------------------+   +------------------+   +------------------+\n| 1. Git working   |   | 2. Tree-sitter   |   | 3. Edit format   |   | 4. REPL loop     |\n|    tree          |   |    repo-map      |   |    (wire proto)  |   |    (你在环里)    |\n|                  |   |                  |   |                  |   |                  |\n| - per-edit       |   | - symbol-level   |   | - diff / udiff   |   | - /ask /code     |\n|   commit         |   |   summary        |   |   / whole /      |   |   /architect     |\n| - /undo          |   | - PageRank over  |   |   patch          |   | - 每轮人手确认   |\n| - dirty 文件     |   |   import graph   |   | - 模型适配选择   |   | - 不自主         |\n|   先 commit 再编 |   | - 动态预算       |   | - JSON 是反模式  |   |                  |\n+------------------+   +------------------+   +------------------+   +------------------+\n```\n\n四者缺一不可：\n- 去掉 git → 失去回滚与审计；\n- 去掉 repo-map → 大仓库里 LLM 找不到正确文件（SWE-Bench Lite 上 repo-map 让 Aider 70.3% 命中正确文件 [aider.chat/2024/05/22/swe-bench-lite.html]）；\n- 用错 edit format → 出现“lazy coding”、SEARCH 块找不到、JSON 句法破坏（udiff 在 GPT-4 Turbo 上把 refactor 基准从 20% 拉到 61% [aider.chat/2023/12/21/unified-diffs.html]）；\n- 放弃 REPL → 退化为自主 agent，但 Aider 在 SWE-Bench 上恰好证明“人在环 + 多次尝试”比纯自主链路更稳。\n\n### 2.2 LLM 看到的上下文分三层（优先级递减）\n\n| 层 | 内容 | 谁能改 |\n|---|---|---|\n| 系统提示 + 编辑格式说明 | Aider 固化 | Aider |\n| 只读上下文 | repo-map + `/read` 文件 + CONVENTIONS.md | 你（通过 `--read`） |\n| 读写上下文 | `/add` 的文件 | LLM **只能编辑**这里的文件 |\n\n> **铁律**：LLM 只允许编辑 `/add`-ed 的文件。这是 Aider 的安全边界。模型“改错了文件”几乎总是因为该文件没 `/add` 或你 `/add` 了太多无关文件。\n\n### 2.3 上下文预算（25k 信号阈）\n\n> \"Above about 25k tokens of context, most models start to become distracted.\" [aider.chat/docs/troubleshooting/edit-errors.html]\n\n把这条当硬约束：超过 25k tokens，编辑准确率断崖式下降。`/tokens` 持续监控。\n\n### 2.4 Repo-map 不是 RAG\n\nrepo-map 是 **tree-sitter 提取的符号清单**（类、函数、签名），用 PageRank 在源文件依赖图上排序，**塞给 LLM 当地图**。这不是 embedd"},{"path":"skills/agentsop-bio-fraud-forensics/SKILL.md","content":"---\nname: agentsop-bio-fraud-forensics\ndomain: research-integrity\ntrigger_keywords:\n  - \"data fraud / image manipulation\"\n  - \"Western blot duplication / splicing\"\n  - \"GRIM / statcheck / impossible statistics\"\n  - \"paper mill / tortured phrases\"\n  - \"PubPeer / Retraction Watch verification\"\ndescription: >-\n  Screens biomedical / life-science papers for signs of data fabrication, image\n  manipulation, and statistical anomalies, using the detection techniques distilled\n  from the field's canonical exposure platforms (PubPeer, Data Colada, Science\n  Integrity Digest, For Better Science) and tools (ImageTwin/Proofig, statcheck,\n  GRIM/GRIMMER, Problematic Paper Screener, Seek & Blastn). Use when asked to check\n  a paper/figure for image duplication, blot splicing, impossible statistics, paper-mill\n  or tortured-phrase signals, research integrity, or \"is this data faked\"; or when a\n  user shares a figure, Western blot, supplementary dataset, or DOI and asks whether it\n  looks manipulated. Reports observable anomalies as questions for clarification — it\n  never accuses anyone of fraud.\nversion: 1.0.0\n---\n\n# Bio-Fraud Forensics · 生物医学论文数据造假筛查\n\nA screening methodology for life-science papers. It reverse-engineers how real cases\nwere caught — the exact panels compared, the transform applied, the statistic recomputed —\nand turns that into a reproducible per-paper checklist. It is a **detective's lens, not a\nverdict machine**: every output stays at \"observed anomaly\" or \"question for the authors,\"\nbecause red flag ≠ proof and an accusation can end a career.\n\n## Activation Rules\n\n**Trigger when:**\n- \"Check this paper / figure / Western blot for manipulation,\" \"does this data look faked,\" \"screen for image duplication.\"\n- A user shares a figure, blot, microscopy panel, supplementary `.xlsx`, or a DOI and asks if it's trustworthy.\n- \"Is this a paper mill?\", \"tortured phrases,\" \"are these statistics possible,\" \"run GRIM/statcheck on this.\"\n- \"Where do I check if this paper has been flagged / retracted?\" (verification routing).\n- Asked to draft a PubPeer-grade, reproducible image/data integrity comment.\n\n**Do NOT trigger when:**\n- The user wants a scientific peer review of validity/novelty (use a peer-review skill) rather than an integrity screen.\n- The user asks you to publicly accuse a named person of fraud, or to write an accusation/social post (refuse — see Boundary Rules).\n- The task is general statistics help or figure-making with no integrity question.\n- The paper is non-biomedical and the request is about a domain whose fraud signatures differ (physics/CS); say so and scope down.\n\n## Agentic Protocol\n\nRun this as a chain-of-steps. Cheapest, fastest signals first; the expensive image/stat\nforensics last (they tell you *where* to dig is often answered for free by the cheap checks).\n\n**Step 1 — Scope & status.** Identify the input: single figure, full paper, supplementary\ndataset, or a batch. Run the status cascade in parallel (it's free and may hand you t"},{"path":"skills/agentsop-bounded-loop/SKILL.md","content":"---\nname: agentsop-bounded-loop\nversion: 0.1.0\ndescription: >-\n  Universal discipline for any LM-driven loop — agent retries, plan-act-observe, multi-agent\n  handoffs, optimiser passes, test-fix cycles. Encodes the one rule every framework\n  documents quietly and every team relearns expensively: the LM in the loop is NEVER a\n  reliable terminator. Termination must be provided by an explicit counter + exit predicate\n  + stagnation signal + escalation path that live OUTSIDE the LM's control. This is a tool-\n  level, framework-agnostic skill. It maps onto LangGraph (recursion_limit + state counter +\n  interrupt), CrewAI (max_iter + max_rpm + human_input), Claude / OpenAI SDKs\n  (max_iterations + tool_use_budget), DSPy (declared evaluation budget), Aider (REPL +\n  explicit retry cap), and AutoGen (max_consecutive_auto_reply). Search keywords: infinite\n  loop, recursion limit, recursion_limit, GraphRecursionError, max iterations, max_iter,\n  agent stuck, agent won't stop, runaway agent, ReAct loop not terminating, agent repeating\n  itself.\n---\n\n# bounded-loop · O7\n\n> Source posture: every load-bearing claim is cited inline with a short tag\n> resolved against `references/R1-source-evidence.md` and\n> `references/R2-cross-framework.md`. Examples cite the real GitHub issues\n> they're distilled from.\n\n---\n\n## 1. 何时激活 (Activation Rules)\n\nActivate this skill when **any** of the following is true:\n\n- The task involves a workflow that contains a **cycle** — tool-call → reflect\n  → retry, plan → act → observe → re-plan, draft → critique → revise,\n  test → fix → re-test.\n- The user is hitting a framework's \"loop too deep\" error:\n  `GRAPH_RECURSION_LIMIT` (LangGraph), `MaxIterationsExceeded` (LangChain\n  `AgentExecutor`), \"agent exceeded max_iter\" (CrewAI), `max_turns reached`\n  (OpenAI Agents SDK), `stop_reason=\"max_tokens\"` mid-tool-use (Anthropic).\n- The user proposes \"let's just raise the limit\" / \"set max_iter to 100\" /\n  `recursion_limit=200` — this is the canonical anti-pattern this skill\n  exists to prevent.\n- The user is building a **multi-agent** system with delegation, handoff,\n  or supervisor patterns — these are exposure-multipliers for unbounded\n  loops (see `[gh/crewai-330]`).\n- The user is building an **optimiser / evaluator loop** (DSPy, AutoEval,\n  RLHF, self-refining agent) where \"stop when good enough\" is the\n  termination criterion — this is *never* sufficient on its own.\n- The user wants a **test-fix loop**, **self-healing code agent**, or\n  **iterative refinement** workflow — every code-agent in production\n  (Cursor, Aider, Devin, Claude Code) ships with an explicit step budget.\n\nDo **not** activate for: single LLM calls, one-shot RAG queries, stateless\ntool pipelines, or flows where the cycle is provably bounded by data (e.g.,\n\"iterate once per row in this fixed list\").\n\n---\n\n## 2. 核心心智模型 (Core Mental Model)\n\n**Every loop body must produce a state change that proves progress — and\nthe proof must be checkable without calling another LM.**\n\n"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"SkillAlchemy — 一念落地，万象成形。输入任意想法或蒸馏目标，输出可安装的 SKILL.md。 内部编排 Lens（看清问题）和 LEAP（执行蒸馏/融合）。用户唯一入口。 Use when 用户说「蒸馏」「生成 skill」「融合」「我想做 X 但不知道从哪下手」。 Skill: Skill Alchemy Main Owner: agentsope Summary: SkillAlchemy — 一念落地，万象成形。输入任意想法或蒸馏目标，输出可安装的 SKILL.md。 内部编排 Lens（看清问题）和 LEAP（执行蒸馏/融合）。用户唯一入口。 Use when 用户说「蒸馏」「生成 skill」「融合」「我想做 X 但不知道从哪下手」。 Tags: latest:0.1.3 Version history: v0.1.3 | 2026-06-15T11:41:28.437Z | user Fixed model names in benchmark table. v0.1.2 | 2026-06-12T14:15:37.330Z | user Updated README with SkillsBench benchmark results. v0.1.1 | 2026-06-02T","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":2053,"uniquenessScore":49,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-11T11:59:00.645Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-11T11:59:00.645Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-11T15:14:21.633Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}