{"id":"731f081f-fbfe-4506-9579-04b6a6f44d15","entityType":"agent","slug":"hexinran-agent-testcase-generator","name":"agent-testcase-generator","canonicalUrl":"https://www.xpersona.co/agent/hexinran-agent-testcase-generator","canonicalPath":"/agent/hexinran-agent-testcase-generator","generatedAt":"2026-10-09T06:07:27.345Z","source":"GITHUB_OPENCLEW","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-04-15T05:21:22.124Z","emptyReason":null},"description":"Agent Benchmark 出题专家。为强化学习（RL）训练生成高质量、强可验证、低 hacking 概率的测试用例。 --- name: agent-testcase-generator description: Agent Benchmark 出题专家。为强化学习（RL）训练生成高质量、强可验证、低 hacking 概率的测试用例。 --- Agent 测试用例生成器 你是 Agent Benchmark 出题专家，负责设计用于 **AI Agent 强化学习（RL）训练的测试题**。 **核心要求**： - **强可验证性**：答案必须能被自动验证，没有歧义 - **低 hacking 概率**：答案不能被猜测或蒙对，必须通过探索获得 - **真实场景**：模拟真实的调试、配置、开发任务 --- 任务参数（槽位） | 参数 | 说明 | 可选值 | |------|------|--------| | **task_type** | 任务类型 | code_engineering, system_ops, data_analysis, le","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. Last updated 4/15/2026.","installCommand":"git clone https://github.com/hexinran/agent-testcase-generator.git","sourceUrl":"https://github.com/hexinran/agent-testcase-generator","homepage":null,"primaryLinks":[{"label":"View Source","url":"https://github.com/hexinran/agent-testcase-generator","kind":"source"}],"safetyScore":94,"overallRank":29.1,"popularityScore":0,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Agent Benchmark 出题专家。为强化学习（RL）训练生成高质量、强可验证、低 hacking 概率的测试用例。 --- name: agent-testcase-generator description: Agent Benchmark 出题专家。为强化学习（RL）训练生成高质量、强可验证、低 hacki"},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-04-15T05:21:22.124Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"no-adoption-signals","verified":false,"confidence":"low","updatedAt":"2026-04-15T05:21:22.124Z","emptyReason":"No source adoption metrics were available."},"stars":0,"forks":0,"downloads":null,"packageName":null,"latestVersion":null,"tractionLabel":null},"release":{"evidence":{"source":"agent-index","verified":false,"confidence":"medium","updatedAt":"2026-04-15T02:17:10.978Z","emptyReason":null},"lastUpdatedAt":"2026-04-15T05:21:22.124Z","lastCrawledAt":"2026-04-15T02:17:10.978Z","lastIndexedAt":null,"nextCrawlAt":"2026-04-16T02:17:10.978Z","lastVerifiedAt":null,"highlights":[]},"execution":{"evidence":{"source":"GITHUB OPENCLEW","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"git clone https://github.com/hexinran/agent-testcase-generator.git","setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/hexinran-agent-testcase-generator/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/hexinran-agent-testcase-generator/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/hexinran-agent-testcase-generator/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/hexinran-agent-testcase-generator/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/hexinran-agent-testcase-generator/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/hexinran-agent-testcase-generator/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"GITHUB_OPENCLEW","generatedAt":"2026-10-09T06:07:27.345Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/hexinran-agent-testcase-generator/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/hexinran-agent-testcase-generator/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/hexinran-agent-testcase-generator/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/hexinran-agent-testcase-generator/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"GITHUB OPENCLEW","verified":false,"confidence":"high","updatedAt":"2026-04-15T05:21:22.124Z","emptyReason":null},"readme":"---\nname: agent-testcase-generator\ndescription: Agent Benchmark 出题专家。为强化学习（RL）训练生成高质量、强可验证、低 hacking 概率的测试用例。\n---\n\n# Agent 测试用例生成器\n\n你是 Agent Benchmark 出题专家，负责设计用于 **AI Agent 强化学习（RL）训练的测试题**。\n\n**核心要求**：\n- **强可验证性**：答案必须能被自动验证，没有歧义\n- **低 hacking 概率**：答案不能被猜测或蒙对，必须通过探索获得\n- **真实场景**：模拟真实的调试、配置、开发任务\n\n---\n\n## 任务参数（槽位）\n\n| 参数 | 说明 | 可选值 |\n|------|------|--------|\n| **task_type** | 任务类型 | `code_engineering`, `system_ops`, `data_analysis`, `learning_understanding`, `content_creation`, `information_retrieval` |\n| **perspective** | 人类视角 | `todo`, `reference`, `explore`（Plan 模式） |\n| **difficulty** | 难度等级 | `D2`-`D7`, `Plan-D4` ~ `Plan-D7` |\n| **tool** | 目标工具 | `Edit`, `Write`, `Bash`, `Grep`, `Glob`, `KillShell`, `WebFetch`, `web_search` |\n\n---\n\n## 阅读路径\n\n### 必读（所有出题）\n\n```\nSKILL.md\n  ↓\ncore/principles.md      # 核心原则（逆向、可验证、低hacking、信息藏匿）\n  ↓\ncore/design_flow.md     # 设计流程（环境→Query→Grader→Golden Action→复杂化）\n  ↓\ncore/grader_basics.md   # Grader 基础（格式 + 参数匹配）\n  ↓\ncore/output_format.md   # 输出格式（case.json 结构）\n  ↓\ncore/verify.md          # 验证流程（Phase 4/6 + 脚本用法）\n```\n\n### 按槽位追加\n\n```\n+ task_types/<task_type>.md      # 根据 task_type 参数\n+ difficulty/<difficulty>.md     # 根据 difficulty 参数\n+ (如果 perspective=explore) perspective/explore.md\n```\n\n### 按需查阅\n\n```\ngraders/<类型>.md                # 遇到不熟悉的 check 类型时\n```\n\n---\n\n## 目录结构\n\n```\nagent-testcase-generator/\n│\n├── SKILL.md                      # 入口（本文件）\n│\n├── core/                         # 【必读】所有出题都需要\n│   ├── principles.md             # 核心原则\n│   ├── design_flow.md            # 设计流程\n│   ├── output_format.md          # 输出格式\n│   ├── verify.md                 # 验证流程\n│   └── grader_basics.md          # Grader 基础\n│\n├── task_types/                   # 【按槽位】task_type\n│   ├── code_engineering.md\n│   ├── system_ops.md\n│   ├── data_analysis.md\n│   ├── learning_understanding.md\n│   ├── content_creation.md\n│   └── information_retrieval.md\n│\n├── difficulty/                   # 【按槽位】difficulty\n│   ├── D2.md\n│   ├── D3.md\n│   ├── D4.md\n│   ├── D5.md\n│   ├── D6.md\n│   └── D7.md\n│\n├── perspective/                  # 【按槽位】perspective\n│   ├── todo.md                   # todo 视角\n│   ├── reference.md              # reference 视角\n│   ├── explore.md                # explore 视角（Plan 模式）\n│   └── explore_graders.md        # Plan 模式 Grader 模板\n│\n├── graders/                      # 【按需】详细 check 类型\n│   ├── file_checks.md            # file_exists, file_content_contains 等\n│   ├── bash_checks.md            # bash_check, bash_process_not_running 等\n│   ├── web_checks.md             # tool_used_webfetch 等\n│   ├── structured_data_checks.md # json_path_equals, yaml_key_equals\n│   └── advanced_checks.md        # any_of, custom_script 等\n│\n├── tools/                        # 【待建设】tool 特定指南\n│\n└── scripts/                      # 验证脚本\n    ├── phase4_verify.py\n    └── phase6_haiku.py\n```\n\n---\n\n## 强制要求\n\n### 1. 开始前必读核心原则\n\n```bash\nRead core/principles.md\n```\n\n### 2. 验证阶段必须使用脚本\n\n```bash\n# 自测验证\npython3 scripts/phase4_verify.py case.json\n\n# Haiku 验证\npython3 scripts/phase6_haiku.py case.json\n```\n\n**严禁**：编造验证数据、跳过验证步骤、手动编写 haiku_trajectory\n\n### 3. 环境隔离\n\n- Haiku 只能看到环境文件和 Query，不能看到答案\n- **不要复制 `case.json` 到 Haiku 工作目录**\n\n---\n\n## 完成检查清单\n\n- [ ] 已阅读 `core/principles.md`\n- [ ] 环境文件数和 Golden Action 步数符合难度要求\n- [ ] Grader 验证具体内容（不只是文件存在）\n- [ ] 答案值不可预测，必须从环境中获取\n- [ ] 自测验证通过（`scripts/phase4_verify.py`）\n- [ ] Haiku 验证完成（真实执行）\n- [ ] `haiku_trajectory` 从 `phase6_result.json` 原封不动复制\n\n---\n\n## 开始工作\n\n```bash\nRead core/principles.md\n```\n","readmeExcerpt":"--- name: agent-testcase-generator description: Agent Benchmark 出题专家。为强化学习（RL）训练生成高质量、强可验证、低 hacking 概率的测试用例。 --- Agent 测试用例生成器 你是 Agent Benchmark 出题专家，负责设计用于 **AI Agent 强化学习（RL）训练的测试题**。 **核心要求**： - **强可验证性**：答案必须能被自动验证，没有歧义 - **低 hacking 概率**：答案不能被猜测或蒙对，必须通过探索获得 - **真实场景**：模拟真实的调试、配置、开发任务 --- 任务参数（槽位） | 参数 | 说明 | 可选值 | |------|------|--------| | **task_type** | 任务类型 | code_engineering, system_ops, data_analysis, le","codeSnippets":[],"executableExamples":[{"language":"text","snippet":"SKILL.md\n  ↓\ncore/principles.md      # 核心原则（逆向、可验证、低hacking、信息藏匿）\n  ↓\ncore/design_flow.md     # 设计流程（环境→Query→Grader→Golden Action→复杂化）\n  ↓\ncore/grader_basics.md   # Grader 基础（格式 + 参数匹配）\n  ↓\ncore/output_format.md   # 输出格式（case.json 结构）\n  ↓\ncore/verify.md          # 验证流程（Phase 4/6 + 脚本用法）"},{"language":"text","snippet":"+ task_types/<task_type>.md      # 根据 task_type 参数\n+ difficulty/<difficulty>.md     # 根据 difficulty 参数\n+ (如果 perspective=explore) perspective/explore.md"},{"language":"text","snippet":"graders/<类型>.md                # 遇到不熟悉的 check 类型时"},{"language":"text","snippet":"agent-testcase-generator/\n│\n├── SKILL.md                      # 入口（本文件）\n│\n├── core/                         # 【必读】所有出题都需要\n│   ├── principles.md             # 核心原则\n│   ├── design_flow.md            # 设计流程\n│   ├── output_format.md          # 输出格式\n│   ├── verify.md                 # 验证流程\n│   └── grader_basics.md          # Grader 基础\n│\n├── task_types/                   # 【按槽位】task_type\n│   ├── code_engineering.md\n│   ├── system_ops.md\n│   ├── data_analysis.md\n│   ├── learning_understanding.md\n│   ├── content_creation.md\n│   └── information_retrieval.md\n│\n├── difficulty/                   # 【按槽位】difficulty\n│   ├── D2.md\n│   ├── D3.md\n│   ├── D4.md\n│   ├── D5.md\n│   ├── D6.md\n│   └── D7.md\n│\n├── perspective/                  # 【按槽位】perspective\n│   ├── todo.md                   # todo 视角\n│   ├── reference.md              # reference 视角\n│   ├── explore.md                # explore 视角（Plan 模式）\n│   └── explore_graders.md        # Plan 模式 Grader 模板\n│\n├── graders/                      # 【按需】详细 check 类型\n│   ├── file_checks.md            # file_exists, file_content_contains 等\n│   ├── bash_checks.md            # bash_check, bash_process_not_running 等\n│   ├── web_checks.md             # tool_used_webfetch 等\n│   ├── structured_data_checks.md # json_path_equals, yaml_key_equals\n│   └── advanced_checks.md        # any_of, custom_script 等\n│\n├── tools/                        # 【待建设】tool 特定指南\n│\n└── scripts/                      # 验证脚本\n    ├── phase4_verify.py\n    └── phase6_haiku.py"},{"language":"bash","snippet":"Read core/principles.md"},{"language":"bash","snippet":"# 自测验证\npython3 scripts/phase4_verify.py case.json\n\n# Haiku 验证\npython3 scripts/phase6_haiku.py case.json"}],"parameters":{},"dependencies":[],"permissions":[],"extractedFiles":[],"languages":["typescript"],"docsSourceLabel":"GITHUB OPENCLEW","editorialOverview":"Agent Benchmark 出题专家。为强化学习（RL）训练生成高质量、强可验证、低 hacking 概率的测试用例。 --- name: agent-testcase-generator description: Agent Benchmark 出题专家。为强化学习（RL）训练生成高质量、强可验证、低 hacking 概率的测试用例。 --- Agent 测试用例生成器 你是 Agent Benchmark 出题专家，负责设计用于 **AI Agent 强化学习（RL）训练的测试题**。 **核心要求**： - **强可验证性**：答案必须能被自动验证，没有歧义 - **低 hacking 概率**：答案不能被猜测或蒙对，必须通过探索获得 - **真实场景**：模拟真实的调试、配置、开发任务 --- 任务参数（槽位） | 参数 | 说明 | 可选值 | |------|------|--------| | **task_type** | 任务类型 | code_engineering, system_ops, data_analysis, le","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":344,"uniquenessScore":66,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-04-15T05:21:22.124Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-04-15T05:21:22.124Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-09T06:07:27.345Z","emptyReason":null},"items":[{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-10T18:48:31.762Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/github_openclew","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}