{"id":"84370fab-2798-49c1-9493-4a2ecebb257f","entityType":"agent","slug":"clawhub-harrylabsj-skillopt","name":"SkillOpt","canonicalUrl":"https://www.xpersona.co/agent/clawhub-harrylabsj-skillopt","canonicalPath":"/agent/clawhub-harrylabsj-skillopt","generatedAt":"2026-10-10T21:43:01.439Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-10T17:44:17.612Z","emptyReason":null},"description":"Train, evaluate, and improve Agent skill files as reusable external capabilities. Use when a user wants to optimize SKILL.md, prompt procedures, OpenClaw/Her... Skill: SkillOpt Owner: harrylabsj Summary: Train, evaluate, and improve Agent skill files as reusable external capabilities. Use when a user wants to optimize SKILL.md, prompt procedures, OpenClaw/Her... Tags: agent:0.1.0, evaluation:0.1.0, latest:0.1.0, optimization:0.1.0, skillopt:0.1.0, skills:0.1.0 Version history: v0.1.0 | 2026-06-07T03:05:31.554Z | user Initial release: SkillOpt workflow for train/validation sk","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.3K downloads reported by the source. Last updated 10/10/2026.","installCommand":"clawhub skill install s17a8m9q4jybb46cv60h4fxard83hmsn:skillopt","sourceUrl":"https://clawhub.ai/harrylabsj/skillopt","homepage":"https://clawhub.ai/harrylabsj/skills/skillopt","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/harrylabsj/skillopt","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/harrylabsj/skills/skillopt","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":62,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Train, evaluate, and improve Agent skill files as reusable external capabilities. Use when a user wants to optimize SKILL.md, prompt procedures, OpenClaw/Her..."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-10T17:44:17.612Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T17:44:17.612Z","emptyReason":null},"stars":null,"forks":null,"downloads":1312,"packageName":null,"latestVersion":"0.1.0","tractionLabel":"1.3K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T17:44:17.612Z","emptyReason":null},"lastUpdatedAt":"2026-10-10T17:44:17.612Z","lastCrawledAt":"2026-10-10T17:44:17.612Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-11T17:44:17.612Z","lastVerifiedAt":null,"highlights":[{"version":"0.1.0","createdAt":"2026-06-07T03:05:31.554Z","changelog":"Initial release: SkillOpt workflow for train/validation skill optimization, rollout scoring, validation gates, and best_skill.md export.","fileCount":6,"zipByteSize":10959}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s17a8m9q4jybb46cv60h4fxard83hmsn:skillopt","setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-harrylabsj-skillopt/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-harrylabsj-skillopt/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-harrylabsj-skillopt/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-harrylabsj-skillopt/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-harrylabsj-skillopt/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-harrylabsj-skillopt/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-10T21:43:01.438Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-harrylabsj-skillopt/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-harrylabsj-skillopt/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-harrylabsj-skillopt/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-harrylabsj-skillopt/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-10T17:44:17.612Z","emptyReason":null},"readme":"Skill: SkillOpt\n\nOwner: harrylabsj\n\nSummary: Train, evaluate, and improve Agent skill files as reusable external capabilities. Use when a user wants to optimize SKILL.md, prompt procedures, OpenClaw/Her...\n\nTags: agent:0.1.0, evaluation:0.1.0, latest:0.1.0, optimization:0.1.0, skillopt:0.1.0, skills:0.1.0\n\nVersion history:\n\nv0.1.0 | 2026-06-07T03:05:31.554Z | user\n\nInitial release: SkillOpt workflow for train/validation skill optimization, rollout scoring, validation gates, and best_skill.md export.\n\nArchive index:\n\nArchive v0.1.0: 6 files, 10959 bytes\n\nFiles: agents/openai.yaml (235b), references/evaluation.md (2751b), scripts/skillopt.py (16975b), skill-card.md (1964b), SKILL.md (6956b), _meta.json (127b)\n\nFile v0.1.0:SKILL.md\n\n---\nname: skillopt\ndescription: Train, evaluate, and improve Agent skill files as reusable external capabilities. Use when a user wants to optimize SKILL.md, prompt procedures, OpenClaw/Hermes/Codex/Claude Code skills, agent workflows, skill factories, benchmark-driven skill iteration, rollout analysis, validation gates, best_skill.md export, or controlled self-evolving skills inspired by Microsoft SkillOpt.\n---\n\n# SkillOpt\n\n## Operating Idea\n\nTreat a skill document as trainable external state. Keep the target model, tools, and runtime fixed; optimize only the skill text through measured task rollouts, failure reflection, small edits, validation gating, and versioned export.\n\nDefault output is a deployable `best_skill.md` plus a short optimization report. Training may use many traces and candidate files; deployment should require only the final skill file.\n\n## Invariants\n\n- Preserve the original skill before editing.\n- Separate train and validation tasks. Never accept an edit based only on the examples used to propose it.\n- Prefer small, reviewable edits over full rewrites. Keep the skill's public contract stable unless the task suite proves the contract is wrong.\n- Score behavior, not eloquence. A prettier skill that does not improve validation is rejected.\n- Record rejected edits and the reason, then consult that buffer before proposing another edit.\n- Do not add model-specific hacks unless the target deployment is explicitly model-specific.\n- Do not leak validation answers into the skill. Validation data may guide accept/reject decisions, not become memorized instructions.\n\n## Run Directory\n\nCreate a run directory near the skill being optimized unless the user specifies another path:\n\n```text\nskillopt_runs/<target-skill-slug>/\n  source_skill.md\n  candidates/\n    candidate_000.md\n    candidate_001.md\n  tasks/\n    train.jsonl\n    val.jsonl\n  rollouts/\n    train/\n    val/\n  rejected_edits.md\n  best_skill.md\n  report.md\n```\n\nUse `scripts/skillopt.py` for deterministic run setup, JSONL validation, simple command-backed rollouts, score aggregation, validation gates, and report generation. Read `references/evaluation.md` when defining task schemas or scorers.\n\n## Workflow\n\n### 1. Define the Optimization Contract\n\nIdentify:\n\n- target skill path and deployment agents\n- target model/runtime/tool constraints to keep fixed during evaluation\n- success metric and acceptance threshold\n- task distribution the skill should serve\n- allowed edit budget, such as max 3 sections or max 25% changed lines per round\n\nIf no task suite exists, create a small proxy suite first, label it as proxy data, and tell the user that real production traces are needed for stronger conclusions.\n\n### 2. Build Train and Validation Sets\n\nRepresent each task as JSONL with an id, prompt, optional inputs, and a scorer. Keep validation examples independent and representative.\n\nMinimum split:\n\n- `train.jsonl`: failure discovery and edit proposal\n- `val.jsonl`: accept/reject gate\n\nFor fragile or high-stakes skills, add a hidden or holdout split outside the optimization loop and use it only for final reporting.\n\n### 3. Run Baseline Rollouts\n\nEvaluate the unmodified skill on train and validation tasks using the same target agent that will later deploy it.\n\nExamples:\n\n```bash\npython3 scripts/skillopt.py init --skill path/to/SKILL.md --out skillopt_runs/my-skill\npython3 scripts/skillopt.py validate-tasks skillopt_runs/my-skill/tasks/train.jsonl\npython3 scripts/skillopt.py run --tasks skillopt_runs/my-skill/tasks/val.jsonl --skill skillopt_runs/my-skill/source_skill.md --out skillopt_runs/my-skill/rollouts/val_baseline --agent-command \"hermes -s {skill_path} -z {prompt}\"\n```\n\nFor OpenClaw or any other agent, replace `--agent-command` with a command template that accepts `{skill_path}`, `{prompt}`, `{task_id}`, and optionally `{output_path}`.\n\n### 4. Reflect on Traces\n\nAnalyze successful and failed rollouts separately.\n\nFor each failure, classify the root cause:\n\n- missing procedure\n- wrong tool order\n- weak verification\n- ambiguous output contract\n- missing edge case\n- over-broad instruction\n- environment assumption\n- scoring mismatch\n\nExtract patterns across failures before editing. Do not chase one-off errors unless they reveal a generalizable instruction.\n\n### 5. Propose a Controlled Edit\n\nGenerate one candidate skill with a concise edit rationale:\n\n- add: missing guardrail, checklist, or workflow step\n- delete: harmful or distracting instruction\n- replace: ambiguous wording with operational criteria\n- reorder: move high-leverage instructions earlier\n\nKeep the candidate deployable as a normal skill. Avoid embedding run logs, benchmark answers, private traces, or optimizer notes in the final skill text.\n\n### 6. Gate on Validation\n\nRun the same validation set on the candidate. Accept only when the candidate beats the baseline by the configured threshold and does not introduce unacceptable regressions.\n\nDefault acceptance:\n\n- validation average improves by at least `0.02`\n- no critical task regresses from pass to fail\n- skill remains shorter or only grows for a clear procedural reason\n- output format and trigger metadata remain valid\n\nIf rejected, append a short note to `rejected_edits.md`:\n\n```text\n## candidate_003\nRejected because validation avg +0.00 and task val_docx_04 regressed.\nAvoid adding broad \"always rewrite\" instructions; they caused format drift.\n```\n\n### 7. Iterate\n\nRepeat rollout, reflection, candidate edit, and validation gate until:\n\n- validation score plateaus for 2 rounds\n- edit budget is exhausted\n- regressions become persistent\n- the skill is good enough for the user's target use\n\nTrack the best candidate, not merely the latest candidate.\n\n### 8. Export\n\nCopy the best accepted candidate to `best_skill.md`. If the user wants installation, replace or install the deployed skill only after showing the report summary.\n\nThe final report should include:\n\n- baseline train/validation scores\n- best candidate train/validation scores\n- accepted edits\n- rejected edit patterns\n- known overfitting risks\n- deployment instructions for OpenClaw, Hermes, or the current agent\n\n## Cross-Agent Notes\n\n- For Codex-style skills, keep required YAML frontmatter to `name` and `description`.\n- For Hermes, prefer standard `SKILL.md` folders and invoke with `hermes -s <skill-or-path>` when testing locally.\n- For OpenClaw, keep the same folder portable and install from the local directory when needed.\n- For unknown agents, use the skill as plain Markdown instructions plus any bundled scripts. The only required contract is: load the candidate skill, run the task prompt, capture output, score it, and compare against the baseline.\n\n## Quality Bar\n\nA good SkillOpt run feels like engineering, not vibes:\n\n- claims are backed by recorded rollouts\n- edits are small enough to review\n- validation decides acceptance\n- rejected edits teach the next round\n- final deployment is one clean skill file\n\nFile v0.1.0:_meta.json\n\n{\n  \"ownerId\": \"kn77zzg9p845zanvy6vrf76k7d81mcnm\",\n  \"slug\": \"skillopt\",\n  \"version\": \"0.1.0\",\n  \"publishedAt\": 1780801531554\n}\n\nFile v0.1.0:references/evaluation.md\n\n# SkillOpt Evaluation Reference\n\nUse this reference when building task suites or interpreting scores.\n\n## JSONL Task Schema\n\nEach line is one task:\n\n```json\n{\n  \"id\": \"val_spreadsheet_001\",\n  \"prompt\": \"Use the skill to inspect the workbook and report the revenue delta.\",\n  \"inputs\": [\"fixtures/revenue.xlsx\"],\n  \"tags\": [\"spreadsheet\", \"calculation\"],\n  \"scorer\": {\n    \"type\": \"contains\",\n    \"expected\": \"$42,100\"\n  }\n}\n```\n\nRequired fields:\n\n- `id`: Stable unique id. Use split prefixes such as `train_` or `val_`.\n- `prompt`: The user-facing task prompt.\n- `scorer`: A scoring object.\n\nOptional fields:\n\n- `inputs`: Files, URLs, or notes needed for the task.\n- `tags`: Capabilities or risk areas covered by the task.\n- `metadata`: Any non-secret context useful for reporting.\n\n## Scorer Types\n\n### exact\n\nPass when normalized output equals normalized expected text.\n\n```json\n{\"type\": \"exact\", \"expected\": \"PASS\"}\n```\n\n### contains\n\nPass when output contains the expected string. If `expected` is a list, every item must appear.\n\n```json\n{\"type\": \"contains\", \"expected\": [\"root cause\", \"rollback plan\"]}\n```\n\n### regex\n\nPass when the regular expression matches the output.\n\n```json\n{\"type\": \"regex\", \"pattern\": \"\\\\b[0-9]+\\\\.[0-9]{2}%\\\\b\"}\n```\n\n### command\n\nRun an external scorer. The command may use `{output_path}`, `{expected}`, `{task_id}`, and `{skill_path}` placeholders. The score is pass when the command exits `0`.\n\n```json\n{\n  \"type\": \"command\",\n  \"command\": \"python3 scorers/check_report.py --output {output_path}\"\n}\n```\n\n### manual\n\nUse when judgment is required. Store the output and record the score separately in the report.\n\n```json\n{\"type\": \"manual\", \"rubric\": \"0-1 score for factual correctness and format compliance\"}\n```\n\n## Split Discipline\n\n- Use train tasks to discover failures and propose edits.\n- Use validation tasks only for gating.\n- Do not copy validation answers, ids, or benchmark-specific tricks into the skill.\n- If validation becomes familiar after many rounds, create a fresh holdout split.\n\n## Suggested Metrics\n\nFor each split, report:\n\n- `avg_score`: Mean score over scored tasks.\n- `pass_rate`: Share of tasks with score `1.0`.\n- `scored_tasks`: Number of tasks with automatic or completed manual scores.\n- `unscored_tasks`: Number of manual or failed-to-score tasks.\n- `critical_regressions`: Validation tasks that changed from pass to fail.\n\n## Acceptance Defaults\n\nAccept a candidate only if:\n\n- validation `avg_score` improves by at least `0.02`\n- no critical validation task regresses\n- no new safety, privacy, or tool-use issue appears\n- skill metadata remains valid\n\nRaise the threshold for noisy scorers; lower it only when tasks are expensive and the observed improvement is qualitatively strong.\n\nFile v0.1.0:skill-card.md\n\n## Description:\n\nSkillOpt helps agents train, evaluate, and improve reusable skill files through rollout scoring, validation gates, and best_skill.md export.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[harrylabsj](https://clawhub.ai/user/harrylabsj)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and engineers use SkillOpt to optimize agent skill documents against train and validation task suites, compare baseline and candidate rollouts, and export the best accepted skill with a report.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Benchmark task files and agent-command templates can cause local shell commands to run with the user's privileges.\n\nMitigation: Use trusted task suites, review command scorers and agent-command templates before running them, prefer non-command scorers, and run the harness in a restricted workspace or container.\n\nRisk: Secrets exposed in the local environment could be reachable to commands launched by the harness.\n\nMitigation: Avoid exposing secrets in the environment when running SkillOpt and isolate runs from sensitive files or credentials.\n\n## Reference(s):\n\n- [SkillOpt Evaluation Reference](references/evaluation.md)\n\n## Skill Output:\n\n**Output Type(s):** [Markdown, JSON, Shell commands, Guidance]\n\n**Output Format:** [Markdown guidance with inline shell commands and generated JSON and Markdown files]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Produces run directories containing candidate skill files, rollout records, summary JSON, reports, and best_skill.md when exported.]\n\n## Skill Version(s):\n\n0.1.0 (source: server release evidence)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v0.1.0:agents/openai.yaml\n\ninterface:\n  display_name: \"SkillOpt\"\n  short_description: \"训练、验证并迭代提升可复用 Agent skill。\"\n  default_prompt: \"Use $skillopt to improve this skill with a train/validation task suite and export a best_skill.md.\"","readmeExcerpt":"Skill: SkillOpt Owner: harrylabsj Summary: Train, evaluate, and improve Agent skill files as reusable external capabilities. Use when a user wants to optimize SKILL.md, prompt procedures, OpenClaw/Her... Tags: agent:0.1.0, evaluation:0.1.0, latest:0.1.0, optimization:0.1.0, skillopt:0.1.0, skills:0.1.0 Version history: v0.1.0 | 2026-06-07T03:05:31.554Z | user Initial release: SkillOpt workflow for train/validation sk","codeSnippets":[],"executableExamples":[{"language":"text","snippet":"skillopt_runs/<target-skill-slug>/\n  source_skill.md\n  candidates/\n    candidate_000.md\n    candidate_001.md\n  tasks/\n    train.jsonl\n    val.jsonl\n  rollouts/\n    train/\n    val/\n  rejected_edits.md\n  best_skill.md\n  report.md"},{"language":"bash","snippet":"python3 scripts/skillopt.py init --skill path/to/SKILL.md --out skillopt_runs/my-skill\npython3 scripts/skillopt.py validate-tasks skillopt_runs/my-skill/tasks/train.jsonl\npython3 scripts/skillopt.py run --tasks skillopt_runs/my-skill/tasks/val.jsonl --skill skillopt_runs/my-skill/source_skill.md --out skillopt_runs/my-skill/rollouts/val_baseline --agent-command \"hermes -s {skill_path} -z {prompt}\""},{"language":"text","snippet":"## candidate_003\nRejected because validation avg +0.00 and task val_docx_04 regressed.\nAvoid adding broad \"always rewrite\" instructions; they caused format drift."},{"language":"json","snippet":"{\n  \"id\": \"val_spreadsheet_001\",\n  \"prompt\": \"Use the skill to inspect the workbook and report the revenue delta.\",\n  \"inputs\": [\"fixtures/revenue.xlsx\"],\n  \"tags\": [\"spreadsheet\", \"calculation\"],\n  \"scorer\": {\n    \"type\": \"contains\",\n    \"expected\": \"$42,100\"\n  }\n}"},{"language":"json","snippet":"{\"type\": \"exact\", \"expected\": \"PASS\"}"},{"language":"json","snippet":"{\"type\": \"contains\", \"expected\": [\"root cause\", \"rollback plan\"]}"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: skillopt\ndescription: Train, evaluate, and improve Agent skill files as reusable external capabilities. Use when a user wants to optimize SKILL.md, prompt procedures, OpenClaw/Hermes/Codex/Claude Code skills, agent workflows, skill factories, benchmark-driven skill iteration, rollout analysis, validation gates, best_skill.md export, or controlled self-evolving skills inspired by Microsoft SkillOpt.\n---\n\n# SkillOpt\n\n## Operating Idea\n\nTreat a skill document as trainable external state. Keep the target model, tools, and runtime fixed; optimize only the skill text through measured task rollouts, failure reflection, small edits, validation gating, and versioned export.\n\nDefault output is a deployable `best_skill.md` plus a short optimization report. Training may use many traces and candidate files; deployment should require only the final skill file.\n\n## Invariants\n\n- Preserve the original skill before editing.\n- Separate train and validation tasks. Never accept an edit based only on the examples used to propose it.\n- Prefer small, reviewable edits over full rewrites. Keep the skill's public contract stable unless the task suite proves the contract is wrong.\n- Score behavior, not eloquence. A prettier skill that does not improve validation is rejected.\n- Record rejected edits and the reason, then consult that buffer before proposing another edit.\n- Do not add model-specific hacks unless the target deployment is explicitly model-specific.\n- Do not leak validation answers into the skill. Validation data may guide accept/reject decisions, not become memorized instructions.\n\n## Run Directory\n\nCreate a run directory near the skill being optimized unless the user specifies another path:\n\n```text\nskillopt_runs/<target-skill-slug>/\n  source_skill.md\n  candidates/\n    candidate_000.md\n    candidate_001.md\n  tasks/\n    train.jsonl\n    val.jsonl\n  rollouts/\n    train/\n    val/\n  rejected_edits.md\n  best_skill.md\n  report.md\n```\n\nUse `scripts/skillopt.py` for deterministic run setup, JSONL validation, simple command-backed rollouts, score aggregation, validation gates, and report generation. Read `references/evaluation.md` when defining task schemas or scorers.\n\n## Workflow\n\n### 1. Define the Optimization Contract\n\nIdentify:\n\n- target skill path and deployment agents\n- target model/runtime/tool constraints to keep fixed during evaluation\n- success metric and acceptance threshold\n- task distribution the skill should serve\n- allowed edit budget, such as max 3 sections or max 25% changed lines per round\n\nIf no task suite exists, create a small proxy suite first, label it as proxy data, and tell the user that real production traces are needed for stronger conclusions.\n\n### 2. Build Train and Validation Sets\n\nRepresent each task as JSONL with an id, prompt, optional inputs, and a scorer. Keep validation examples independent and representative.\n\nMinimum split:\n\n- `train.jsonl`: failure discovery and edit proposal\n- `val.jsonl`: accept/reject gate\n\nFor fragil"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn77zzg9p845zanvy6vrf76k7d81mcnm\",\n  \"slug\": \"skillopt\",\n  \"version\": \"0.1.0\",\n  \"publishedAt\": 1780801531554\n}"},{"path":"references/evaluation.md","content":"# SkillOpt Evaluation Reference\n\nUse this reference when building task suites or interpreting scores.\n\n## JSONL Task Schema\n\nEach line is one task:\n\n```json\n{\n  \"id\": \"val_spreadsheet_001\",\n  \"prompt\": \"Use the skill to inspect the workbook and report the revenue delta.\",\n  \"inputs\": [\"fixtures/revenue.xlsx\"],\n  \"tags\": [\"spreadsheet\", \"calculation\"],\n  \"scorer\": {\n    \"type\": \"contains\",\n    \"expected\": \"$42,100\"\n  }\n}\n```\n\nRequired fields:\n\n- `id`: Stable unique id. Use split prefixes such as `train_` or `val_`.\n- `prompt`: The user-facing task prompt.\n- `scorer`: A scoring object.\n\nOptional fields:\n\n- `inputs`: Files, URLs, or notes needed for the task.\n- `tags`: Capabilities or risk areas covered by the task.\n- `metadata`: Any non-secret context useful for reporting.\n\n## Scorer Types\n\n### exact\n\nPass when normalized output equals normalized expected text.\n\n```json\n{\"type\": \"exact\", \"expected\": \"PASS\"}\n```\n\n### contains\n\nPass when output contains the expected string. If `expected` is a list, every item must appear.\n\n```json\n{\"type\": \"contains\", \"expected\": [\"root cause\", \"rollback plan\"]}\n```\n\n### regex\n\nPass when the regular expression matches the output.\n\n```json\n{\"type\": \"regex\", \"pattern\": \"\\\\b[0-9]+\\\\.[0-9]{2}%\\\\b\"}\n```\n\n### command\n\nRun an external scorer. The command may use `{output_path}`, `{expected}`, `{task_id}`, and `{skill_path}` placeholders. The score is pass when the command exits `0`.\n\n```json\n{\n  \"type\": \"command\",\n  \"command\": \"python3 scorers/check_report.py --output {output_path}\"\n}\n```\n\n### manual\n\nUse when judgment is required. Store the output and record the score separately in the report.\n\n```json\n{\"type\": \"manual\", \"rubric\": \"0-1 score for factual correctness and format compliance\"}\n```\n\n## Split Discipline\n\n- Use train tasks to discover failures and propose edits.\n- Use validation tasks only for gating.\n- Do not copy validation answers, ids, or benchmark-specific tricks into the skill.\n- If validation becomes familiar after many rounds, create a fresh holdout split.\n\n## Suggested Metrics\n\nFor each split, report:\n\n- `avg_score`: Mean score over scored tasks.\n- `pass_rate`: Share of tasks with score `1.0`.\n- `scored_tasks`: Number of tasks with automatic or completed manual scores.\n- `unscored_tasks`: Number of manual or failed-to-score tasks.\n- `critical_regressions`: Validation tasks that changed from pass to fail.\n\n## Acceptance Defaults\n\nAccept a candidate only if:\n\n- validation `avg_score` improves by at least `0.02`\n- no critical validation task regresses\n- no new safety, privacy, or tool-use issue appears\n- skill metadata remains valid\n\nRaise the threshold for noisy scorers; lower it only when tasks are expensive and the observed improvement is qualitatively strong."},{"path":"skill-card.md","content":"## Description:\n\nSkillOpt helps agents train, evaluate, and improve reusable skill files through rollout scoring, validation gates, and best_skill.md export.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[harrylabsj](https://clawhub.ai/user/harrylabsj)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and engineers use SkillOpt to optimize agent skill documents against train and validation task suites, compare baseline and candidate rollouts, and export the best accepted skill with a report.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Benchmark task files and agent-command templates can cause local shell commands to run with the user's privileges.\n\nMitigation: Use trusted task suites, review command scorers and agent-command templates before running them, prefer non-command scorers, and run the harness in a restricted workspace or container.\n\nRisk: Secrets exposed in the local environment could be reachable to commands launched by the harness.\n\nMitigation: Avoid exposing secrets in the environment when running SkillOpt and isolate runs from sensitive files or credentials.\n\n## Reference(s):\n\n- [SkillOpt Evaluation Reference](references/evaluation.md)\n\n## Skill Output:\n\n**Output Type(s):** [Markdown, JSON, Shell commands, Guidance]\n\n**Output Format:** [Markdown guidance with inline shell commands and generated JSON and Markdown files]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Produces run directories containing candidate skill files, rollout records, summary JSON, reports, and best_skill.md when exported.]\n\n## Skill Version(s):\n\n0.1.0 (source: server release evidence)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment."},{"path":"agents/openai.yaml","content":"interface:\n  display_name: \"SkillOpt\"\n  short_description: \"训练、验证并迭代提升可复用 Agent skill。\"\n  default_prompt: \"Use $skillopt to improve this skill with a train/validation task suite and export a best_skill.md.\""}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Train, evaluate, and improve Agent skill files as reusable external capabilities. Use when a user wants to optimize SKILL.md, prompt procedures, OpenClaw/Her... Skill: SkillOpt Owner: harrylabsj Summary: Train, evaluate, and improve Agent skill files as reusable external capabilities. Use when a user wants to optimize SKILL.md, prompt procedures, OpenClaw/Her... Tags: agent:0.1.0, evaluation:0.1.0, latest:0.1.0, optimization:0.1.0, skillopt:0.1.0, skills:0.1.0 Version history: v0.1.0 | 2026-06-07T03:05:31.554Z | user Initial release: SkillOpt workflow for train/validation sk","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1353,"uniquenessScore":49,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-10T17:44:17.612Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-10T17:44:17.612Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T21:43:01.439Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}