{"id":"20850960-cf0c-4862-bbc3-e63842d0e5f6","entityType":"agent","slug":"clawhub-gongyu0918-debug-skill-usefulness-audit","name":"skill-usefulness-audit","canonicalUrl":"https://www.xpersona.co/agent/clawhub-gongyu0918-debug-skill-usefulness-audit","canonicalPath":"/agent/clawhub-gongyu0918-debug-skill-usefulness-audit","generatedAt":"2026-10-09T15:19:05.433Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T06:14:25.202Z","emptyReason":null},"description":"Review your installed agent skills to see what you actually use, what overlaps, and what may no longer be worth keeping.","descriptionLabel":"Source description","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 4K downloads reported by the source. Last updated 10/9/2026.","installCommand":"clawhub skill install s172a0qxsw6kfdee064rcn06es83edse:skill-usefulness-audit","sourceUrl":"https://clawhub.ai/gongyu0918-debug/skill-usefulness-audit","homepage":"https://clawhub.ai/gongyu0918-debug/skills/skill-usefulness-audit","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/gongyu0918-debug/skill-usefulness-audit","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/gongyu0918-debug/skills/skill-usefulness-audit","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":72,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"skill-usefulness-audit technical dossier on Xpersona with agent coverage, OPENCLEW support, and live trust metadata."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-09T06:14:25.202Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T06:14:25.202Z","emptyReason":null},"stars":null,"forks":null,"downloads":3982,"packageName":null,"latestVersion":"0.3.24","tractionLabel":"4K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T06:14:25.202Z","emptyReason":null},"lastUpdatedAt":"2026-10-09T06:14:25.202Z","lastCrawledAt":"2026-10-09T06:14:25.202Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-10T06:14:25.202Z","lastVerifiedAt":null,"highlights":[{"version":"0.3.24","createdAt":"2026-08-24T10:08:57.284Z","changelog":"Add a package-marker profile for Chinese-only derivatives; the ClawHub edition remains bilingual.","fileCount":23,"zipByteSize":72674},{"version":"0.3.22","createdAt":"2026-08-09T09:00:14.107Z","changelog":"Trim scoring-rubric context load 2891->2674 units and ratchet the rubric ceiling; no scoring, risk, or JSON contract change.","fileCount":23,"zipByteSize":72312},{"version":"0.3.21","createdAt":"2026-08-08T07:28:16.479Z","changelog":"Centralize shared scoring thresholds, split the concise report builder, align the rubric with runtime facts, tighten SKILL.md claims, and apply low-risk cleanups. No scoring or JSON contract change.","fileCount":23,"zipByteSize":72690},{"version":"0.3.20","createdAt":"2026-07-31T07:51:43.804Z","changelog":"Reduced routing and reference load, aligned the misfire rubric, and improved truthful bilingual Markdown evidence without changing scoring or JSON.","fileCount":23,"zipByteSize":71434},{"version":"0.3.19","createdAt":"2026-07-29T15:26:35.075Z","changelog":"Reduced entry and reference load while preserving routing, scoring, ablation, action thresholds, and report contracts.","fileCount":23,"zipByteSize":70583},{"version":"0.3.18","createdAt":"2026-07-27T09:32:49.927Z","changelog":"Removed store and maintenance residue from the runtime entry, shortened the ClawHub summary, and kept audit behavior unchanged.","fileCount":23,"zipByteSize":71237},{"version":"0.3.17","createdAt":"2026-07-25T07:55:11.064Z","changelog":"Reduce duplicated runtime rules and split audit orchestration god functions without changing reports, scoring, or risk behavior.","fileCount":23,"zipByteSize":71540},{"version":"0.3.16","createdAt":"2026-07-16T10:46:27.541Z","changelog":"Default to concise natural-language usefulness reports, add entry-prompt character/token totals, and keep full Markdown evidence available.","fileCount":23,"zipByteSize":71872}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s172a0qxsw6kfdee064rcn06es83edse:skill-usefulness-audit","setupComplexity":"low","setupSteps":["Install using `clawhub skill install s172a0qxsw6kfdee064rcn06es83edse:skill-usefulness-audit` in an isolated environment before connecting it to live workloads.","No published capability contract is available yet, so validate auth and request/response behavior manually.","Review the upstream CLAWHUB listing at https://clawhub.ai/gongyu0918-debug/skill-usefulness-audit before using production credentials."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-gongyu0918-debug-skill-usefulness-audit/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-gongyu0918-debug-skill-usefulness-audit/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-gongyu0918-debug-skill-usefulness-audit/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-gongyu0918-debug-skill-usefulness-audit/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-gongyu0918-debug-skill-usefulness-audit/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-gongyu0918-debug-skill-usefulness-audit/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-09T15:19:05.428Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-gongyu0918-debug-skill-usefulness-audit/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-gongyu0918-debug-skill-usefulness-audit/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-gongyu0918-debug-skill-usefulness-audit/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-gongyu0918-debug-skill-usefulness-audit/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T06:14:25.202Z","emptyReason":null},"readme":"Skill: skill-usefulness-audit\n\nOwner: gongyu0918-debug\n\nSummary: Review your installed agent skills to see what you actually use, what overlaps, and what may no longer be worth keeping.\n\nTags: ablation:0.3.24, audit:0.3.24, latest:0.3.24, latest audit skills:0.2.7, latest audit skills openclaw:0.3.14, latest audit skills openclaw hermes claude-code:0.2.11, openclaw:0.3.24, skills:0.3.24\n\nVersion history:\n\nv0.3.24 | 2026-08-24T10:08:57.284Z | user\n\nAdd a package-marker profile for Chinese-only derivatives; the ClawHub edition remains bilingual.\n\nv0.3.22 | 2026-08-09T09:00:14.107Z | user\n\nTrim scoring-rubric context load 2891->2674 units and ratchet the rubric ceiling; no scoring, risk, or JSON contract change.\n\nv0.3.21 | 2026-08-08T07:28:16.479Z | user\n\nCentralize shared scoring thresholds, split the concise report builder, align the rubric with runtime facts, tighten SKILL.md claims, and apply low-risk cleanups. No scoring or JSON contract change.\n\nv0.3.20 | 2026-07-31T07:51:43.804Z | user\n\nReduced routing and reference load, aligned the misfire rubric, and improved truthful bilingual Markdown evidence without changing scoring or JSON.\n\nv0.3.19 | 2026-07-29T15:26:35.075Z | user\n\nReduced entry and reference load while preserving routing, scoring, ablation, action thresholds, and report contracts.\n\nv0.3.18 | 2026-07-27T09:32:49.927Z | user\n\nRemoved store and maintenance residue from the runtime entry, shortened the ClawHub summary, and kept audit behavior unchanged.\n\nv0.3.17 | 2026-07-25T07:55:11.064Z | user\n\nReduce duplicated runtime rules and split audit orchestration god functions without changing reports, scoring, or risk behavior.\n\nv0.3.16 | 2026-07-16T10:46:27.541Z | user\n\nDefault to concise natural-language usefulness reports, add entry-prompt character/token totals, and keep full Markdown evidence available.\n\nv0.3.15 | 2026-07-12T19:48:44.308Z | user\n\nHarden audit evidence handling, OpenClaw routing, dependency classification, Chinese tool protection, and repository navigation.\n\nv0.3.14 | 2026-07-11T06:56:16.182Z | user\n\nRestore compatibility-wrapper state and document complete scoring contracts\n\nv0.3.13 | 2026-07-09T20:21:35.345Z | user\n\nFix audit classification, risk scanning, credential detection, and concise locale-aware report summaries\n\nv0.3.12 | 2026-07-09T03:40:31.299Z | user\n\nFix fenced-code, risk-scan, classification, and private-content edge cases\n\nv0.3.11 | 2026-07-04T22:47:48.018Z | user\n\nClarify report wording, delete reasons, and skipped-directory notes\n\nv0.3.10 | 2026-07-04T13:15:05.408Z | user\n\nDetect env credential bundles and scan SKILL.md fenced code risks\n\nv0.3.9 | 2026-07-03T07:46:27.303Z | user\n\nUpdate review baseline notes and align source-bundle self-audit behavior\n\nv0.3.8 | 2026-06-29T22:26:04.686Z | user\n\nMitigate Hermes install-scan false positives and preserve fallback frontmatter parity\n\nv0.3.7 | 2026-06-28T16:01:26.186Z | user\n\nParse YAML inline lists in no-PyYAML fallback frontmatter and preserve required env evidence\n\nv0.3.6 | 2026-06-27T11:36:34.034Z | user\n\nDetect empty contracts, missing imports, reference pollution, nested metadata fallback, and duplicate install audits\n\nv0.3.5 | 2026-06-26T23:25:27.572Z | user\n\nPreserve CJK skill names, parse fallback frontmatter lists, and compact zh-CN decision summaries\n\nv0.3.4 | 2026-06-23T09:06:37.748Z | user\n\nClarify trigger boundaries and add prompt-matrix regression coverage\n\nv0.3.3 | 2026-06-22T09:15:23.076Z | user\n\nAdd decision summary, localized Markdown reports, and opt-in raw JSON workflow\n\nv0.3.2 | 2026-06-21T14:21:34.148Z | user\n\nMake ablation planning opt-in, preserve medium-risk review, trim runtime context, and harden local test cleanup\n\nv0.3.1 | 2026-06-14T16:47:22.181Z | user\n\nFix risk scoring, credential risk escalation, broken-script health caps, and low-evidence verdicts\n\nv0.3.0 | 2026-06-05T09:57:48.502Z | user\n\nAdd safe-first-run docs, strict input validation, version output, and no-skills diagnostics\n\nv0.2.17 | 2026-06-03T07:56:06.693Z | user\n\nRestore MIT-0 license files in the GitHub repository and OpenClaw bundle.\n\nv0.2.16 | 2026-06-02T19:10:02.316Z | user\n\nFix readiness required-env parsing boundaries: ignore raw frontmatter env-like fields, support camelCase declared secrets, and prefer env/envVar fields over display names in object schemas.\n\nv0.2.15 | 2026-06-02T08:37:22.016Z | user\n\nTighten install identity dedupe and flag missing declared API keys or required environment variables\n\nv0.2.14 | 2026-06-02T08:14:38.825Z | user\n\nRead ClawHub metadata, preserve OpenClaw skill identity, and deduplicate duplicate installs in ranking and ablation planning\n\nv0.2.13 | 2026-05-24T13:04:34.490Z | user\n\nTrim audit skill docs, add reference contents, plain-language action advice, sync hardening, and pseudo-skill regression tests\n\nv0.2.12 | 2026-05-22T07:50:50.575Z | user\n\nAdd lightweight risk review checks and keep the ClawHub bundle OpenClaw-specific\n\nv0.2.11 | 2026-05-22T06:46:51.945Z | user\n\nAdd OpenClaw, Hermes, and Claude Code compatibility metadata and discovery defaults\n\nv0.2.10 | 2026-05-08T09:23:41.129Z | user\n\nSpecialize the ClawHub bundle for OpenClaw and point other agent users to GitHub\n\nv0.2.9 | 2026-05-01T13:35:06.912Z | user\n\nSplit risk signatures out of executable-looking constants to avoid static scanner false positives\n\nv0.2.8 | 2026-05-01T13:20:37.666Z | user\n\nModularize audit code, fix weak history evidence, harden frontmatter sync, and clarify static risk signals\n\nv0.2.7 | 2026-04-26T16:15:01.511Z | user\n\nFix script burden detection, refresh stale ablation candidates, and improve mixed-language context estimates\n\nv0.2.6 | 2026-04-25T18:17:39.644Z | user\n\nAdd quality burden scoring, cost-efficient ablation planning, and model-cost reduction estimates\n\nv0.2.5 | 2026-04-24T02:05:06.895Z | user\n\nImprove README positioning, shorten ClawHub summary, and add score/report breakdowns\n\nv0.2.4 | 2026-04-23T15:08:45.601Z | user\n\nUse lifetime community signals, improve history fallback recall, remove redundant risk reads, and validate bundle drift in CI\n\nv0.2.3 | 2026-04-22T13:31:15.476Z | user\n\nFix duplicate-name evidence routing, make bundle sync CRLF-safe, add missing-file warnings, and tighten risk scanning\n\nv0.2.2 | 2026-04-21T12:20:01.079Z | user\n\nReduce ClawHub scan false positives, rename protected-path risk labels, and keep audit behavior intact\n\nv0.2.1 | 2026-04-21T12:04:47.745Z | user\n\nFix duplicate-name evidence routing, doc-only risk false positives, host-prompt history inflation, and README quickstart\n\nv0.2.0 | 2026-04-20T02:34:03.465Z | user\n\nRecency-weighted evidence, confidence score, risk scan, and offline community metrics\n\nv0.1.1 | 2026-04-18T21:06:00.401Z | user\n\nCompatibility expansion and bilingual introduction\n\nv0.1.0 | 2026-04-18T20:50:42.634Z | user\n\nInitial release\n\nArchive index:\n\nArchive v0.3.24: 23 files, 72674 bytes\n\nFiles: LICENSE (911b), references/ablation-protocol.md (2820b), references/report-narration-prompt.md (819b), references/scoring-rubric.md (10696b), scripts/skill_usefulness_audit_lib/__init__.py (219b), scripts/skill_usefulness_audit_lib/ablation.py (5024b), scripts/skill_usefulness_audit_lib/cli.py (38480b), scripts/skill_usefulness_audit_lib/common.py (24381b), scripts/skill_usefulness_audit_lib/community.py (7605b), scripts/skill_usefulness_audit_lib/constants.py (12681b), scripts/skill_usefulness_audit_lib/reporting.py (62612b), scripts/skill_usefulness_audit_lib/risk_quality.py (46289b), scripts/skill_usefulness_audit_lib/risk_signatures_encoding.py (271b), scripts/skill_usefulness_audit_lib/risk_signatures_execution.py (687b), scripts/skill_usefulness_audit_lib/risk_signatures_network.py (581b), scripts/skill_usefulness_audit_lib/risk_signatures_sensitive.py (326b), scripts/skill_usefulness_audit_lib/risk_signatures.py (394b), scripts/skill_usefulness_audit_lib/scoring.py (19650b), scripts/skill_usefulness_audit_lib/usage_loader.py (11893b), scripts/skill_usefulness_audit.py (793b), skill-card.md (2161b), SKILL.md (6901b), _meta.json (142b)\n\nFile v0.3.24:SKILL.md\n\n---\nname: skill-usefulness-audit\nslug: skill-usefulness-audit\ndescription: Review your installed agent skills to see what you actually use, what overlaps, and what may no longer be worth keeping.\nversion: 0.3.24\ntags: [\"audit\",\"skills\",\"ablation\",\"openclaw\"]\nuser-invocable: true\ndisable-model-invocation: true\nargument-hint: --skills-root PATH --usage-file FILE\nhomepage: https://github.com/gongyu0918-debug/skill-usefulness-audit\nmetadata: {\"openclaw\":{\"skillKey\":\"skill-usefulness-audit\",\"requires\":{\"bins\":[\"python\"]},\"homepage\":\"https://github.com/gongyu0918-debug/skill-usefulness-audit\"}}\n---\n# Skill Usefulness Audit\n\n## Manual Trigger Only\n\nUse this skill only after a direct request to audit installed agent skills, their usage, overlap, cleanup options, or a structure-only inventory.\nDo not invoke it during normal tasks or use it for ordinary repository/source-code review, general security audit, or employee/human skill assessment.\n\n## Safety\n\nNever delete, merge, quarantine, isolate, or disable skills automatically.\nTreat `delete`, `merge-delete`, and `quarantine-review` as manual-review recommendations.\nDo not delete skills based only on a structure-only report.\nThis tool does not automatically replay historical conversations; it generates ablation plans and reads ablation result files that the user provides.\n\n## Audit Scope\n\nAudit these layers in order:\n\n1. Usage evidence, including recency and source quality.\n2. Installed metadata, instructions, and functional overlap.\n3. User-provided skill-on versus skill-off results for general skills.\n4. Runtime and bundle burden, including over-triggering, context cost, weak progressive disclosure, redundant resources, script failures, and private-looking files.\n5. Static health and risk hints.\n6. Optional offline community or registry metrics.\n\nTreat API and tool skills as protected capability skills during ablation.\nExamples: Excel, DOCX, PDF, browser automation, deployment, OCR, external API wrappers, MCP/API gateway helpers.\n\n## Workflow\n\n1. Collect user-provided roots before host-local defaults.\n2. Load only the usage, history, ablation, and community evidence that is available.\n3. Inspect each `SKILL.md` and its script/reference/asset metrics.\n4. Let the bundled script classify each skill as `api`, `tool`, or `general` and calculate its score. Read `{baseDir}/references/scoring-rubric.md` only when checking or explaining a score, verdict, or action.\n5. Print the short usefulness report and, when requested, write Markdown evidence or an ablation plan.\n\n## Ablation Rules\n\nRead `{baseDir}/references/ablation-protocol.md` only when running replays, preparing normalized ablation records, or reviewing mixed or delete-boundary results. The script can generate an ablation plan without loading the protocol.\nReplay only selected `general` candidates with identical prompts/artifacts and pairwise judging.\nDo not fake no-tool ablation for `api` or `tool` skills; use the rubric's protected-capability branch.\n\n## Run the Audit\n\nRun the audit after collecting available evidence:\n\n```bash\nREPORT_LANGUAGE=en  # use zh-CN when the current user invocation is Chinese\npython \"{baseDir}/scripts/skill_usefulness_audit.py\" audit \\\n  --skills-root ./skills \\\n  --report-language \"$REPORT_LANGUAGE\" \\\n  --markdown-out ./skill-audit-report.md\n```\n\nOpenClaw expands `{baseDir}` to the installed skill directory. Use it for bundled scripts and references.\n\nAdd evidence only when available:\n\n- `--usage-file`: JSON, JSONL, CSV, or TSV with per-skill usage.\n- `--history-file`: raw transcripts used only when direct usage is weak or missing; mentions remain `history_mentions` / `suspected_invocations`, not `calls`.\n- `--ablation-file`: normalized JSON or JSONL skill-on/skill-off results.\n- `--community-file`: offline JSON, JSONL, CSV, or TSV registry metrics.\n- `--ablation-plan-out`: a cost estimate and focused replay plan; its case counts can be overridden with the four `--ablation-*-cases` options documented by `--help`.\n- `--json-out`: machine-readable evidence only when requested or needed by another tool.\n\nPass `--report-language zh-CN` for a Chinese invocation and `--report-language en` for an English invocation. `auto` reads `SKILL_AUDIT_REPORT_LANGUAGE` or the process locale, then falls back to English.\n\nRun without extra files only when you need a structure-only audit.\nUsage, community, and ablation evidence become lower-confidence in that mode.\nHistory and usage files may contain sensitive conversations, local paths, project names, and customer data.\nMissing env means not configured in the current audit process, not proof that the skill is broken in every host.\n\n## Output Contract\n\nUse one run for both output layers; do not ask the user to choose a quick or full mode.\nStandard output is a short natural-language report. Its opening paragraph states the audited skill count and the total characters plus approximate tokens of loaded entry descriptions. Lead with actual usage, not static risk or bundle health, and keep scores, internal codes, risk flags, and tables out of this layer.\nWhen `--markdown-out` is provided, write the detailed evidence—with scores, action codes, missing evidence, burden, and risk notes—in the same run.\nMatch the user's language: clean Chinese for `zh-CN` and clean English for `en`, except for skill names and unavoidable paths or commands.\n\nCopy the short report to chat verbatim, apart from making its evidence path clickable. Do not paste raw JSON or the full Markdown evidence unless the user asks. Read `{baseDir}/references/report-narration-prompt.md` only when another agent or host must deliver an already-generated report.\n\nJSON includes `report_mode`, per-skill `score_breakdown`, `quality_penalty`, `quality_penalty_uncapped`, `quality_evidence`, `community_breakdown`, `action_advice`, and `risk_review`. It includes `ablation_plan` only when `--ablation-plan-out` is used. JSON emits both `risk_*` and `static_risk_*` with identical values, and `total_score` as an alias of `local_score`; treat `risk_*` and `local_score` as canonical.\n\nKeep deletion advice conservative for system or host-core skills, and prefer narrowing or merging when overlapping skills still serve distinct host integrations.\n\n## Resources\n\n- `{baseDir}/scripts/skill_usefulness_audit.py`: compatibility wrapper for the modular audit package.\n- `{baseDir}/scripts/skill_usefulness_audit_lib/`: collect metadata, score skills, scan static risk hints, and render Markdown reports plus optional JSON artifacts.\n- `{baseDir}/references/report-narration-prompt.md`: concise prompt for turning the report into a user-facing conversational summary.\n- `{baseDir}/references/scoring-rubric.md`: 10-point scoring rules, confidence logic, community prior, and action thresholds.\n- `{baseDir}/references/ablation-protocol.md`: normalized replay method for historical conversations.\n\nFile v0.3.24:_meta.json\n\n{\n  \"ownerId\": \"kn7em0w89d0zac35fzt84qm2a182j54b\",\n  \"slug\": \"skill-usefulness-audit\",\n  \"version\": \"0.3.24\",\n  \"publishedAt\": 1787566137284\n}\n\nFile v0.3.24:references/ablation-protocol.md\n\n# Ablation Protocol\n\nUse this protocol for `general` skills selected by the ablation plan.\n\n## Goal\n\nMeasure whether the skill changes outcomes in a meaningful way.\nHigh consistency between skill-on and skill-off runs means the skill adds little value.\n\n## Sampling\n\nStart with `3` historical tasks where the skill should plausibly matter. Prefer real user turns over synthetic prompts. Expand to `5` when results are mixed and to `10` only for high-impact or delete-boundary decisions.\n\n## Replay Method\n\nFor each selected case, run two isolated replays:\n\n1. `with_skill`\n2. `without_skill`\n\nKeep these constant:\n\n- same prompt\n- same files and artifacts\n- same model class when possible\n- same tool permissions\n- same success criteria\n\nUse a fresh thread or isolated run if the host supports it.\n\n## Judge Method\n\nFor open-ended outputs:\n\n1. Compare `with_skill` and `without_skill` side by side.\n2. Randomize A/B order.\n3. Spot-check reversed order on boundary cases.\n4. Prefer `pass/fail`, `same/better/worse`, and short reasons over long open-ended grading.\n\nRecord a standard `verdict` and one short `notes` reason. Each arm may also include optional `pass` and/or `score` from `0.0-1.0` for fallback inference. Optional `tool_cost` may describe calls, latency, or retries and currently does not affect the audit score.\n\n### Normalized JSON\n\n```json\n[\n  {\n    \"skill\": \"emotion-orchestrator\",\n    \"case_id\": \"case-001\",\n    \"with_skill\": {\"pass\": true, \"score\": 0.92},\n    \"without_skill\": {\"pass\": true, \"score\": 0.81},\n    \"verdict\": \"better\"\n  }\n]\n```\n\n## Judgment Rule\n\nUse `same` when the final answer, correctness, and workflow remain materially equivalent.\nUse `better` when the skill improves correctness, speed, structure, or user-fit in a way the baseline did not.\nUse `worse` when the skill adds friction, drift, or errors.\n\nIgnore verdict-only cases with unsupported values. A case with an unknown or missing verdict is usable only when both arms provide comparable `pass` and/or `score` fields for inference.\n\n## Early Stop Rules\n\n- Stop as low-value when `3/3` cases are `same` and `better_rate` is `0`.\n- Stop as useful when at least `2/3` cases are `better` and no case is `worse`.\n- Expand to `5` when the first batch is mixed.\n- Expand to `10` only for delete-boundary or high-impact decisions.\n\nDelete-boundary means \"needs stronger human review evidence\", not automatic deletion authority.\n\n## Planning and Model Cost\n\nCreate the replay plan with `--ablation-plan-out`. Planning uses local evidence and does not call an LLM.\n\nThe plan estimates replay cost for `light`, `realistic`, and `coding` profiles. Each case assumes two replays and one compact pairwise judge. `model_cost_estimates.unit` records `estimated_context_units_per_case`.\n\nFeed normalized results back with `--ablation-file`.\n\nFile v0.3.24:references/report-narration-prompt.md\n\n# Report Delivery Contract\n\n- Treat standard output as the final short report. Copy it verbatim into chat, apart from making the evidence path clickable when supported.\n- Do not add headings, bullets, extra counts, verification notes, or facts from the Markdown evidence unless requested.\n- Keep the short report on actual use, missing use evidence, overlap, and verified outcome impact. Leave scores, internal codes, risk, bundle health, and tables in the Markdown evidence.\n- Match the user's language: clean Chinese for `zh-CN` and clean English for `en`, except for skill names and unavoidable paths or commands.\n- Do not paste raw JSON or read back the full Markdown report unless requested.\n- Treat removal results as manual-review recommendations. Never remove, merge, isolate, or disable a skill automatically.\n\nFile v0.3.24:references/scoring-rubric.md\n\n# Scoring Rubric\n\n## Contents\n\n- Core Outputs\n- Usage Score\n- Uniqueness Score\n- Impact Score\n- Confidence Score\n- Quality Penalty\n- Community Prior Score\n- Static Risk Level\n- Verdict Bands\n- Action Rules\n\n## Core Outputs\n\n- `local_score = usage_score + uniqueness_score + impact_score`\n- `quality_penalty`: `0.0-2.5`\n- `quality_penalty_uncapped`: raw quality burden before the cap\n- `static_quality_penalty`: `0.0-1.4`\n- `final_score = clamp(local_score - quality_penalty, 0.0, 10.0)`\n- `risk_level` / `static_risk_level`: `none / low / medium / high`\n\n## 1. Usage Score (`0.0-3.0`)\n\nPrefer direct host usage logs.\nUse transcript mentions only as weaker fallback evidence.\n\n### Input Fields\n\n- Direct usage: `calls`, `recent_30d_calls`, `recent_90d_calls`, `last_used_at`, and `active_days`.\n- History fallback: `history_mentions` and `suspected_invocations`. These are weak evidence weighted through history and must not be reported as direct `calls`.\n- Evidence and runtime burden: `usage_source`, `evidence_weight`, `executions`, `script_failures`, `repair_turns`, `reference_loads`, and `false_triggers`.\n\n### Base Usage Strength\n\n- When `recent_30d_calls` exists:\n  - `0.0`: `0`\n  - `1.0`: `1-2`\n  - `2.0`: `3-7`\n  - `3.0`: `8+`\n- When only `recent_90d_calls` exists:\n  - `0.0`: `0`\n  - `0.75`: `1-2`\n  - `1.5`: `3-9`\n  - `2.5`: `10+`\n- When only total `calls` exists:\n  - `0.0`: `0`\n  - `1.0`: `1-2`\n  - `2.0`: `3-9`\n  - `3.0`: `10+`\n\n### Recency Adjustments\n\n- add `0.5` when `last_used_at <= 7 days`\n- add `0.25` when `last_used_at <= 30 days`\n- subtract `0.5` when `last_used_at > 180 days`\n- add `0.25` when `active_days >= 10`\n- add `0.10` when `active_days >= 3`\n\n### Evidence Weight\n\n- `1.00`: direct usage file\n- `0.45`: transcript-history fallback based on `suspected_invocations`\n- `0.00`: missing usage evidence\n\n## 2. Uniqueness Score (`0.0-3.0`)\n\nMeasure the highest functional-overlap similarity against any other installed skill using descriptions, headings, and resource names.\n\nBuckets:\n\n- `0.0`: highest overlap `>= 0.85`\n- `1.0`: highest overlap `0.65-0.84`\n- `2.0`: highest overlap `0.40-0.64`\n- `3.0`: highest overlap `< 0.40`\n\n## 3. Impact Score (`0.0-4.0`)\n\n### General skills\n\nUse ablation on historical conversations.\nCompute:\n\n- `consistency_rate`: skill-on and skill-off produce materially equivalent outcomes\n- `better_rate`: skill-on clearly improves the result\n- `worse_rate`: skill-on clearly harms the result\n\nBase score from consistency:\n\n- `0.0`: `consistency_rate >= 0.85`\n- `1.0`: `0.70-0.84`\n- `2.0`: `0.55-0.69`\n- `3.0`: `0.35-0.54`\n- `4.0`: `< 0.35`\n\nAdjustments:\n\n- add `1.0` when `better_rate - worse_rate >= 0.30`\n- subtract `1.0` when `worse_rate > better_rate`\n\nWhen ablation is missing, use low-evidence score `1.0` for zero-call skills.\nFor skills with direct usage evidence but no ablation yet, keep temporary neutral score `2.0` and lower confidence.\n\n### API and tool skills\n\nSkip history ablation.\nUse protected-capability scoring instead:\n\n- start at `2.0`\n- add `1.0` when the skill ships executable scripts or reference files\n- add `0.5` when highest overlap `< 0.35`\n- add `0.5` when calls `>= 3`\n- subtract `1.0` when highest overlap `>= 0.75`\n- subtract `0.5` when calls are `0`\n\n## 4. Confidence Score (`0.0-1.0`)\n\nConfidence describes evidence quality, not usefulness.\n\nAdd:\n\n- `0.35` for direct usage files\n- `0.15` for history fallback\n- `0.20` when recent usage fields exist\n- `0.10` when only total direct calls exist\n- `0.25` for protected `api/tool` classification\n- `0.25` for `general` skills with `>= 5` ablation cases\n- `0.15` for `general` skills with `1-4` ablation cases\n- `0.10` when more than one skill is in scope\n- `0.05` when only one skill exists in scope\n- `0.10` when community metadata exists\n\n## 5. Quality Penalty (`0.0-2.5`)\n\nQuality penalty captures the cost of keeping a skill and is deducted from `local_score`; it is not a risk flag.\n\n### Runtime burden\n\nUse direct usage logs when available:\n\n- `overtrigger-low-execution`: `0.45` when `calls >= 8` and `executions / calls < 0.25`\n- `overtrigger-misfire`: `0.35` when `calls >= 5` and (`false_triggers >= 3` or `false_triggers / calls >= 0.25`)\n- `overtrigger-no-impact`: `0.40` when `calls >= 5`, `consistency_rate >= 0.85`, and `better_rate <= 0.10`\n- `reference-overload`: `0.30` when `reference_loads >= 10` and `reference_loads / calls >= 3.0`\n- `script-failure-burden`: `0.45` when `script_failures >= 3` or the failure rate against executions (or calls) reaches `0.30`; `0.20` below that\n- `agent-repair-burden`: `0.30` when `repair_turns >= 3`\n\n### Readiness burden\n\n- `missing-required-env`: `0.90` when declared required environment variables are not configured in the current audit process\n\n### Catalog burden\n\n- `near-duplicate-instructions`: `0.10` when the instruction fingerprint closely matches another installed skill\n\n### Static bundle burden\n\nScan installed skill files:\n\n- `empty-skill-contract`: `0.80` when the skill has no meaningful runtime contract beyond minimal or missing metadata\n- `prompt-bloat`: `0.40` when `SKILL.md` body is at least `5000` context units\n- `prompt-bloat`: `0.20` when `SKILL.md` body is at least `2500` context units\n- `broad-trigger-surface`: `0.25` for at least two broad trigger matches, or one match with description at `30+` context units\n- `description-bloat`: `0.25` when the frontmatter description is at least `120` context units\n- `description-bloat`: `0.10` when the frontmatter description is at least `60` context units\n- `reference-disclosure-gap`: `0.30` when at least 3 reference files exist and none are directly discoverable from `SKILL.md`\n- `reference-disclosure-gap`: `0.10` when 1-2 reference files exist and none are directly discoverable from `SKILL.md`\n- `reference-disclosure-gap`: `0.20` when at least 8 reference files exist and fewer than 30% are directly linked from `SKILL.md`\n- `reference-link-broken`: `0.25` when `SKILL.md` points to missing reference files\n- `reference-bloat`: `0.50` when references are at least 50 files or 50000 context units\n- `reference-bloat`: `0.25` when references are at least 20 files or 15000 context units\n- `long-reference-without-toc`: `0.20` when at least 3 reference files over `100` lines lack a visible table of contents\n- `long-reference-without-toc`: `0.10` when 1-2 such files lack one\n- `reference-content-pollution`: `0.35` when references include advertising, upsells, unrelated text, or other-tool/skill promotion\n- `asset-bloat`: `0.50` when assets are at least 200 files or 100 MB\n- `asset-bloat`: `0.25` when assets are at least 50 files or 25 MB\n- `vague-resource-names`: `0.20` when at least 5 scripts, references, or assets use generic filenames\n- `private-bundle-artifact`: `0.60` when bundled filenames look private or environment-specific\n- `private-content-artifact`: `0.60` when bundled content looks like credentials or keys\n- `executable-asset`: `0.30` when assets contain executable binaries or installers\n- `script-count-bloat`: `0.20` when the bundle has at least 40 scripts\n- `script-count-bloat`: `0.10` when the bundle has at least 20 scripts\n- `script-maintenance-smell`: `0.40` when at least 8 scripts contain placeholders, local absolute paths, or maintenance smells\n- `script-maintenance-smell`: `0.25` when 1-7 scripts contain placeholders, local absolute paths, or maintenance smells\n- `script-syntax-error`: `0.50` when Python scripts contain syntax errors\n- `script-import-error`: `0.50` when Python scripts import modules missing from the local environment or bundle\n\nClamp `static_quality_penalty` to `0.0-1.4`, then clamp the combined `quality_penalty` to `0.0-2.5`.\n\n## 6. Community Prior Score (`0.0-1.0`)\n\nTreat community data as external prior, not a local verdict.\n\nWeighted components:\n\n- `0.30`: normalized rating\n- `0.20`: current installs or downloads\n- `0.10`: all-time installs\n- `0.15`: trending metric\n- `0.10`: stars\n- `0.05`: comments\n- `0.10`: maintenance freshness from `last_updated`\n\nNormalization: rating divides by `5.0`; volume signals use `log1p` with saturation `5000` (current), `20000` (all-time), `250` (trending, stars), and `100` (comments); maintenance scores `1.0/0.7/0.4/0.1` at `<=180/<=365/<=730/>730` days.\n\n## 7. Static Risk Level\n\nRun static scans against runnable scripts and resource files.\nOnly fenced code blocks in `SKILL.md` and directly linked Markdown references are scanned as commands; prose outside fences and unlinked references are not command-scanned.\nCredential-like content checks still cover `SKILL.md`, scripts, assets, references, and root text files without echoing matched values.\nThis is lint-style evidence only; it cannot prove a skill is safe.\n\nTypical flags: `curl-pipe-shell`, `dynamic-exec`, `protected-path-access`, `persistence-hook`, `external-post`, `shell-exec`, `network-download`, and `base64-payload`.\n\nStatic risk levels:\n\n- `none`: `0.0`\n- `low`: `0.0 < score < 2.0`\n- `medium`: `2.0-3.9`\n- `high`: `4.0+`\n\nIf static quality finds `private-content-artifact`, that evidence is promoted to `high` risk so credential-like bundled content receives `quarantine-review`.\n\n## Health Cap\n\nSome quality findings cap the final score even when usage or protected-capability signals are strong:\n\n- `script-syntax-error`: final score cap `4.0`\n- `empty-skill-contract`: final score cap `5.5`\n- `script-import-error`: final score cap `5.5`\n- `script-failure-burden`: final score cap `4.0` when the penalty is at least `0.45`\n\n## Verdict Bands\n\nUse `final_score` for verdict bands.\n\n- confidence `< 0.55` and `final_score < 4.5`: `insufficient-evidence`\n- `8.0-10.0`: `keep`\n- `6.0-7.9`: `keep-narrow`\n- `4.5-5.9`: `review`\n- `3.0-4.4`: `merge-delete`\n- `0.0-2.9`: `delete`\n\n## Action Rules\n\nEvaluate top to bottom; the first matching rule wins.\n\n| Condition | Action |\n| --- | --- |\n| system source, high risk / otherwise | `review-system` / `keep-system` |\n| high risk | `quarantine-review` |\n| medium risk and score `>= 6.0` | `keep-review-risk` |\n| quality penalty `>= 1.2` and score `>= 6.0` / `>= 4.5` | `keep-review-burden` / `review-burden` |\n| score `>= 8.0` / `>= 6.0` | `keep` / `keep-narrow` |\n| remaining medium risk | `review-risk` |\n| confidence `< 0.55` | `observe-30d` |\n| score `>= 4.5`: overlap `>= 0.65` / community prior `>= 0.6` / otherwise | `merge-or-review` / `review-vs-community` / `review` |\n| API/tool skill: zero calls and overlap `>= 0.75` / community prior `>= 0.6` / otherwise | `merge-delete` / `review-vs-community` / `merge-or-review` |\n| community prior `>= 0.6` | `review-vs-community` |\n| score `< 3.0` | `delete` |\n| otherwise, including overlap `>= 0.65` with calls `<= 1` | `merge-delete` |\n\nFile v0.3.24:skill-card.md\n\n## Description:\n\nReview your installed agent skills to see what you actually use, what overlaps, and what may no longer be worth keeping.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[gongyu0918-debug](https://clawhub.ai/user/gongyu0918-debug)\n\n### License/Terms of Use:\n\nMIT No Attribution (MIT-0)\n\n## Use Case:\n\nDevelopers and agent operators use this skill to audit installed agent skills for usage, overlap, cleanup options, bundle burden, and conservative ablation planning. It helps produce short user-facing reports and optional detailed evidence files without automatically deleting or disabling skills.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The skill reads installed skill folders and user-provided evidence files during local audits.\n\nMitigation: Pass --skills-root to limit the audited scope and review which evidence files are provided.\n\nRisk: Usage or history files may contain conversation details, local paths, project names, or customer data.\n\nMitigation: Provide only necessary evidence files and review or redact sensitive data before running the audit.\n\n## Reference(s):\n\n- [Project homepage](https://github.com/gongyu0918-debug/skill-usefulness-audit)\n- [Ablation Protocol](references/ablation-protocol.md)\n- [Report Narration Prompt](references/report-narration-prompt.md)\n- [Scoring Rubric](references/scoring-rubric.md)\n\n## Skill Output:\n\n**Output Type(s):** [Text, Markdown, JSON, Shell commands, Guidance]\n\n**Output Format:** [Natural-language report, Markdown evidence report, and optional JSON audit or ablation-plan files]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Can consume user-provided usage, history, ablation, and community evidence files; supports English or Chinese reports.]\n\n## Skill Version(s):\n\n0.3.24 (source: frontmatter and server release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v0.3.24:LICENSE\n\nMIT No Attribution\n\nCopyright 2026 gongyu0918-debug\n\nPermission is hereby granted, free of charge, to any person obtaining a copy of this software and associated documentation files (the \"Software\"), to deal in the Software without restriction, including without limitation the rights to use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the Software, and to permit persons to whom the Software is furnished to do so.\n\nTHE SOFTWARE IS PROVIDED \"AS IS\", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.\n\nArchive v0.3.22: 23 files, 72312 bytes\n\nFiles: LICENSE (911b), references/ablation-protocol.md (2820b), references/report-narration-prompt.md (819b), references/scoring-rubric.md (10696b), scripts/skill_usefulness_audit_lib/__init__.py (219b), scripts/skill_usefulness_audit_lib/ablation.py (5024b), scripts/skill_usefulness_audit_lib/cli.py (38170b), scripts/skill_usefulness_audit_lib/common.py (24381b), scripts/skill_usefulness_audit_lib/community.py (7605b), scripts/skill_usefulness_audit_lib/constants.py (12582b), scripts/skill_usefulness_audit_lib/reporting.py (62122b), scripts/skill_usefulness_audit_lib/risk_quality.py (46289b), scripts/skill_usefulness_audit_lib/risk_signatures_encoding.py (271b), scripts/skill_usefulness_audit_lib/risk_signatures_execution.py (687b), scripts/skill_usefulness_audit_lib/risk_signatures_network.py (581b), scripts/skill_usefulness_audit_lib/risk_signatures_sensitive.py (326b), scripts/skill_usefulness_audit_lib/risk_signatures.py (394b), scripts/skill_usefulness_audit_lib/scoring.py (19650b), scripts/skill_usefulness_audit_lib/usage_loader.py (11893b), scripts/skill_usefulness_audit.py (793b), skill-card.md (2015b), SKILL.md (6901b), _meta.json (142b)\n\nFile v0.3.22:SKILL.md\n\n---\nname: skill-usefulness-audit\nslug: skill-usefulness-audit\ndescription: Review your installed agent skills to see what you actually use, what overlaps, and what may no longer be worth keeping.\nversion: 0.3.22\ntags: [\"audit\",\"skills\",\"ablation\",\"openclaw\"]\nuser-invocable: true\ndisable-model-invocation: true\nargument-hint: --skills-root PATH --usage-file FILE\nhomepage: https://github.com/gongyu0918-debug/skill-usefulness-audit\nmetadata: {\"openclaw\":{\"skillKey\":\"skill-usefulness-audit\",\"requires\":{\"bins\":[\"python\"]},\"homepage\":\"https://github.com/gongyu0918-debug/skill-usefulness-audit\"}}\n---\n# Skill Usefulness Audit\n\n## Manual Trigger Only\n\nUse this skill only after a direct request to audit installed agent skills, their usage, overlap, cleanup options, or a structure-only inventory.\nDo not invoke it during normal tasks or use it for ordinary repository/source-code review, general security audit, or employee/human skill assessment.\n\n## Safety\n\nNever delete, merge, quarantine, isolate, or disable skills automatically.\nTreat `delete`, `merge-delete`, and `quarantine-review` as manual-review recommendations.\nDo not delete skills based only on a structure-only report.\nThis tool does not automatically replay historical conversations; it generates ablation plans and reads ablation result files that the user provides.\n\n## Audit Scope\n\nAudit these layers in order:\n\n1. Usage evidence, including recency and source quality.\n2. Installed metadata, instructions, and functional overlap.\n3. User-provided skill-on versus skill-off results for general skills.\n4. Runtime and bundle burden, including over-triggering, context cost, weak progressive disclosure, redundant resources, script failures, and private-looking files.\n5. Static health and risk hints.\n6. Optional offline community or registry metrics.\n\nTreat API and tool skills as protected capability skills during ablation.\nExamples: Excel, DOCX, PDF, browser automation, deployment, OCR, external API wrappers, MCP/API gateway helpers.\n\n## Workflow\n\n1. Collect user-provided roots before host-local defaults.\n2. Load only the usage, history, ablation, and community evidence that is available.\n3. Inspect each `SKILL.md` and its script/reference/asset metrics.\n4. Let the bundled script classify each skill as `api`, `tool`, or `general` and calculate its score. Read `{baseDir}/references/scoring-rubric.md` only when checking or explaining a score, verdict, or action.\n5. Print the short usefulness report and, when requested, write Markdown evidence or an ablation plan.\n\n## Ablation Rules\n\nRead `{baseDir}/references/ablation-protocol.md` only when running replays, preparing normalized ablation records, or reviewing mixed or delete-boundary results. The script can generate an ablation plan without loading the protocol.\nReplay only selected `general` candidates with identical prompts/artifacts and pairwise judging.\nDo not fake no-tool ablation for `api` or `tool` skills; use the rubric's protected-capability branch.\n\n## Run the Audit\n\nRun the audit after collecting available evidence:\n\n```bash\nREPORT_LANGUAGE=en  # use zh-CN when the current user invocation is Chinese\npython \"{baseDir}/scripts/skill_usefulness_audit.py\" audit \\\n  --skills-root ./skills \\\n  --report-language \"$REPORT_LANGUAGE\" \\\n  --markdown-out ./skill-audit-report.md\n```\n\nOpenClaw expands `{baseDir}` to the installed skill directory. Use it for bundled scripts and references.\n\nAdd evidence only when available:\n\n- `--usage-file`: JSON, JSONL, CSV, or TSV with per-skill usage.\n- `--history-file`: raw transcripts used only when direct usage is weak or missing; mentions remain `history_mentions` / `suspected_invocations`, not `calls`.\n- `--ablation-file`: normalized JSON or JSONL skill-on/skill-off results.\n- `--community-file`: offline JSON, JSONL, CSV, or TSV registry metrics.\n- `--ablation-plan-out`: a cost estimate and focused replay plan; its case counts can be overridden with the four `--ablation-*-cases` options documented by `--help`.\n- `--json-out`: machine-readable evidence only when requested or needed by another tool.\n\nPass `--report-language zh-CN` for a Chinese invocation and `--report-language en` for an English invocation. `auto` reads `SKILL_AUDIT_REPORT_LANGUAGE` or the process locale, then falls back to English.\n\nRun without extra files only when you need a structure-only audit.\nUsage, community, and ablation evidence become lower-confidence in that mode.\nHistory and usage files may contain sensitive conversations, local paths, project names, and customer data.\nMissing env means not configured in the current audit process, not proof that the skill is broken in every host.\n\n## Output Contract\n\nUse one run for both output layers; do not ask the user to choose a quick or full mode.\nStandard output is a short natural-language report. Its opening paragraph states the audited skill count and the total characters plus approximate tokens of loaded entry descriptions. Lead with actual usage, not static risk or bundle health, and keep scores, internal codes, risk flags, and tables out of this layer.\nWhen `--markdown-out` is provided, write the detailed evidence—with scores, action codes, missing evidence, burden, and risk notes—in the same run.\nMatch the user's language: clean Chinese for `zh-CN` and clean English for `en`, except for skill names and unavoidable paths or commands.\n\nCopy the short report to chat verbatim, apart from making its evidence path clickable. Do not paste raw JSON or the full Markdown evidence unless the user asks. Read `{baseDir}/references/report-narration-prompt.md` only when another agent or host must deliver an already-generated report.\n\nJSON includes `report_mode`, per-skill `score_breakdown`, `quality_penalty`, `quality_penalty_uncapped`, `quality_evidence`, `community_breakdown`, `action_advice`, and `risk_review`. It includes `ablation_plan` only when `--ablation-plan-out` is used. JSON emits both `risk_*` and `static_risk_*` with identical values, and `total_score` as an alias of `local_score`; treat `risk_*` and `local_score` as canonical.\n\nKeep deletion advice conservative for system or host-core skills, and prefer narrowing or merging when overlapping skills still serve distinct host integrations.\n\n## Resources\n\n- `{baseDir}/scripts/skill_usefulness_audit.py`: compatibility wrapper for the modular audit package.\n- `{baseDir}/scripts/skill_usefulness_audit_lib/`: collect metadata, score skills, scan static risk hints, and render Markdown reports plus optional JSON artifacts.\n- `{baseDir}/references/report-narration-prompt.md`: concise prompt for turning the report into a user-facing conversational summary.\n- `{baseDir}/references/scoring-rubric.md`: 10-point scoring rules, confidence logic, community prior, and action thresholds.\n- `{baseDir}/references/ablation-protocol.md`: normalized replay method for historical conversations.\n\nFile v0.3.22:_meta.json\n\n{\n  \"ownerId\": \"kn7em0w89d0zac35fzt84qm2a182j54b\",\n  \"slug\": \"skill-usefulness-audit\",\n  \"version\": \"0.3.22\",\n  \"publishedAt\": 1786266014107\n}\n\nFile v0.3.22:references/ablation-protocol.md\n\n# Ablation Protocol\n\nUse this protocol for `general` skills selected by the ablation plan.\n\n## Goal\n\nMeasure whether the skill changes outcomes in a meaningful way.\nHigh consistency between skill-on and skill-off runs means the skill adds little value.\n\n## Sampling\n\nStart with `3` historical tasks where the skill should plausibly matter. Prefer real user turns over synthetic prompts. Expand to `5` when results are mixed and to `10` only for high-impact or delete-boundary decisions.\n\n## Replay Method\n\nFor each selected case, run two isolated replays:\n\n1. `with_skill`\n2. `without_skill`\n\nKeep these constant:\n\n- same prompt\n- same files and artifacts\n- same model class when possible\n- same tool permissions\n- same success criteria\n\nUse a fresh thread or isolated run if the host supports it.\n\n## Judge Method\n\nFor open-ended outputs:\n\n1. Compare `with_skill` and `without_skill` side by side.\n2. Randomize A/B order.\n3. Spot-check reversed order on boundary cases.\n4. Prefer `pass/fail`, `same/better/worse`, and short reasons over long open-ended grading.\n\nRecord a standard `verdict` and one short `notes` reason. Each arm may also include optional `pass` and/or `score` from `0.0-1.0` for fallback inference. Optional `tool_cost` may describe calls, latency, or retries and currently does not affect the audit score.\n\n### Normalized JSON\n\n```json\n[\n  {\n    \"skill\": \"emotion-orchestrator\",\n    \"case_id\": \"case-001\",\n    \"with_skill\": {\"pass\": true, \"score\": 0.92},\n    \"without_skill\": {\"pass\": true, \"score\": 0.81},\n    \"verdict\": \"better\"\n  }\n]\n```\n\n## Judgment Rule\n\nUse `same` when the final answer, correctness, and workflow remain materially equivalent.\nUse `better` when the skill improves correctness, speed, structure, or user-fit in a way the baseline did not.\nUse `worse` when the skill adds friction, drift, or errors.\n\nIgnore verdict-only cases with unsupported values. A case with an unknown or missing verdict is usable only when both arms provide comparable `pass` and/or `score` fields for inference.\n\n## Early Stop Rules\n\n- Stop as low-value when `3/3` cases are `same` and `better_rate` is `0`.\n- Stop as useful when at least `2/3` cases are `better` and no case is `worse`.\n- Expand to `5` when the first batch is mixed.\n- Expand to `10` only for delete-boundary or high-impact decisions.\n\nDelete-boundary means \"needs stronger human review evidence\", not automatic deletion authority.\n\n## Planning and Model Cost\n\nCreate the replay plan with `--ablation-plan-out`. Planning uses local evidence and does not call an LLM.\n\nThe plan estimates replay cost for `light`, `realistic`, and `coding` profiles. Each case assumes two replays and one compact pairwise judge. `model_cost_estimates.unit` records `estimated_context_units_per_case`.\n\nFeed normalized results back with `--ablation-file`.\n\nFile v0.3.22:references/report-narration-prompt.md\n\n# Report Delivery Contract\n\n- Treat standard output as the final short report. Copy it verbatim into chat, apart from making the evidence path clickable when supported.\n- Do not add headings, bullets, extra counts, verification notes, or facts from the Markdown evidence unless requested.\n- Keep the short report on actual use, missing use evidence, overlap, and verified outcome impact. Leave scores, internal codes, risk, bundle health, and tables in the Markdown evidence.\n- Match the user's language: clean Chinese for `zh-CN` and clean English for `en`, except for skill names and unavoidable paths or commands.\n- Do not paste raw JSON or read back the full Markdown report unless requested.\n- Treat removal results as manual-review recommendations. Never remove, merge, isolate, or disable a skill automatically.\n\nFile v0.3.22:references/scoring-rubric.md\n\n# Scoring Rubric\n\n## Contents\n\n- Core Outputs\n- Usage Score\n- Uniqueness Score\n- Impact Score\n- Confidence Score\n- Quality Penalty\n- Community Prior Score\n- Static Risk Level\n- Verdict Bands\n- Action Rules\n\n## Core Outputs\n\n- `local_score = usage_score + uniqueness_score + impact_score`\n- `quality_penalty`: `0.0-2.5`\n- `quality_penalty_uncapped`: raw quality burden before the cap\n- `static_quality_penalty`: `0.0-1.4`\n- `final_score = clamp(local_score - quality_penalty, 0.0, 10.0)`\n- `risk_level` / `static_risk_level`: `none / low / medium / high`\n\n## 1. Usage Score (`0.0-3.0`)\n\nPrefer direct host usage logs.\nUse transcript mentions only as weaker fallback evidence.\n\n### Input Fields\n\n- Direct usage: `calls`, `recent_30d_calls`, `recent_90d_calls`, `last_used_at`, and `active_days`.\n- History fallback: `history_mentions` and `suspected_invocations`. These are weak evidence weighted through history and must not be reported as direct `calls`.\n- Evidence and runtime burden: `usage_source`, `evidence_weight`, `executions`, `script_failures`, `repair_turns`, `reference_loads`, and `false_triggers`.\n\n### Base Usage Strength\n\n- When `recent_30d_calls` exists:\n  - `0.0`: `0`\n  - `1.0`: `1-2`\n  - `2.0`: `3-7`\n  - `3.0`: `8+`\n- When only `recent_90d_calls` exists:\n  - `0.0`: `0`\n  - `0.75`: `1-2`\n  - `1.5`: `3-9`\n  - `2.5`: `10+`\n- When only total `calls` exists:\n  - `0.0`: `0`\n  - `1.0`: `1-2`\n  - `2.0`: `3-9`\n  - `3.0`: `10+`\n\n### Recency Adjustments\n\n- add `0.5` when `last_used_at <= 7 days`\n- add `0.25` when `last_used_at <= 30 days`\n- subtract `0.5` when `last_used_at > 180 days`\n- add `0.25` when `active_days >= 10`\n- add `0.10` when `active_days >= 3`\n\n### Evidence Weight\n\n- `1.00`: direct usage file\n- `0.45`: transcript-history fallback based on `suspected_invocations`\n- `0.00`: missing usage evidence\n\n## 2. Uniqueness Score (`0.0-3.0`)\n\nMeasure the highest functional-overlap similarity against any other installed skill using descriptions, headings, and resource names.\n\nBuckets:\n\n- `0.0`: highest overlap `>= 0.85`\n- `1.0`: highest overlap `0.65-0.84`\n- `2.0`: highest overlap `0.40-0.64`\n- `3.0`: highest overlap `< 0.40`\n\n## 3. Impact Score (`0.0-4.0`)\n\n### General skills\n\nUse ablation on historical conversations.\nCompute:\n\n- `consistency_rate`: skill-on and skill-off produce materially equivalent outcomes\n- `better_rate`: skill-on clearly improves the result\n- `worse_rate`: skill-on clearly harms the result\n\nBase score from consistency:\n\n- `0.0`: `consistency_rate >= 0.85`\n- `1.0`: `0.70-0.84`\n- `2.0`: `0.55-0.69`\n- `3.0`: `0.35-0.54`\n- `4.0`: `< 0.35`\n\nAdjustments:\n\n- add `1.0` when `better_rate - worse_rate >= 0.30`\n- subtract `1.0` when `worse_rate > better_rate`\n\nWhen ablation is missing, use low-evidence score `1.0` for zero-call skills.\nFor skills with direct usage evidence but no ablation yet, keep temporary neutral score `2.0` and lower confidence.\n\n### API and tool skills\n\nSkip history ablation.\nUse protected-capability scoring instead:\n\n- start at `2.0`\n- add `1.0` when the skill ships executable scripts or reference files\n- add `0.5` when highest overlap `< 0.35`\n- add `0.5` when calls `>= 3`\n- subtract `1.0` when highest overlap `>= 0.75`\n- subtract `0.5` when calls are `0`\n\n## 4. Confidence Score (`0.0-1.0`)\n\nConfidence describes evidence quality, not usefulness.\n\nAdd:\n\n- `0.35` for direct usage files\n- `0.15` for history fallback\n- `0.20` when recent usage fields exist\n- `0.10` when only total direct calls exist\n- `0.25` for protected `api/tool` classification\n- `0.25` for `general` skills with `>= 5` ablation cases\n- `0.15` for `general` skills with `1-4` ablation cases\n- `0.10` when more than one skill is in scope\n- `0.05` when only one skill exists in scope\n- `0.10` when community metadata exists\n\n## 5. Quality Penalty (`0.0-2.5`)\n\nQuality penalty captures the cost of keeping a skill and is deducted from `local_score`; it is not a risk flag.\n\n### Runtime burden\n\nUse direct usage logs when available:\n\n- `overtrigger-low-execution`: `0.45` when `calls >= 8` and `executions / calls < 0.25`\n- `overtrigger-misfire`: `0.35` when `calls >= 5` and (`false_triggers >= 3` or `false_triggers / calls >= 0.25`)\n- `overtrigger-no-impact`: `0.40` when `calls >= 5`, `consistency_rate >= 0.85`, and `better_rate <= 0.10`\n- `reference-overload`: `0.30` when `reference_loads >= 10` and `reference_loads / calls >= 3.0`\n- `script-failure-burden`: `0.45` when `script_failures >= 3` or the failure rate against executions (or calls) reaches `0.30`; `0.20` below that\n- `agent-repair-burden`: `0.30` when `repair_turns >= 3`\n\n### Readiness burden\n\n- `missing-required-env`: `0.90` when declared required environment variables are not configured in the current audit process\n\n### Catalog burden\n\n- `near-duplicate-instructions`: `0.10` when the instruction fingerprint closely matches another installed skill\n\n### Static bundle burden\n\nScan installed skill files:\n\n- `empty-skill-contract`: `0.80` when the skill has no meaningful runtime contract beyond minimal or missing metadata\n- `prompt-bloat`: `0.40` when `SKILL.md` body is at least `5000` context units\n- `prompt-bloat`: `0.20` when `SKILL.md` body is at least `2500` context units\n- `broad-trigger-surface`: `0.25` for at least two broad trigger matches, or one match with description at `30+` context units\n- `description-bloat`: `0.25` when the frontmatter description is at least `120` context units\n- `description-bloat`: `0.10` when the frontmatter description is at least `60` context units\n- `reference-disclosure-gap`: `0.30` when at least 3 reference files exist and none are directly discoverable from `SKILL.md`\n- `reference-disclosure-gap`: `0.10` when 1-2 reference files exist and none are directly discoverable from `SKILL.md`\n- `reference-disclosure-gap`: `0.20` when at least 8 reference files exist and fewer than 30% are directly linked from `SKILL.md`\n- `reference-link-broken`: `0.25` when `SKILL.md` points to missing reference files\n- `reference-bloat`: `0.50` when references are at least 50 files or 50000 context units\n- `reference-bloat`: `0.25` when references are at least 20 files or 15000 context units\n- `long-reference-without-toc`: `0.20` when at least 3 reference files over `100` lines lack a visible table of contents\n- `long-reference-without-toc`: `0.10` when 1-2 such files lack one\n- `reference-content-pollution`: `0.35` when references include advertising, upsells, unrelated text, or other-tool/skill promotion\n- `asset-bloat`: `0.50` when assets are at least 200 files or 100 MB\n- `asset-bloat`: `0.25` when assets are at least 50 files or 25 MB\n- `vague-resource-names`: `0.20` when at least 5 scripts, references, or assets use generic filenames\n- `private-bundle-artifact`: `0.60` when bundled filenames look private or environment-specific\n- `private-content-artifact`: `0.60` when bundled content looks like credentials or keys\n- `executable-asset`: `0.30` when assets contain executable binaries or installers\n- `script-count-bloat`: `0.20` when the bundle has at least 40 scripts\n- `script-count-bloat`: `0.10` when the bundle has at least 20 scripts\n- `script-maintenance-smell`: `0.40` when at least 8 scripts contain placeholders, local absolute paths, or maintenance smells\n- `script-maintenance-smell`: `0.25` when 1-7 scripts contain placeholders, local absolute paths, or maintenance smells\n- `script-syntax-error`: `0.50` when Python scripts contain syntax errors\n- `script-import-error`: `0.50` when Python scripts import modules missing from the local environment or bundle\n\nClamp `static_quality_penalty` to `0.0-1.4`, then clamp the combined `quality_penalty` to `0.0-2.5`.\n\n## 6. Community Prior Score (`0.0-1.0`)\n\nTreat community data as external prior, not a local verdict.\n\nWeighted components:\n\n- `0.30`: normalized rating\n- `0.20`: current installs or downloads\n- `0.10`: all-time installs\n- `0.15`: trending metric\n- `0.10`: stars\n- `0.05`: comments\n- `0.10`: maintenance freshness from `last_updated`\n\nNormalization: rating divides by `5.0`; volume signals use `log1p` with saturation `5000` (current), `20000` (all-time), `250` (trending, stars), and `100` (comments); maintenance scores `1.0/0.7/0.4/0.1` at `<=180/<=365/<=730/>730` days.\n\n## 7. Static Risk Level\n\nRun static scans against runnable scripts and resource files.\nOnly fenced code blocks in `SKILL.md` and directly linked Markdown references are scanned as commands; prose outside fences and unlinked references are not command-scanned.\nCredential-like content checks still cover `SKILL.md`, scripts, assets, references, and root text files without echoing matched values.\nThis is lint-style evidence only; it cannot prove a skill is safe.\n\nTypical flags: `curl-pipe-shell`, `dynamic-exec`, `protected-path-access`, `persistence-hook`, `external-post`, `shell-exec`, `network-download`, and `base64-payload`.\n\nStatic risk levels:\n\n- `none`: `0.0`\n- `low`: `0.0 < score < 2.0`\n- `medium`: `2.0-3.9`\n- `high`: `4.0+`\n\nIf static quality finds `private-content-artifact`, that evidence is promoted to `high` risk so credential-like bundled content receives `quarantine-review`.\n\n## Health Cap\n\nSome quality findings cap the final score even when usage or protected-capability signals are strong:\n\n- `script-syntax-error`: final score cap `4.0`\n- `empty-skill-contract`: final score cap `5.5`\n- `script-import-error`: final score cap `5.5`\n- `script-failure-burden`: final score cap `4.0` when the penalty is at least `0.45`\n\n## Verdict Bands\n\nUse `final_score` for verdict bands.\n\n- confidence `< 0.55` and `final_score < 4.5`: `insufficient-evidence`\n- `8.0-10.0`: `keep`\n- `6.0-7.9`: `keep-narrow`\n- `4.5-5.9`: `review`\n- `3.0-4.4`: `merge-delete`\n- `0.0-2.9`: `delete`\n\n## Action Rules\n\nEvaluate top to bottom; the first matching rule wins.\n\n| Condition | Action |\n| --- | --- |\n| system source, high risk / otherwise | `review-system` / `keep-system` |\n| high risk | `quarantine-review` |\n| medium risk and score `>= 6.0` | `keep-review-risk` |\n| quality penalty `>= 1.2` and score `>= 6.0` / `>= 4.5` | `keep-review-burden` / `review-burden` |\n| score `>= 8.0` / `>= 6.0` | `keep` / `keep-narrow` |\n| remaining medium risk | `review-risk` |\n| confidence `< 0.55` | `observe-30d` |\n| score `>= 4.5`: overlap `>= 0.65` / community prior `>= 0.6` / otherwise | `merge-or-review` / `review-vs-community` / `review` |\n| API/tool skill: zero calls and overlap `>= 0.75` / community prior `>= 0.6` / otherwise | `merge-delete` / `review-vs-community` / `merge-or-review` |\n| community prior `>= 0.6` | `review-vs-community` |\n| score `< 3.0` | `delete` |\n| otherwise, including overlap `>= 0.65` with calls `<= 1` | `merge-delete` |\n\nFile v0.3.22:skill-card.md\n\n## Description:\n\nReview your installed agent skills to see what you actually use, what overlaps, and what may no longer be worth keeping.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[gongyu0918-debug](https://clawhub.ai/user/gongyu0918-debug)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and agent users use this skill to audit installed agent skills for actual usage, overlap, cleanup candidates, runtime burden, static health hints, and optional ablation evidence.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Usage and history files may contain sensitive conversations, local paths, project names, or customer data.\n\nMitigation: Provide only files that are appropriate for local processing and avoid unnecessary sensitive evidence.\n\nRisk: Cleanup recommendations could lead to accidental loss of useful installed skills if acted on automatically.\n\nMitigation: Review delete, merge, quarantine, isolate, or disable recommendations manually before changing installed skills.\n\n## Reference(s):\n\n- [Project homepage](https://github.com/gongyu0918-debug/skill-usefulness-audit)\n- [Ablation Protocol](references/ablation-protocol.md)\n- [Report Delivery Contract](references/report-narration-prompt.md)\n- [Scoring Rubric](references/scoring-rubric.md)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, json, shell commands, guidance]\n\n**Output Format:** [Natural-language report, Markdown evidence, and optional JSON audit artifacts]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Can also produce an ablation plan and write local report files when requested.]\n\n## Skill Version(s):\n\n0.3.22 (source: frontmatter and server release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v0.3.22:LICENSE\n\nMIT No Attribution\n\nCopyright 2026 gongyu0918-debug\n\nPermission is hereby granted, free of charge, to any person obtaining a copy of this software and associated documentation files (the \"Software\"), to deal in the Software without restriction, including without limitation the rights to use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the Software, and to permit persons to whom the Software is furnished to do so.\n\nTHE SOFTWARE IS PROVIDED \"AS IS\", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.\n\nArchive v0.3.21: 23 files, 72690 bytes\n\nFiles: LICENSE (911b), references/ablation-protocol.md (2820b), references/report-narration-prompt.md (819b), references/scoring-rubric.md (11563b), scripts/skill_usefulness_audit_lib/__init__.py (219b), scripts/skill_usefulness_audit_lib/ablation.py (5024b), scripts/skill_usefulness_audit_lib/cli.py (38170b), scripts/skill_usefulness_audit_lib/common.py (24381b), scripts/skill_usefulness_audit_lib/community.py (7605b), scripts/skill_usefulness_audit_lib/constants.py (12582b), scripts/skill_usefulness_audit_lib/reporting.py (62122b), scripts/skill_usefulness_audit_lib/risk_quality.py (46289b), scripts/skill_usefulness_audit_lib/risk_signatures_encoding.py (271b), scripts/skill_usefulness_audit_lib/risk_signatures_execution.py (687b), scripts/skill_usefulness_audit_lib/risk_signatures_network.py (581b), scripts/skill_usefulness_audit_lib/risk_signatures_sensitive.py (326b), scripts/skill_usefulness_audit_lib/risk_signatures.py (394b), scripts/skill_usefulness_audit_lib/scoring.py (19650b), scripts/skill_usefulness_audit_lib/usage_loader.py (11893b), scripts/skill_usefulness_audit.py (793b), skill-card.md (2159b), SKILL.md (6901b), _meta.json (142b)\n\nFile v0.3.21:SKILL.md\n\n---\nname: skill-usefulness-audit\nslug: skill-usefulness-audit\ndescription: Review your installed agent skills to see what you actually use, what overlaps, and what may no longer be worth keeping.\nversion: 0.3.21\ntags: [\"audit\",\"skills\",\"ablation\",\"openclaw\"]\nuser-invocable: true\ndisable-model-invocation: true\nargument-hint: --skills-root PATH --usage-file FILE\nhomepage: https://github.com/gongyu0918-debug/skill-usefulness-audit\nmetadata: {\"openclaw\":{\"skillKey\":\"skill-usefulness-audit\",\"requires\":{\"bins\":[\"python\"]},\"homepage\":\"https://github.com/gongyu0918-debug/skill-usefulness-audit\"}}\n---\n# Skill Usefulness Audit\n\n## Manual Trigger Only\n\nUse this skill only after a direct request to audit installed agent skills, their usage, overlap, cleanup options, or a structure-only inventory.\nDo not invoke it during normal tasks or use it for ordinary repository/source-code review, general security audit, or employee/human skill assessment.\n\n## Safety\n\nNever delete, merge, quarantine, isolate, or disable skills automatically.\nTreat `delete`, `merge-delete`, and `quarantine-review` as manual-review recommendations.\nDo not delete skills based only on a structure-only report.\nThis tool does not automatically replay historical conversations; it generates ablation plans and reads ablation result files that the user provides.\n\n## Audit Scope\n\nAudit these layers in order:\n\n1. Usage evidence, including recency and source quality.\n2. Installed metadata, instructions, and functional overlap.\n3. User-provided skill-on versus skill-off results for general skills.\n4. Runtime and bundle burden, including over-triggering, context cost, weak progressive disclosure, redundant resources, script failures, and private-looking files.\n5. Static health and risk hints.\n6. Optional offline community or registry metrics.\n\nTreat API and tool skills as protected capability skills during ablation.\nExamples: Excel, DOCX, PDF, browser automation, deployment, OCR, external API wrappers, MCP/API gateway helpers.\n\n## Workflow\n\n1. Collect user-provided roots before host-local defaults.\n2. Load only the usage, history, ablation, and community evidence that is available.\n3. Inspect each `SKILL.md` and its script/reference/asset metrics.\n4. Let the bundled script classify each skill as `api`, `tool`, or `general` and calculate its score. Read `{baseDir}/references/scoring-rubric.md` only when checking or explaining a score, verdict, or action.\n5. Print the short usefulness report and, when requested, write Markdown evidence or an ablation plan.\n\n## Ablation Rules\n\nRead `{baseDir}/references/ablation-protocol.md` only when running replays, preparing normalized ablation records, or reviewing mixed or delete-boundary results. The script can generate an ablation plan without loading the protocol.\nReplay only selected `general` candidates with identical prompts/artifacts and pairwise judging.\nDo not fake no-tool ablation for `api` or `tool` skills; use the rubric's protected-capability branch.\n\n## Run the Audit\n\nRun the audit after collecting available evidence:\n\n```bash\nREPORT_LANGUAGE=en  # use zh-CN when the current user invocation is Chinese\npython \"{baseDir}/scripts/skill_usefulness_audit.py\" audit \\\n  --skills-root ./skills \\\n  --report-language \"$REPORT_LANGUAGE\" \\\n  --markdown-out ./skill-audit-report.md\n```\n\nOpenClaw expands `{baseDir}` to the installed skill directory. Use it for bundled scripts and references.\n\nAdd evidence only when available:\n\n- `--usage-file`: JSON, JSONL, CSV, or TSV with per-skill usage.\n- `--history-file`: raw transcripts used only when direct usage is weak or missing; mentions remain `history_mentions` / `suspected_invocations`, not `calls`.\n- `--ablation-file`: normalized JSON or JSONL skill-on/skill-off results.\n- `--community-file`: offline JSON, JSONL, CSV, or TSV registry metrics.\n- `--ablation-plan-out`: a cost estimate and focused replay plan; its case counts can be overridden with the four `--ablation-*-cases` options documented by `--help`.\n- `--json-out`: machine-readable evidence only when requested or needed by another tool.\n\nPass `--report-language zh-CN` for a Chinese invocation and `--report-language en` for an English invocation. `auto` reads `SKILL_AUDIT_REPORT_LANGUAGE` or the process locale, then falls back to English.\n\nRun without extra files only when you need a structure-only audit.\nUsage, community, and ablation evidence become lower-confidence in that mode.\nHistory and usage files may contain sensitive conversations, local paths, project names, and customer data.\nMissing env means not configured in the current audit process, not proof that the skill is broken in every host.\n\n## Output Contract\n\nUse one run for both output layers; do not ask the user to choose a quick or full mode.\nStandard output is a short natural-language report. Its opening paragraph states the audited skill count and the total characters plus approximate tokens of loaded entry descriptions. Lead with actual usage, not static risk or bundle health, and keep scores, internal codes, risk flags, and tables out of this layer.\nWhen `--markdown-out` is provided, write the detailed evidence—with scores, action codes, missing evidence, burden, and risk notes—in the same run.\nMatch the user's language: clean Chinese for `zh-CN` and clean English for `en`, except for skill names and unavoidable paths or commands.\n\nCopy the short report to chat verbatim, apart from making its evidence path clickable. Do not paste raw JSON or the full Markdown evidence unless the user asks. Read `{baseDir}/references/report-narration-prompt.md` only when another agent or host must deliver an already-generated report.\n\nJSON includes `report_mode`, per-skill `score_breakdown`, `quality_penalty`, `quality_penalty_uncapped`, `quality_evidence`, `community_breakdown`, `action_advice`, and `risk_review`. It includes `ablation_plan` only when `--ablation-plan-out` is used. JSON emits both `risk_*` and `static_risk_*` with identical values, and `total_score` as an alias of `local_score`; treat `risk_*` and `local_score` as canonical.\n\nKeep deletion advice conservative for system or host-core skills, and prefer narrowing or merging when overlapping skills still serve distinct host integrations.\n\n## Resources\n\n- `{baseDir}/scripts/skill_usefulness_audit.py`: compatibility wrapper for the modular audit package.\n- `{baseDir}/scripts/skill_usefulness_audit_lib/`: collect metadata, score skills, scan static risk hints, and render Markdown reports plus optional JSON artifacts.\n- `{baseDir}/references/report-narration-prompt.md`: concise prompt for turning the report into a user-facing conversational summary.\n- `{baseDir}/references/scoring-rubric.md`: 10-point scoring rules, confidence logic, community prior, and action thresholds.\n- `{baseDir}/references/ablation-protocol.md`: normalized replay method for historical conversations.\n\nFile v0.3.21:_meta.json\n\n{\n  \"ownerId\": \"kn7em0w89d0zac35fzt84qm2a182j54b\",\n  \"slug\": \"skill-usefulness-audit\",\n  \"version\": \"0.3.21\",\n  \"publishedAt\": 1786174096479\n}\n\nFile v0.3.21:references/ablation-protocol.md\n\n# Ablation Protocol\n\nUse this protocol for `general` skills selected by the ablation plan.\n\n## Goal\n\nMeasure whether the skill changes outcomes in a meaningful way.\nHigh consistency between skill-on and skill-off runs means the skill adds little value.\n\n## Sampling\n\nStart with `3` historical tasks where the skill should plausibly matter. Prefer real user turns over synthetic prompts. Expand to `5` when results are mixed and to `10` only for high-impact or delete-boundary decisions.\n\n## Replay Method\n\nFor each selected case, run two isolated replays:\n\n1. `with_skill`\n2. `without_skill`\n\nKeep these constant:\n\n- same prompt\n- same files and artifacts\n- same model class when possible\n- same tool permissions\n- same success criteria\n\nUse a fresh thread or isolated run if the host supports it.\n\n## Judge Method\n\nFor open-ended outputs:\n\n1. Compare `with_skill` and `without_skill` side by side.\n2. Randomize A/B order.\n3. Spot-check reversed order on boundary cases.\n4. Prefer `pass/fail`, `same/better/worse`, and short reasons over long open-ended grading.\n\nRecord a standard `verdict` and one short `notes` reason. Each arm may also include optional `pass` and/or `score` from `0.0-1.0` for fallback inference. Optional `tool_cost` may describe calls, latency, or retries and currently does not affect the audit score.\n\n### Normalized JSON\n\n```json\n[\n  {\n    \"skill\": \"emotion-orchestrator\",\n    \"case_id\": \"case-001\",\n    \"with_skill\": {\"pass\": true, \"score\": 0.92},\n    \"without_skill\": {\"pass\": true, \"score\": 0.81},\n    \"verdict\": \"better\"\n  }\n]\n```\n\n## Judgment Rule\n\nUse `same` when the final answer, correctness, and workflow remain materially equivalent.\nUse `better` when the skill improves correctness, speed, structure, or user-fit in a way the baseline did not.\nUse `worse` when the skill adds friction, drift, or errors.\n\nIgnore verdict-only cases with unsupported values. A case with an unknown or missing verdict is usable only when both arms provide comparable `pass` and/or `score` fields for inference.\n\n## Early Stop Rules\n\n- Stop as low-value when `3/3` cases are `same` and `better_rate` is `0`.\n- Stop as useful when at least `2/3` cases are `better` and no case is `worse`.\n- Expand to `5` when the first batch is mixed.\n- Expand to `10` only for delete-boundary or high-impact decisions.\n\nDelete-boundary means \"needs stronger human review evidence\", not automatic deletion authority.\n\n## Planning and Model Cost\n\nCreate the replay plan with `--ablation-plan-out`. Planning uses local evidence and does not call an LLM.\n\nThe plan estimates replay cost for `light`, `realistic`, and `coding` profiles. Each case assumes two replays and one compact pairwise judge. `model_cost_estimates.unit` records `estimated_context_units_per_case`.\n\nFeed normalized results back with `--ablation-file`.\n\nFile v0.3.21:references/report-narration-prompt.md\n\n# Report Delivery Contract\n\n- Treat standard output as the final short report. Copy it verbatim into chat, apart from making the evidence path clickable when supported.\n- Do not add headings, bullets, extra counts, verification notes, or facts from the Markdown evidence unless requested.\n- Keep the short report on actual use, missing use evidence, overlap, and verified outcome impact. Leave scores, internal codes, risk, bundle health, and tables in the Markdown evidence.\n- Match the user's language: clean Chinese for `zh-CN` and clean English for `en`, except for skill names and unavoidable paths or commands.\n- Do not paste raw JSON or read back the full Markdown report unless requested.\n- Treat removal results as manual-review recommendations. Never remove, merge, isolate, or disable a skill automatically.\n\nFile v0.3.21:references/scoring-rubric.md\n\n# Scoring Rubric\n\n## Contents\n\n- Core Outputs\n- Usage Score\n- Uniqueness Score\n- Impact Score\n- Confidence Score\n- Quality Penalty\n- Community Prior Score\n- Static Risk Level\n- Verdict Bands\n- Action Rules\n\n## Core Outputs\n\n- `local_score = usage_score + uniqueness_score + impact_score`\n- `quality_penalty`: `0.0-2.5`\n- `quality_penalty_uncapped`: raw quality burden before the cap\n- `static_quality_penalty`: `0.0-1.4`\n- `final_score = clamp(local_score - quality_penalty, 0.0, 10.0)`\n- `risk_level` / `static_risk_level`: `none / low / medium / high`\n\nKeep community prior and risk separate from `local_score`; use them with quality burden to shape review and action.\n\n## 1. Usage Score (`0.0-3.0`)\n\nPrefer direct host usage logs.\nUse transcript mentions only as weaker fallback evidence.\n\n### Input Fields\n\n- Direct usage: `calls`, `recent_30d_calls`, `recent_90d_calls`, `last_used_at`, and `active_days`.\n- History fallback: `history_mentions` and `suspected_invocations`. These are weak evidence weighted through history and must not be reported as direct `calls`.\n- Evidence and runtime burden: `usage_source`, `evidence_weight`, `executions`, `script_failures`, `repair_turns`, `reference_loads`, and `false_triggers`.\n\n### Base Usage Strength\n\n- When `recent_30d_calls` exists:\n  - `0.0`: `0`\n  - `1.0`: `1-2`\n  - `2.0`: `3-7`\n  - `3.0`: `8+`\n- When only `recent_90d_calls` exists:\n  - `0.0`: `0`\n  - `0.75`: `1-2`\n  - `1.5`: `3-9`\n  - `2.5`: `10+`\n- When only total `calls` exists:\n  - `0.0`: `0`\n  - `1.0`: `1-2`\n  - `2.0`: `3-9`\n  - `3.0`: `10+`\n\n### Recency Adjustments\n\n- add `0.5` when `last_used_at <= 7 days`\n- add `0.25` when `last_used_at <= 30 days`\n- subtract `0.5` when `last_used_at > 180 days`\n- add `0.25` when `active_days >= 10`\n- add `0.10` when `active_days >= 3`\n\n### Evidence Weight\n\n- `1.00`: direct usage file\n- `0.45`: transcript-history fallback based on `suspected_invocations`\n- `0.00`: missing usage evidence\n\nClamp the final usage score to `0.0-3.0`.\n\n## 2. Uniqueness Score (`0.0-3.0`)\n\nMeasure the highest functional-overlap similarity against any other installed skill using descriptions, headings, and resource names.\n\nBuckets:\n\n- `0.0`: highest overlap `>= 0.85`\n- `1.0`: highest overlap `0.65-0.84`\n- `2.0`: highest overlap `0.40-0.64`\n- `3.0`: highest overlap `< 0.40`\n\n## 3. Impact Score (`0.0-4.0`)\n\n### General skills\n\nUse ablation on historical conversations.\nCompute:\n\n- `consistency_rate`: skill-on and skill-off produce materially equivalent outcomes\n- `better_rate`: skill-on clearly improves the result\n- `worse_rate`: skill-on clearly harms the result\n\nBase score from consistency:\n\n- `0.0`: `consistency_rate >= 0.85`\n- `1.0`: `0.70-0.84`\n- `2.0`: `0.55-0.69`\n- `3.0`: `0.35-0.54`\n- `4.0`: `< 0.35`\n\nAdjustments:\n\n- add `1.0` when `better_rate - worse_rate >= 0.30`\n- subtract `1.0` when `worse_rate > better_rate`\n- clamp the final impact score to `0.0-4.0`\n\nWhen ablation is missing, use low-evidence score `1.0` for zero-call skills.\nFor skills with direct usage evidence but no ablation yet, keep temporary neutral score `2.0` and lower confidence.\n\n### API and tool skills\n\nSkip history ablation.\nUse protected-capability scoring instead:\n\n- start at `2.0`\n- add `1.0` when the skill ships executable scripts or reference files\n- add `0.5` when highest overlap `< 0.35`\n- add `0.5` when calls `>= 3`\n- subtract `1.0` when highest overlap `>= 0.75`\n- subtract `0.5` when calls are `0`\n- clamp the final impact score to `0.0-4.0`\n\n## 4. Confidence Score (`0.0-1.0`)\n\nConfidence describes evidence quality, not usefulness.\n\nAdd:\n\n- `0.35` for direct usage files\n- `0.15` for history fallback\n- `0.20` when recent usage fields exist\n- `0.10` when only total direct calls exist\n- `0.25` for protected `api/tool` classification\n- `0.25` for `general` skills with `>= 5` ablation cases\n- `0.15` for `general` skills with `1-4` ablation cases\n- `0.10` when more than one skill is in scope\n- `0.05` when only one skill exists in scope\n- `0.10` when community metadata exists\n\nClamp the final confidence score to `0.0-1.0`.\n\n## 5. Quality Penalty (`0.0-2.5`)\n\nQuality penalty captures the cost of keeping a skill and is deducted from `local_score`; it is not a risk flag.\n\n### Runtime burden\n\nUse direct usage logs when available:\n\n- `overtrigger-low-execution`: `0.45` when `calls >= 8` and `executions / calls < 0.25`\n- `overtrigger-misfire`: `0.35` when `calls >= 5` and (`false_triggers >= 3` or `false_triggers / calls >= 0.25`)\n- `overtrigger-no-impact`: `0.40` when `calls >= 5`, `consistency_rate >= 0.85`, and `better_rate <= 0.10`\n- `reference-overload`: `0.30` when `reference_loads >= 10` and `reference_loads / calls >= 3.0`\n- `script-failure-burden`: `0.45` when `script_failures >= 3` or the failure rate against executions (or calls) reaches `0.30`; `0.20` below that\n- `agent-repair-burden`: `0.30` when `repair_turns >= 3`\n\n### Readiness burden\n\n- `missing-required-env`: `0.90` when declared required environment variables are not configured in the current audit process\n\n### Catalog burden\n\n- `near-duplicate-instructions`: `0.10` when the instruction fingerprint closely matches another installed skill\n\n### Static bundle burden\n\nScan installed skill files:\n\n- `empty-skill-contract`: `0.80` when the skill has no meaningful runtime contract beyond minimal or missing metadata\n- `prompt-bloat`: `0.40` when `SKILL.md` body is at least `5000` context units\n- `prompt-bloat`: `0.20` when `SKILL.md` body is at least `2500` context units\n- `broad-trigger-surface`: `0.25` for at least two broad trigger matches, or one match with description at `30+` context units\n- `description-bloat`: `0.25` when the frontmatter description is at least `120` context units\n- `description-bloat`: `0.10` when the frontmatter description is at least `60` context units\n- `reference-disclosure-gap`: `0.30` when at least 3 reference files exist and none are directly discoverable from `SKILL.md`\n- `reference-disclosure-gap`: `0.10` when 1-2 reference files exist and none are directly discoverable from `SKILL.md`\n- `reference-disclosure-gap`: `0.20` when at least 8 reference files exist and fewer than 30% are directly linked from `SKILL.md`\n- `reference-link-broken`: `0.25` when `SKILL.md` points to missing reference files\n- `reference-bloat`: `0.50` when references are at least 50 files or 50000 context units\n- `reference-bloat`: `0.25` when references are at least 20 files or 15000 context units\n- `long-reference-without-toc`: `0.20` when at least 3 reference files over `100` lines lack a visible table of contents\n- `long-reference-without-toc`: `0.10` when 1-2 such files lack one\n- `reference-content-pollution`: `0.35` when references include advertising, upsells, unrelated text, or other-tool/skill promotion\n- `asset-bloat`: `0.50` when assets are at least 200 files or 100 MB\n- `asset-bloat`: `0.25` when assets are at least 50 files or 25 MB\n- `vague-resource-names`: `0.20` when at least 5 scripts, references, or assets use generic filenames\n- `private-bundle-artifact`: `0.60` when bundled filenames look private or environment-specific\n- `private-content-artifact`: `0.60` when bundled content looks like credentials or keys\n- `executable-asset`: `0.30` when assets contain executable binaries or installers\n- `script-count-bloat`: `0.20` when the bundle has at least 40 scripts\n- `script-count-bloat`: `0.10` when the bundle has at least 20 scripts\n- `script-maintenance-smell`: `0.40` when at least 8 scripts contain placeholders, local absolute paths, or maintenance smells\n- `script-maintenance-smell`: `0.25` when 1-7 scripts contain placeholders, local absolute paths, or maintenance smells\n- `script-syntax-error`: `0.50` when Python scripts contain syntax errors\n- `script-import-error`: `0.50` when Python scripts import modules missing from the local environment or bundle\n\nClamp `static_quality_penalty` to `0.0-1.4`, then clamp the combined `quality_penalty` to `0.0-2.5`.\nEmit `quality_flags`, `quality_evidence`, `resource_metrics`, `quality_penalty_uncapped`, and `score_breakdown.quality`.\n\n## 6. Community Prior Score (`0.0-1.0`)\n\nTreat community data as external prior, not a local verdict.\n\nWeighted components:\n\n- `0.30`: normalized rating\n- `0.20`: current installs or downloads\n- `0.10`: all-time installs\n- `0.15`: trending metric\n- `0.10`: stars\n- `0.05`: comments\n- `0.10`: maintenance freshness from `last_updated`\n\nNormalization: rating divides by `5.0`; volume signals use `log1p` with saturation `5000` (current), `20000` (all-time), `250` (trending, stars), and `100` (comments); maintenance scores `1.0/0.7/0.4/0.1` at `<=180/<=365/<=730/>730` days.\n\nUse it to rank review priority and benchmark replacements. Emit `community_breakdown` in JSON so users can see which registry signals contributed.\n\n## 7. Static Risk Level\n\nRun static scans against runnable scripts and resource files.\nOnly fenced code blocks in `SKILL.md` and directly linked Markdown references are scanned as commands; prose outside fences and unlinked references are not command-scanned.\nCredential-like content checks still cover `SKILL.md`, scripts, assets, references, and root text files without echoing matched values.\nThis is lint-style evidence only. It cannot prove a skill is safe, because indirection, dynamic imports, encoded payloads, aliases, or external downloads can evade simple pattern matching.\n\nTypical flags: `curl-pipe-shell`, `dynamic-exec`, `protected-path-access`, `persistence-hook`, `external-post`, `shell-exec`, `network-download`, and `base64-payload`.\n\nStatic risk levels:\n\n- `none`: `0.0`\n- `low`: `0.0 < score < 2.0`\n- `medium`: `2.0-3.9`\n- `high`: `4.0+`\n\nIf static quality finds `private-content-artifact`, that evidence is promoted to `high` risk so credential-like bundled content receives `quarantine-review`.\n\n## Health Cap\n\nSome quality findings cap the final score even when usage or protected-capability signals are strong:\n\n- `script-syntax-error`: final score cap `4.0`\n- `empty-skill-contract`: final score cap `5.5`\n- `script-import-error`: final score cap `5.5`\n- `script-failure-burden`: final score cap `4.0` when the penalty is at least `0.45`\n\n## Verdict Bands\n\nUse `final_score` for verdict bands.\n\n- confidence `< 0.55` and `final_score < 4.5`: `insufficient-evidence`\n- `8.0-10.0`: `keep`\n- `6.0-7.9`: `keep-narrow`\n- `4.5-5.9`: `review`\n- `3.0-4.4`: `merge-delete`\n- `0.0-2.9`: `delete`\n\n## Action Rules\n\nEvaluate top to bottom; the first matching rule wins.\n\n| Condition | Action |\n| --- | --- |\n| system source, high risk / otherwise | `review-system` / `keep-system` |\n| high risk | `quarantine-review` |\n| medium risk and score `>= 6.0` | `keep-review-risk` |\n| quality penalty `>= 1.2` and score `>= 6.0` / `>= 4.5` | `keep-review-burden` / `review-burden` |\n| score `>= 8.0` / `>= 6.0` | `keep` / `keep-narrow` |\n| remaining medium risk | `review-risk` |\n| confidence `< 0.55` | `observe-30d` |\n| score `>= 4.5`: overlap `>= 0.65` / community prior `>= 0.6` / otherwise | `merge-or-review` / `review-vs-community` / `review` |\n| API/tool skill: zero calls and overlap `>= 0.75` / community prior `>= 0.6` / otherwise | `merge-delete` / `review-vs-community` / `merge-or-review` |\n| community prior `>= 0.6` | `review-vs-community` |\n| score `< 3.0` | `delete` |\n| otherwise, including overlap `>= 0.65` with calls `<= 1` | `merge-delete` |\n\n`delete`, `merge-delete`, and `quarantine-review` are report recommendations only. They are never permission for automatic deletion, isolation, or disabling without human review.\n\nFile v0.3.21:skill-card.md\n\n## Description:\n\nReview your installed agent skills to see what you actually use, what overlaps, and what may no longer be worth keeping.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[gongyu0918-debug](https://clawhub.ai/user/gongyu0918-debug)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and agent maintainers use this skill to audit installed agent skills for usage, overlap, cleanup candidates, static health, and optional ablation evidence before deciding what to keep, narrow, merge, or remove.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The skill can read local skill folders and optional usage or history files that may contain sensitive conversations, paths, project names, or customer data.\n\nMitigation: Use explicit input paths, limit evidence files to the audit scope, and review generated reports before sharing them.\n\nRisk: Cleanup, deletion, merge, or quarantine recommendations could be mistaken for automatic actions.\n\nMitigation: Treat those outputs as manual-review recommendations only; confirm evidence before changing or removing any skill.\n\n## Reference(s):\n\n- [Ablation Protocol](references/ablation-protocol.md)\n- [Report Narration Prompt](references/report-narration-prompt.md)\n- [Scoring Rubric](references/scoring-rubric.md)\n- [Project homepage](https://github.com/gongyu0918-debug/skill-usefulness-audit)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, JSON, shell commands, guidance]\n\n**Output Format:** [Short natural-language report, optional Markdown evidence report, and optional JSON audit or ablation-plan files.]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Recommendations are manual-review guidance; reports are written only when output paths are provided.]\n\n## Skill Version(s):\n\n0.3.21 (source: frontmatter and server release evidence)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v0.3.21:LICENSE\n\nMIT No Attribution\n\nCopyright 2026 gongyu0918-debug\n\nPermission is hereby granted, free of charge, to any person obtaining a copy of this software and associated documentation files (the \"Software\"), to deal in the Software without restriction, including without limitation the rights to use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the Software, and to permit persons to whom the Software is furnished to do so.\n\nTHE SOFTWARE IS PROVIDED \"AS IS\", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.\n\nArchive v0.3.20: 23 files, 71434 bytes\n\nFiles: LICENSE (911b), references/ablation-protocol.md (2820b), references/report-narration-prompt.md (819b), references/scoring-rubric.md (11312b), scripts/skill_usefulness_audit_lib/__init__.py (219b), scripts/skill_usefulness_audit_lib/ablation.py (5078b), scripts/skill_usefulness_audit_lib/cli.py (38170b), scripts/skill_usefulness_audit_lib/common.py (24079b), scripts/skill_usefulness_audit_lib/community.py (7317b), scripts/skill_usefulness_audit_lib/constants.py (11293b), scripts/skill_usefulness_audit_lib/reporting.py (60621b), scripts/skill_usefulness_audit_lib/risk_quality.py (46000b), scripts/skill_usefulness_audit_lib/risk_signatures_encoding.py (271b), scripts/skill_usefulness_audit_lib/risk_signatures_execution.py (687b), scripts/skill_usefulness_audit_lib/risk_signatures_network.py (581b), scripts/skill_usefulness_audit_lib/risk_signatures_sensitive.py (326b), scripts/skill_usefulness_audit_lib/risk_signatures.py (394b), scripts/skill_usefulness_audit_lib/scoring.py (18994b), scripts/skill_usefulness_audit_lib/usage_loader.py (11893b), scripts/skill_usefulness_audit.py (793b), skill-card.md (2667b), SKILL.md (6738b), _meta.json (142b)\n\nFile v0.3.20:SKILL.md\n\n---\nname: skill-usefulness-audit\nslug: skill-usefulness-audit\ndescription: Review your installed agent skills to see what you actually use, what overlaps, and what may no longer be worth keeping.\nversion: 0.3.20\ntags: [\"audit\",\"skills\",\"ablation\",\"openclaw\"]\nuser-invocable: true\ndisable-model-invocation: true\nargument-hint: --skills-root PATH --usage-file FILE\nhomepage: https://github.com/gongyu0918-debug/skill-usefulness-audit\nmetadata: {\"openclaw\":{\"skillKey\":\"skill-usefulness-audit\",\"requires\":{\"bins\":[\"python\"]},\"homepage\":\"https://github.com/gongyu0918-debug/skill-usefulness-audit\"}}\n---\n# Skill Usefulness Audit\n\n## Manual Trigger Only\n\nUse this skill only after a direct request to audit installed agent skills, their usage, overlap, cleanup options, or a structure-only inventory.\nDo not invoke it during normal tasks or use it for ordinary repository/source-code review, general security audit, or employee/human skill assessment.\n\n## Safety\n\nNever delete, merge, quarantine, isolate, or disable skills automatically.\nTreat `delete`, `merge-delete`, and `quarantine-review` as manual-review recommendations.\nDo not delete skills based only on a structure-only report.\nThis tool does not automatically replay historical conversations; it generates ablation plans and reads ablation result files that the user provides.\n\n## Audit Scope\n\nAudit these layers in order:\n\n1. Usage evidence, including recency and source quality.\n2. Installed metadata, instructions, and functional overlap.\n3. User-provided skill-on versus skill-off results for general skills.\n4. Runtime and bundle burden, including over-triggering, context cost, weak progressive disclosure, redundant resources, script failures, and private-looking files.\n5. Static health and risk hints.\n6. Optional offline community or registry metrics.\n\nTreat API and tool skills as protected capability skills during ablation.\nExamples: Excel, DOCX, PDF, browser automation, deployment, OCR, external API wrappers, MCP/API gateway helpers.\n\n## Workflow\n\n1. Collect user-provided roots before host-local defaults.\n2. Load only the usage, history, ablation, and community evidence that is available.\n3. Inspect each `SKILL.md` and its script/reference/asset metrics.\n4. Let the bundled script classify each skill as `api`, `tool`, or `general` and calculate its score. Read `{baseDir}/references/scoring-rubric.md` only when checking or explaining a score, verdict, or action.\n5. Print the short usefulness report and, when requested, write Markdown evidence or an ablation plan.\n\n## Ablation Rules\n\nRead `{baseDir}/references/ablation-protocol.md` only when running replays, preparing normalized ablation records, or reviewing mixed or delete-boundary results. The script can generate an ablation plan without loading the protocol.\nReplay only selected `general` candidates with identical prompts/artifacts and pairwise judging.\nDo not fake no-tool ablation for `api` or `tool` skills; use the rubric's protected-capability branch.\n\n## Run the Audit\n\nRun the audit after collecting available evidence:\n\n```bash\nREPORT_LANGUAGE=en  # use zh-CN when the current user invocation is Chinese\npython \"{baseDir}/scripts/skill_usefulness_audit.py\" audit \\\n  --skills-root ./skills \\\n  --report-language \"$REPORT_LANGUAGE\" \\\n  --markdown-out ./skill-audit-report.md\n```\n\nOpenClaw expands `{baseDir}` to the installed skill directory. Use it for bundled scripts and references.\n\nAdd evidence only when available:\n\n- `--usage-file`: JSON, JSONL, CSV, or TSV with per-skill usage.\n- `--history-file`: raw transcripts used only when direct usage is weak or missing; mentions remain `history_mentions` / `suspected_invocations`, not `calls`.\n- `--ablation-file`: normalized JSON or JSONL skill-on/skill-off results.\n- `--community-file`: offline JSON, JSONL, CSV, or TSV registry metrics.\n- `--ablation-plan-out`: a cost estimate and focused replay plan; its case counts can be overridden with the four `--ablation-*-cases` options documented by `--help`.\n- `--json-out`: machine-readable evidence only when requested or needed by another tool.\n\nPass `--report-language zh-CN` for a Chinese invocation and `--report-language en` for an English invocation. `auto` reads `SKILL_AUDIT_REPORT_LANGUAGE` or the process locale, then falls back to English.\n\nRun without extra files only when you need a structure-only audit.\nUsage, community, and ablation evidence become lower-confidence in that mode.\nHistory and usage files may contain sensitive conversations, local paths, project names, and customer data.\nMissing env means not configured in the current audit process, not proof that the skill is broken in every host.\n\n## Output Contract\n\nUse one run for both output layers; do not ask the user to choose a quick or full mode.\nStandard output is a short natural-language report. Its opening paragraph states the audited skill count and the total characters plus approximate tokens of loaded entry descriptions. Lead with actual usage, not static risk or bundle health, and keep scores, internal codes, risk flags, and tables out of this layer.\nWhen `--markdown-out` is provided, write the detailed evidence—with scores, action codes, missing evidence, burden, and risk notes—in the same run.\nMatch the user's language: clean Chinese for `zh-CN` and clean English for `en`, except for skill names and unavoidable paths or commands.\n\nCopy the short report to chat verbatim, apart from making its evidence path clickable. Do not paste raw JSON or the full Markdown evidence unless the user asks. Read `{baseDir}/references/report-narration-prompt.md` only when another agent or host must deliver an already-generated report.\n\nJSON includes `report_mode`, per-skill `score_breakdown`, `quality_penalty`, `quality_penalty_uncapped`, `quality_evidence`, `community_breakdown`, `action_advice`, and `risk_review`. It includes `ablation_plan` only when `--ablation-plan-out` is used.\n\nKeep deletion advice conservative for system or host-core skills, and prefer narrowing or merging when overlapping skills still serve distinct host integrations.\n\n## Resources\n\n- `{baseDir}/scripts/skill_usefulness_audit.py`: compatibility wrapper for the modular audit package.\n- `{baseDir}/scripts/skill_usefulness_audit_lib/`: collect metadata, score skills, scan static risk hints, and render Markdown reports plus optional JSON artifacts.\n- `{baseDir}/references/report-narration-prompt.md`: concise prompt for turning the report into a user-facing conversational summary.\n- `{baseDir}/references/scoring-rubric.md`: 10-point scoring rules, confidence logic, community prior, and action thresholds.\n- `{baseDir}/references/ablation-protocol.md`: normalized replay method for historical conversations.\n\nFile v0.3.20:_meta.json\n\n{\n  \"ownerId\": \"kn7em0w89d0zac35fzt84qm2a182j54b\",\n  \"slug\": \"skill-usefulness-audit\",\n  \"version\": \"0.3.20\",\n  \"publishedAt\": 1785484303804\n}\n\nFile v0.3.20:references/ablation-protocol.md\n\n# Ablation Protocol\n\nUse this protocol for `general` skills selected by the ablation plan.\n\n## Goal\n\nMeasure whether the skill changes outcomes in a meaningful way.\nHigh consistency between skill-on and skill-off runs means the skill adds little value.\n\n## Sampling\n\nStart with `3` historical tasks where the skill should plausibly matter. Prefer real user turns over synthetic prompts. Expand to `5` when results are mixed and to `10` only for high-impact or delete-boundary decisions.\n\n## Replay Method\n\nFor each selected case, run two isolated replays:\n\n1. `with_skill`\n2. `without_skill`\n\nKeep these constant:\n\n- same prompt\n- same files and artifacts\n- same model class when possible\n- same tool permissions\n- same success criteria\n\nUse a fresh thread or isolated run if the host supports it.\n\n## Judge Method\n\nFor open-ended outputs:\n\n1. Compare `with_skill` and `without_skill` side by side.\n2. Randomize A/B order.\n3. Spot-check reversed order on boundary cases.\n4. Prefer `pass/fail`, `same/better/worse`, and short reasons over long open-ended grading.\n\nRecord a standard `verdict` and one short `notes` reason. Each arm may also include optional `pass` and/or `score` from `0.0-1.0` for fallback inference. Optional `tool_cost` may describe calls, latency, or retries and currently does not affect the audit score.\n\n### Normalized JSON\n\n```json\n[\n  {\n    \"skill\": \"emotion-orchestrator\",\n    \"case_id\": \"case-001\",\n    \"with_skill\": {\"pass\": true, \"score\": 0.92},\n    \"without_skill\": {\"pass\": true, \"score\": 0.81},\n    \"verdict\": \"better\"\n  }\n]\n```\n\n## Judgment Rule\n\nUse `same` when the final answer, correctness, and workflow remain materially equivalent.\nUse `better` when the skill improves correctness, speed, structure, or user-fit in a way the baseline did not.\nUse `worse` when the skill adds friction, drift, or errors.\n\nIgnore verdict-only cases with unsupported values. A case with an unknown or missing verdict is usable only when both arms provide comparable `pass` and/or `score` fields for inference.\n\n## Early Stop Rules\n\n- Stop as low-value when `3/3` cases are `same` and `better_rate` is `0`.\n- Stop as useful when at least `2/3` cases are `better` and no case is `worse`.\n- Expand to `5` when the first batch is mixed.\n- Expand to `10` only for delete-boundary or high-impact decisions.\n\nDelete-boundary means \"needs stronger human review evidence\", not automatic deletion authority.\n\n## Planning and Model Cost\n\nCreate the replay plan with `--ablation-plan-out`. Planning uses local evidence and does not call an LLM.\n\nThe plan estimates replay cost for `light`, `realistic`, and `coding` profiles. Each case assumes two replays and one compact pairwise judge. `model_cost_estimates.unit` records `estimated_context_units_per_case`.\n\nFeed normalized results back with `--ablation-file`.\n\nFile v0.3.20:references/report-narration-prompt.md\n\n# Report Delivery Contract\n\n- Treat standard output as the final short report. Copy it verbatim into chat, apart from making the evidence path clickable when supported.\n- Do not add headings, bullets, extra counts, verification notes, or facts from the Markdown evidence unless requested.\n- Keep the short report on actual use, missing use evidence, overlap, and verified outcome impact. Leave scores, internal codes, risk, bundle health, and tables in the Markdown evidence.\n- Match the user's language: clean Chinese for `zh-CN` and clean English for `en`, except for skill names and unavoidable paths or commands.\n- Do not paste raw JSON or read back the full Markdown report unless requested.\n- Treat removal results as manual-review recommendations. Never remove, merge, isolate, or disable a skill automatically.\n\nFile v0.3.20:references/scoring-rubric.md\n\n# Scoring Rubric\n\n## Contents\n\n- Core Outputs\n- Usage Score\n- Uniqueness Score\n- Impact Score\n- Confidence Score\n- Quality Penalty\n- Community Prior Score\n- Static Risk Level\n- Verdict Bands\n- Action Rules\n\n## Core Outputs\n\n- `local_score = usage_score + uniqueness_score + impact_score`\n- `quality_penalty`: `0.0-2.5`\n- `quality_penalty_uncapped`: raw quality burden before the cap\n- `static_quality_penalty`: `0.0-1.4`\n- `final_score = clamp(local_score - quality_penalty, 0.0, 10.0)`\n- `risk_level` / `static_risk_level`: `none / low / medium / high`\n\nKeep community prior and risk separate from `local_score`; use them with quality burden to shape review and action.\n\n## 1. Usage Score (`0.0-3.0`)\n\nPrefer direct host usage logs.\nUse transcript mentions only as weaker fallback evidence.\n\n### Input Fields\n\n- Direct usage: `calls`, `recent_30d_calls`, `recent_90d_calls`, `last_used_at`, and `active_days`.\n- History fallback: `history_mentions` and `suspected_invocations`. These are weak evidence weighted through history and must not be reported as direct `calls`.\n- Evidence and runtime burden: `usage_source`, `evidence_weight`, `executions`, `script_failures`, `repair_turns`, `reference_loads`, and `false_triggers`.\n\n### Base Usage Strength\n\n- When `recent_30d_calls` exists:\n  - `0.0`: `0`\n  - `1.0`: `1-2`\n  - `2.0`: `3-7`\n  - `3.0`: `8+`\n- When only `recent_90d_calls` exists:\n  - `0.0`: `0`\n  - `0.75`: `1-2`\n  - `1.5`: `3-9`\n  - `2.5`: `10+`\n- When only total `calls` exists:\n  - `0.0`: `0`\n  - `1.0`: `1-2`\n  - `2.0`: `3-9`\n  - `3.0`: `10+`\n\n### Recency Adjustments\n\n- add `0.5` when `last_used_at <= 7 days`\n- add `0.25` when `last_used_at <= 30 days`\n- subtract `0.5` when `last_used_at > 180 days`\n- add `0.25` when `active_days >= 10`\n- add `0.10` when `active_days >= 3`\n\n### Evidence Weight\n\n- `1.00`: direct usage file\n- `0.45`: transcript-history fallback based on `suspected_invocations`\n- `0.00`: missing usage evidence\n\nClamp the final usage score to `0.0-3.0`.\n\n## 2. Uniqueness Score (`0.0-3.0`)\n\nMeasure the highest functional-overlap similarity against any other installed skill using descriptions, headings, and resource names.\n\nBuckets:\n\n- `0.0`: highest overlap `>= 0.85`\n- `1.0`: highest overlap `0.65-0.84`\n- `2.0`: highest overlap `0.40-0.64`\n- `3.0`: highest overlap `< 0.40`\n\n## 3. Impact Score (`0.0-4.0`)\n\n### General skills\n\nUse ablation on historical conversations.\nCompute:\n\n- `consistency_rate`: skill-on and skill-off produce materially equivalent outcomes\n- `better_rate`: skill-on clearly improves the result\n- `worse_rate`: skill-on clearly harms the result\n\nBase score from consistency:\n\n- `0.0`: `consistency_rate >= 0.85`\n- `1.0`: `0.70-0.84`\n- `2.0`: `0.55-0.69`\n- `3.0`: `0.35-0.54`\n- `4.0`: `< 0.35`\n\nAdjustments:\n\n- add `1.0` when `better_rate - worse_rate >= 0.30`\n- subtract `1.0` when `worse_rate > better_rate`\n- clamp the final impact score to `0.0-4.0`\n\nWhen ablation is missing, use low-evidence score `1.0` for zero-call skills.\nFor skills with direct usage evidence but no ablation yet, keep temporary neutral score `2.0` and lower confidence.\n\n### API and tool skills\n\nSkip history ablation.\nUse protected-capability scoring instead:\n\n- start at `2.0`\n- add `1.0` when the skill ships executable scripts or hard capability resources\n- add `0.5` when highest overlap `< 0.35`\n- add `0.5` when calls `>= 3`\n- subtract `1.0` when highest overlap `>= 0.75`\n- subtract `0.5` when calls are `0`\n- clamp the final impact score to `0.0-4.0`\n\n## 4. Confidence Score (`0.0-1.0`)\n\nConfidence describes evidence quality, not usefulness.\n\nAdd:\n\n- `0.35` for direct usage files\n- `0.15` for history fallback\n- `0.20` when recent usage fields exist\n- `0.10` when only total direct calls exist\n- `0.25` for protected `api/tool` classification\n- `0.25` for `general` skills with `>= 5` ablation cases\n- `0.15` for `general` skills with `1-4` ablation cases\n- `0.10` when overlap comparison has peers\n- `0.05` when only one skill exists in scope\n- `0.10` when community metadata exists\n\nClamp the final confidence score to `0.0-1.0`.\n\n## 5. Quality Penalty (`0.0-2.5`)\n\nQuality penalty captures the cost of keeping a skill and is deducted from `local_score`; it is not a risk flag.\n\n### Runtime burden\n\nUse direct usage logs when available:\n\n- `overtrigger-low-execution`: `0.45` when `calls >= 8` and `executions / calls < 0.25`\n- `overtrigger-misfire`: `0.35` when `calls >= 5` and (`false_triggers >= 3` or `false_triggers / calls >= 0.25`)\n- `overtrigger-no-impact`: `0.40` when `calls >= 5`, `consistency_rate >= 0.85`, and `better_rate <= 0.10`\n- `reference-overload`: `0.30` when `reference_loads >= 10` and `reference_loads / calls >= 3.0`\n- `script-failure-burden`: `0.45` when script failures are frequent\n- `script-failure-burden`: `0.20` when script failures are occasional\n- `agent-repair-burden`: `0.30` when `repair_turns >= 3`\n\n### Readiness burden\n\n- `missing-required-env`: `0.90` when declared required environment variables are not configured in the current audit process\n\n### Catalog burden\n\n- `near-duplicate-instructions`: `0.10` when the instruction fingerprint closely matches another installed skill\n\n### Static bundle burden\n\nScan installed skill files:\n\n- `empty-skill-contract`: `0.80` when the skill has no meaningful runtime contract beyond minimal or missing metadata\n- `prompt-bloat`: `0.40` when `SKILL.md` body is at least `5000` context units\n- `prompt-bloat`: `0.20` when `SKILL.md` body is at least `2500` context units\n- `broad-trigger-surface`: `0.25` when the frontmatter description uses broad trigger language\n- `description-bloat`: `0.25` when the frontmatter description is at least `120` context units\n- `description-bloat`: `0.10` when the frontmatter description is at least `60` context units\n- `reference-disclosure-gap`: `0.30` when at least 3 reference files exist and none are directly discoverable from `SKILL.md`\n- `reference-disclosure-gap`: `0.10` when 1-2 reference files exist and none are directly discoverable from `SKILL.md`\n- `reference-disclosure-gap`: `0.20` when at least 8 reference files exist and fewer than 30% are directly linked from `SKILL.md`\n- `reference-link-broken`: `0.25` when `SKILL.md` points to missing reference files\n- `reference-bloat`: `0.50` when references are at least 50 files or 50000 context units\n- `reference-bloat`: `0.25` when references are at least 20 files or 15000 context units\n- `long-reference-without-toc`: `0.20` when at least 3 long reference files lack a visible table of contents\n- `long-reference-without-toc`: `0.10` when 1-2 long reference files lack a visible table of contents\n- `reference-content-pollution`: `0.35` when references include advertising, upsells, unrelated text, or other-tool/skill promotion\n- `asset-bloat`: `0.50` when assets are at least 200 files or 100 MB\n- `asset-bloat`: `0.25` when assets are at least 50 files or 25 MB\n- `vague-resource-names`: `0.20` when at least 5 scripts, references, or assets use generic filenames\n- `private-bundle-artifact`: `0.60` when bundled filenames look private or environment-specific\n- `private-content-artifact`: `0.60` when bundled content looks like credentials or keys\n- `executable-asset`: `0.30` when assets contain executable binaries or installers\n- `script-count-bloat`: `0.20` when the bundle has at least 40 scripts\n- `script-count-bloat`: `0.10` when the bundle has at least 20 scripts\n- `script-maintenance-smell`: `0.40` when at least 8 scripts contain placeholders, local absolute paths, or maintenance smells\n- `script-maintenance-smell`: `0.25` when 1-7 scripts contain placeholders, local absolute paths, or maintenance smells\n- `script-syntax-error`: `0.50` when Python scripts contain syntax errors\n- `script-import-error`: `0.50` when Python scripts import modules missing from the local environment or bundle\n\nClamp `static_quality_penalty` to `0.0-1.4`, then clamp the combined `quality_penalty` to `0.0-2.5`.\nEmit `quality_flags`, `quality_evidence`, `resource_metrics`, `quality_penalty_uncapped`, and `score_breakdown.quality`.\n\n## 6. Community Prior Score (`0.0-1.0`)\n\nTreat community data as external prior, not a local verdict.\n\nWeighted components:\n\n- `0.30`: normalized rating\n- `0.20`: current installs or downloads\n- `0.10`: all-time installs\n- `0.15`: trending metric\n- `0.10`: stars\n- `0.05`: comments\n- `0.10`: maintenance freshness from `last_updated`\n\nUse it to rank review priority and benchmark replacements. Emit `community_breakdown` in JSON so users can see which registry signals contributed.\n\n## 7. Static Risk Level\n\nRun static scans against runnable scripts and resource files.\nOnly fenced code blocks in `SKILL.md` and directly linked Markdown references are scanned as commands; prose outside fences and unlinked references are not command-scanned.\nCredential-like content checks still cover `SKILL.md`, scripts, assets, references, and root text files without echoing matched values.\nThis is lint-style evidence only. It cannot prove a skill is safe, because indirection, dynamic imports, encoded payloads, aliases, or external downloads can evade simple pattern matching.\n\nTypical flags: `curl-pipe-shell`, `dynamic-exec`, `protected-path-access`, `persistence-hook`, `external-post`, `shell-exec`, `network-download`, and `base64-payload`.\n\nStatic risk levels:\n\n- `none`: `0.0`\n- `low`: `0.0 < score < 2.0`\n- `medium`: `2.0-3.9`\n- `high`: `4.0+`\n\nIf static quality finds `private-content-artifact`, that evidence is promoted to `high` risk so credential-like bundled content receives `quarantine-review`.\n\n## Health Cap\n\nSome quality findings cap the final score even when usage or protected-capability signals are strong:\n\n- `script-syntax-error`: final score cap `4.0`\n- `empty-skill-contract`: final score cap `5.5`\n- `script-import-error`: final score cap `5.5`\n- `script-failure-burden`: final score cap `4.0` when the penalty is at least `0.45`\n\n## Verdict Bands\n\nUse `final_score` for verdict bands.\n\n- confidence `< 0.55` and `final_score < 4.5`: `insufficient-evidence`\n- `8.0-10.0`: `keep`\n- `6.0-7.9`: `keep-narrow`\n- `4.5-5.9`: `review`\n- `3.0-4.4`: `merge-delete`\n- `0.0-2.9`: `delete`\n\n## Action Rules\n\nEvaluate top to bottom; the first matching rule wins.\n\n| Condition | Action |\n| --- | --- |\n| system source, high risk / otherwise | `review-system` / `keep-system` |\n| high risk | `quarantine-review` |\n| medium risk and score `>= 6.0` | `keep-review-risk` |\n| quality penalty `>= 1.2` and score `>= 6.0` / `>= 4.5` | `keep-review-burden` / `review-burden` |\n| score `>= 8.0` / `>= 6.0` | `keep` / `keep-narrow` |\n| remaining medium risk | `review-risk` |\n| confidence `< 0.55` | `observe-30d` |\n| score `>= 4.5`: overlap `>= 0.65` / community prior `>= 0.6` / otherwise | `merge-or-review` / `review-vs-community` / `review` |\n| API/tool skill: zero calls and overlap `>= 0.75` / community prior `>= 0.6` / otherwise | `merge-delete` / `review-vs-community` / `merge-or-review` |\n| community prior `>= 0.6` | `review-vs-community` |\n| score `< 3.0` | `delete` |\n| otherwise, including overlap `>= 0.65` with calls `<= 1` | `merge-delete` |\n\n`delete`, `merge-delete`, and `quarantine-review` are report recommendations only. They are never permission for automatic deletion, isolation, or disabling without human review.\n\nFile v0.3.20:skill-card.md\n\n## Description: <br>\nReview your installed agent skills to see what you actually use, what overlaps, and what may no longer be worth keeping. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[gongyu0918-debug](https://clawhub.ai/user/gongyu0918-debug) <br>\n\n### License/Terms of Use: <br>\nMIT-0 <br>\n\n\n## Use Case: <br>\nDevelopers and agent users use this skill to audit installed agent skills, compare usage and overlap, identify low-confidence cleanup candidates, and prepare evidence-backed ablation plans without automatically removing skills. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: Installed skill folders and user-provided usage or history files may contain sensitive conversations, local paths, project names, or customer data. <br>\nMitigation: Provide only intended evidence files, keep generated reports local unless reviewed, and avoid sharing raw JSON or full Markdown evidence unless needed. <br>\nRisk: Delete, merge-delete, quarantine, or cleanup labels could be mistaken for automatic removal instructions. <br>\nMitigation: Treat cleanup labels as manual-review recommendations and confirm evidence before changing or removing any installed skill. <br>\nRisk: Structure-only audits have lower-confidence usefulness and cleanup recommendations. <br>\nMitigation: Prefer direct usage logs, history fallback evidence, ablation results, or community metrics before making cleanup decisions. <br>\n\n\n## Reference(s): <br>\n- [ClawHub skill page](https://clawhub.ai/gongyu0918-debug/skills/skill-usefulness-audit) <br>\n- [Project homepage](https://github.com/gongyu0918-debug/skill-usefulness-audit) <br>\n- [Ablation protocol](references/ablation-protocol.md) <br>\n- [Report narration prompt](references/report-narration-prompt.md) <br>\n- [Scoring rubric](references/scoring-rubric.md) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [text, markdown, JSON, shell commands, guidance] <br>\n**Output Format:** [Natural-language report with optional Markdown report, JSON evidence, and ablation plan files] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [May include manual-review cleanup recommendations, score evidence, risk notes, and ablation planning details.] <br>\n\n## Skill Version(s): <br>\n0.3.20 (source: frontmatter and server release metadata) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nFile v0.3.20:LICENSE\n\nMIT No Attribution\n\nCopyright 2026 gongyu0918-debug\n\nPermission is hereby granted, free of charge, to any person obtaining a copy of this software and associated documentation files (the \"Software\"), to deal in the Software without restriction, including without limitation the rights to use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the Software, and to permit persons to whom the Software is furnished to do so.\n\nTHE SOFTWARE IS PROVIDED \"AS IS\", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.\n\nArchive v0.3.19: 23 files, 70583 bytes\n\nFiles: LICENSE (911b), references/ablation-protocol.md (2820b), references/report-narration-prompt.md (819b), references/scoring-rubric.md (11293b), scripts/skill_usefulness_audit_lib/__init__.py (219b), scripts/skill_usefulness_audit_lib/ablation.py (5078b), scripts/skill_usefulness_audit_lib/cli.py (37159b), scripts/skill_usefulness_audit_lib/common.py (24079b), scripts/skill_usefulness_audit_lib/community.py (7317b), scripts/skill_usefulness_audit_lib/constants.py (11293b), scripts/skill_usefulness_audit_lib/reporting.py (58738b), scripts/skill_usefulness_audit_lib/risk_quality.py (46000b), scripts/skill_usefulness_audit_lib/risk_signatures_encoding.py (271b), scripts/skill_usefulness_audit_lib/risk_signatures_execution.py (687b), scripts/skill_usefulness_audit_lib/risk_signatures_network.py (581b), scripts/skill_usefulness_audit_lib/risk_signatures_sensitive.py (326b), scripts/skill_usefulness_audit_lib/risk_signatures.py (394b), scripts/skill_usefulness_audit_lib/scoring.py (18994b), scripts/skill_usefulness_audit_lib/usage_loader.py (11893b), scripts/skill_usefulness_audit.py (793b), skill-card.md (2361b), SKILL.md (6487b), _meta.json (142b)\n\nFile v0.3.19:SKILL.md\n\n---\nname: skill-usefulness-audit\nslug: skill-usefulness-audit\ndescription: Review your installed agent skills to see what you actually use, what overlaps, and what may no longer be worth keeping.\nversion: 0.3.19\ntags: [\"audit\",\"skills\",\"ablation\",\"openclaw\"]\nuser-invocable: true\ndisable-model-invocation: true\nargument-hint: --skills-root PATH --usage-file FILE\nhomepage: https://github.com/gongyu0918-debug/skill-usefulness-audit\nmetadata: {\"openclaw\":{\"skillKey\":\"skill-usefulness-audit\",\"requires\":{\"bins\":[\"python\"]},\"homepage\":\"https://github.com/gongyu0918-debug/skill-usefulness-audit\"}}\n---\n# Skill Usefulness Audit\n\n## Manual Trigger Only\n\nUse this skill only after a direct request to audit installed agent skills, their usage, overlap, cleanup options, or a structure-only inventory.\nDo not invoke it during normal tasks or use it for ordinary repository/source-code review, general security audit, or employee/human skill assessment.\n\n## Safety\n\nNever delete, merge, quarantine, isolate, or disable skills automatically.\nTreat `delete`, `merge-delete`, and `quarantine-review` as manual-review recommendations.\nDo not delete skills based only on a structure-only report.\nThis tool does not automatically replay historical conversations; it generates ablation plans and reads ablation result files that the user provides.\n\n## Audit Scope\n\nAudit these layers in order:\n\n1. Usage evidence, including recency and source quality.\n2. Installed metadata, instructions, and functional overlap.\n3. User-provided skill-on versus skill-off results for general skills.\n4. Runtime and bundle burden, including over-triggering, context cost, weak progressive disclosure, redundant resources, script failures, and private-looking files.\n5. Static health and risk hints.\n6. Optional offline community or registry metrics.\n\nTreat API and tool skills as protected capability skills during ablation.\nExamples: Excel, DOCX, PDF, browser automation, deployment, OCR, external API wrappers, MCP/API gateway helpers.\n\n## Workflow\n\n1. Collect user-provided roots before host-local defaults.\n2. Load only the usage, history, ablation, and community evidence that is available.\n3. Inspect each `SKILL.md` and its script/reference/asset metrics.\n4. Classify each skill as `api`, `tool`, or `general`, then score it with `{baseDir}/references/scoring-rubric.md`.\n5. Print the short usefulness report and, when requested, write Markdown evidence or an ablation plan.\n\n## Ablation Rules\n\nRead `{baseDir}/references/ablation-protocol.md` only when generating an ablation plan or evaluating ablation results.\nReplay only selected `general` candidates with identical prompts/artifacts and pairwise judging.\nDo not fake no-tool ablation for `api` or `tool` skills; use the rubric's protected-capability branch.\n\n## Run the Audit\n\nRun the audit after collecting available evidence:\n\n```bash\nREPORT_LANGUAGE=en  # use zh-CN when the current user invocation is Chinese\npython \"{baseDir}/scripts/skill_usefulness_audit.py\" audit \\\n  --skills-root ./skills \\\n  --report-language \"$REPORT_LANGUAGE\" \\\n  --markdown-out ./skill-audit-report.md\n```\n\nOpenClaw expands `{baseDir}` to the installed skill directory. Use it for bundled scripts and references.\n\nAdd evidence only when available:\n\n- `--usage-file`: JSON, JSONL, CSV, or TSV with per-skill usage.\n- `--history-file`: raw transcripts used only when direct usage is weak or missing; mentions remain `history_mentions` / `suspected_invocations`, not `calls`.\n- `--ablation-file`: normalized JSON or JSONL skill-on/skill-off results.\n- `--community-file`: offline JSON, JSONL, CSV, or TSV registry metrics.\n- `--ablation-plan-out`: a cost estimate and focused replay plan; its case counts can be overridden with the four `--ablation-*-cases` options documented by `--help`.\n- `--json-out`: machine-readable evidence only when requested or needed by another tool.\n\nPass `--report-language zh-CN` for a Chinese invocation and `--report-language en` for an English invocation. `auto` reads `SKILL_AUDIT_REPORT_LANGUAGE` or the process locale, then falls back to English.\n\nRun without extra files only when you need a structure-only audit.\nUsage, community, and ablation evidence become lower-confidence in that mode.\nHistory and usage files may contain sensitive conversations, local paths, project names, and customer data.\nMissing env means not configured in the current audit process, not proof that the skill is broken in every host.\n\n## Output Contract\n\nUse one run for both output layers; do not ask the user to choose a quick or full mode.\nStandard output is a short natural-language report. Its opening paragraph states the audited skill count and the total characters plus approximate tokens of loaded entry descriptions. Lead with actual usage, not static risk or bundle health, and keep scores, internal codes, risk flags, and tables out of this layer.\nWhen `--markdown-out` is provided, write the detailed evidence—with scores, action codes, missing evidence, burden, and risk notes—in the same run.\nMatch the user's language: clean Chinese for `zh-CN` and clean English for `en`, except for skill names and unavoidable paths or commands.\n\nBefore delivering the result, read `{baseDir}/references/report-narration-prompt.md`. Copy the short report to chat verbatim, apart from making its evidence path clickable. Do not paste raw JSON or the full Markdown evidence unless the user asks.\n\nJSON includes `report_mode`, per-skill `score_breakdown`, `quality_penalty`, `quality_penalty_uncapped`, `quality_evidence`, `community_breakdown`, `action_advice`, and `risk_review`. It includes `ablation_plan` only when `--ablation-plan-out` is used.\n\nKeep deletion advice conservative for system or host-core skills, and prefer narrowing or merging when overlapping skills still serve distinct host integrations.\n\n## Resources\n\n- `{baseDir}/scripts/skill_usefulness_audit.py`: compatibility wrapper for the modular audit package.\n- `{baseDir}/scripts/skill_usefulness_audit_lib/`: collect metadata, score skills, scan static risk hints, and render Markdown reports plus optional JSON artifacts.\n- `{baseDir}/references/report-narration-prompt.md`: concise prompt for turning the report into a user-facing conversational summary.\n- `{baseDir}/references/scoring-rubric.md`: 10-point scoring rules, confidence logic, community prior, and action thresholds.\n- `{baseDir}/references/ablation-protocol.md`: normalized replay method for historical conversations.\n\nFile v0.3.19:_meta.json\n\n{\n  \"ownerId\": \"kn7em0w89d0zac35fzt84qm2a182j54b\",\n  \"slug\": \"skill-usefulness-audit\",\n  \"version\": \"0.3.19\",\n  \"publishedAt\": 1785338795075\n}\n\nFile v0.3.19:references/ablation-protocol.md\n\n# Ablation Protocol\n\nUse this protocol for `general` skills selected by the ablation plan.\n\n## Goal\n\nMeasure whether the skill changes outcomes in a meaningful way.\nHigh consistency between skill-on and skill-off runs means the skill adds little value.\n\n## Sampling\n\nStart with `3` historical tasks where the skill should plausibly matter. Prefer real user turns over synthetic prompts. Expand to `5` when results are mixed and to `10` only for high-impact or delete-boundary decisions.\n\n## Replay Method\n\nFor each selected case, run two isolated replays:\n\n1. `with_skill`\n2. `without_skill`\n\nKeep these constant:\n\n- same prompt\n- same files and artifacts\n- same model class when possible\n- same tool permissions\n- same success criteria\n\nUse a fresh thread or isolated run if the host supports it.\n\n## Judge Method\n\nFor open-ended outputs:\n\n1. Compare `with_skill` and `without_skill` side by side.\n2. Randomize A/B order.\n3. Spot-check reversed order on boundary cases.\n4. Prefer `pass/fail`, `same/better/worse`, and short reasons over long open-ended grading.\n\nRecord a standard `verdict` and one short `notes` reason. Each arm may also include optional `pass` and/or `score` from `0.0-1.0` for fallback inference. Optional `tool_cost` may describe calls, latency, or retries and currently does not affect the audit score.\n\n### Normalized JSON\n\n```json\n[\n  {\n    \"skill\": \"emotion-orchestrator\",\n    \"case_id\": \"case-001\",\n    \"with_skill\": {\"pass\": true, \"score\": 0.92},\n    \"without_skill\": {\"pass\": true, \"score\": 0.81},\n    \"verdict\": \"better\"\n  }\n]\n```\n\n## Judgment Rule\n\nUse `same` when the final answer, correctness, and workflow remain materially equivalent.\nUse `better` when the skill improves correctness, speed, structure, or user-fit in a way the baseline did not.\nUse `worse` when the skill adds friction, drift, or errors.\n\nIgnore verdict-only cases with unsupported values. A case with an unknown or missing verdict is usable only when both arms provide comparable `pass` and/or `score` fields for inference.\n\n## Early Stop Rules\n\n- Stop as low-value when `3/3` cases are `same` and `better_rate` is `0`.\n- Stop as useful when at least `2/3` cases are `better` and no case is `worse`.\n- Expand to `5` when the first batch is mixed.\n- Expand to `10` only for delete-boundary or high-impact decisions.\n\nDelete-boundary means \"needs stronger human review evidence\", not automatic deletion authority.\n\n## Planning and Model Cost\n\nCreate the replay plan with `--ablation-plan-out`. Planning uses local evidence and does not call an LLM.\n\nThe plan estimates replay cost for `light`, `realistic`, and `coding` profiles. Each case assumes two replays and one compact pairwise judge. `model_cost_estimates.unit` records `estimated_context_units_per_case`.\n\nFeed normalized results back with `--ablation-file`.\n\nFile v0.3.19:references/report-narration-prompt.md\n\n# Report Delivery Contract\n\n- Treat standard output as the final short report. Copy it verbatim into chat, apart from making the evidence path clickable when supported.\n- Do not add headings, bullets, extra counts, verification notes, or facts from the Markdown evidence unless requested.\n- Keep the short report on actual use, missing use evidence, overlap, and verified outcome impact. Leave scores, internal codes, risk, bundle health, and tables in the Markdown evidence.\n- Match the user's language: clean Chinese for `zh-CN` and clean English for `en`, except for skill names and unavoidable paths or commands.\n- Do not paste raw JSON or read back the full Markdown report unless requested.\n- Treat removal results as manual-review recommendations. Never remove, merge, isolate, or disable a skill automatically.\n\nFile v0.3.19:references/scoring-rubric.md\n\n# Scoring Rubric\n\n## Contents\n\n- Core Outputs\n- Usage Score\n- Uniqueness Score\n- Impact Score\n- Confidence Score\n- Quality Penalty\n- Community Prior Score\n- Static Risk Level\n- Verdict Bands\n- Action Rules\n\n## Core Outputs\n\n- `local_score = usage_score + uniqueness_score + impact_score`\n- `quality_penalty`: `0.0-2.5`\n- `quality_penalty_uncapped`: raw quality burden before the cap\n- `static_quality_penalty`: `0.0-1.4`\n- `final_score = clamp(local_score - quality_penalty, 0.0, 10.0)`\n- `risk_level` / `static_risk_level`: `none / low / medium / high`\n\nKeep community prior and risk separate from `local_score`; use them with quality burden to shape review and action.\n\n## 1. Usage Score (`0.0-3.0`)\n\nPrefer direct host usage logs.\nUse transcript mentions only as weaker fallback evidence.\n\n### Input Fields\n\n- Direct usage: `calls`, `recent_30d_calls`, `recent_90d_calls`, `last_used_at`, and `active_days`.\n- History fallback: `history_mentions` and `suspected_invocations`. These are weak evidence weighted through history and must not be reported as direct `calls`.\n- Evidence and runtime burden: `usage_source`, `evidence_weight`, `executions`, `script_failures`, `repair_turns`, `reference_loads`, and `false_triggers`.\n\n### Base Usage Strength\n\n- When `recent_30d_calls` exists:\n  - `0.0`: `0`\n  - `1.0`: `1-2`\n  - `2.0`: `3-7`\n  - `3.0`: `8+`\n- When only `recent_90d_calls` exists:\n  - `0.0`: `0`\n  - `0.75`: `1-2`\n  - `1.5`: `3-9`\n  - `2.5`: `10+`\n- When only total `calls` exists:\n  - `0.0`: `0`\n  - `1.0`: `1-2`\n  - `2.0`: `3-9`\n  - `3.0`: `10+`\n\n### Recency Adjustments\n\n- add `0.5` when `last_used_at <= 7 days`\n- add `0.25` when `last_used_at <= 30 days`\n- subtract `0.5` when `last_used_at > 180 days`\n- add `0.25` when `active_days >= 10`\n- add `0.10` when `active_days >= 3`\n\n### Evidence Weight\n\n- `1.00`: direct usage file\n- `0.45`: transcript-history fallback based on `suspected_invocations`\n- `0.00`: missing usage evidence\n\nClamp the final usage score to `0.0-3.0`.\n\n## 2. Uniqueness Score (`0.0-3.0`)\n\nMeasure the highest functional-overlap similarity against any other installed skill using descriptions, headings, and resource names.\n\nBuckets:\n\n- `0.0`: highest overlap `>= 0.85`\n- `1.0`: highest overlap `0.65-0.84`\n- `2.0`: highest overlap `0.40-0.64`\n- `3.0`: highest overlap `< 0.40`\n\n## 3. Impact Score (`0.0-4.0`)\n\n### General skills\n\nUse ablation on historical conversations.\nCompute:\n\n- `consistency_rate`: skill-on and skill-off produce materially equivalent outcomes\n- `better_rate`: skill-on clearly improves the result\n- `worse_rate`: skill-on clearly harms the result\n\nBase score from consistency:\n\n- `0.0`: `consistency_rate >= 0.85`\n- `1.0`: `0.70-0.84`\n- `2.0`: `0.55-0.69`\n- `3.0`: `0.35-0.54`\n- `4.0`: `< 0.35`\n\nAdjustments:\n\n- add `1.0` when `better_rate - worse_rate >= 0.30`\n- subtract `1.0` when `worse_rate > better_rate`\n- clamp the final impact score to `0.0-4.0`\n\nWhen ablation is missing, use low-evidence score `1.0` for zero-call skills.\nFor skills with direct usage evidence but no ablation yet, keep temporary neutral score `2.0` and lower confidence.\n\n### API and tool skills\n\nSkip history ablation.\nUse protected-capability scoring instead:\n\n- start at `2.0`\n- add `1.0` when the skill ships executable scripts or hard capability resources\n- add `0.5` when highest overlap `< 0.35`\n- add `0.5` when calls `>= 3`\n- subtract `1.0` when highest overlap `>= 0.75`\n- subtract `0.5` when calls are `0`\n- clamp the final impact score to `0.0-4.0`\n\n## 4. Confidence Score (`0.0-1.0`)\n\nConfidence describes evidence quality, not usefulness.\n\nAdd:\n\n- `0.35` for direct usage files\n- `0.15` for history fallback\n- `0.20` when recent usage fields exist\n- `0.10` when only total direct calls exist\n- `0.25` for protected `api/tool` classification\n- `0.25` for `general` skills with `>= 5` ablation cases\n- `0.15` for `general` skills with `1-4` ablation cases\n- `0.10` when overlap comparison has peers\n- `0.05` when only one skill exists in scope\n- `0.10` when community metadata exists\n\nClamp the final confidence score to `0.0-1.0`.\n\n## 5. Quality Penalty (`0.0-2.5`)\n\nQuality penalty captures the cost of keeping a skill and is deducted from `local_score`; it is not a risk flag.\n\n### Runtime burden\n\nUse direct usage logs when available:\n\n- `overtrigger-low-execution`: `0.45` when `calls >= 8` and `executions / calls < 0.25`\n- `overtrigger-misfire`: `0.35` when `false_triggers >= 3` or `false_triggers / calls >= 0.25`\n- `overtrigger-no-impact`: `0.40` when `calls >= 5`, `consistency_rate >= 0.85`, and `better_rate <= 0.10`\n- `reference-overload`: `0.30` when `reference_loads >= 10` and `reference_loads / calls >= 3.0`\n- `script-failure-burden`: `0.45` when script failures are frequent\n- `script-failure-burden`: `0.20` when script failures are occasional\n- `agent-repair-burden`: `0.30` when `repair_turns >= 3`\n\n### Readiness burden\n\n- `missing-required-env`: `0.90` when declared required environment variables are not configured in the current audit process\n\n### Catalog burden\n\n- `near-duplicate-instructions`: `0.10` when the instruction fingerprint closely matches another installed skill\n\n### Static bundle burden\n\nScan installed skill files:\n\n- `empty-skill-contract`: `0.80` when the skill has no meaningful runtime contract beyond minimal or missing metadata\n- `prompt-bloat`: `0.40` when `SKILL.md` body is at least `5000` context units\n- `prompt-bloat`: `0.20` when `SKILL.md` body is at least `2500` context units\n- `broad-trigger-surface`: `0.25` when the frontmatter description uses broad trigger language\n- `description-bloat`: `0.25` when the frontmatter description is at least `120` context units\n- `description-bloat`: `0.10` when the frontmatter description is at least `60` context units\n- `reference-disclosure-gap`: `0.30` when at least 3 reference files exist and none are directly discoverable from `SKILL.md`\n- `reference-disclosure-gap`: `0.10` when 1-2 reference files exist and none are directly discoverable from `SKILL.md`\n- `reference-disclosure-gap`: `0.20` when at least 8 reference files exist and fewer than 30% are directly linked from `SKILL.md`\n- `reference-link-broken`: `0.25` when `SKILL.md` points to missing reference files\n- `reference-bloat`: `0.50` when references are at least 50 files or 50000 context units\n- `reference-bloat`: `0.25` when references are at least 20 files or 15000 context units\n- `long-reference-without-toc`: `0.20` when at least 3 long reference files lack a visible table of contents\n- `long-reference-without-toc`: `0.10` when 1-2 long reference files lack a visible table of contents\n- `reference-content-pollution`: `0.35` when references include advertising, upsells, unrelated text, or other-tool/skill promotion\n- `asset-bloat`: `0.50` when assets are at least 200 files or 100 MB\n- `asset-bloat`: `0.25` when assets are at least 50 files or 25 MB\n- `vague-resource-names`: `0.20` when at least 5 scripts, references, or assets use generic filenames\n- `private-bundle-artifact`: `0.60` when bundled filenames look private or environment-specific\n- `private-content-artifact`: `0.60` when bundled content looks like credentials or keys\n- `executable-asset`: `0.30` when assets contain executable binaries or installers\n- `script-count-bloat`: `0.20` when the bundle has at least 40 scripts\n- `script-count-bloat`: `0.10` when the bundle has at least 20 scripts\n- `script-maintenance-smell`: `0.40` when at least 8 scripts contain placeholders, local absolute paths, or maintenance smells\n- `script-maintenance-smell`: `0.25` when 1-7 scripts contain placeholders, local absolute paths, or maintenance smells\n- `script-syntax-error`: `0.50` when Python scripts contain syntax errors\n- `script-import-error`: `0.50` when Python scripts import modules missing from the local environment or bundle\n\nClamp `static_quality_penalty` to `0.0-1.4`, then clamp the combined `quality_penalty` to `0.0-2.5`.\nEmit `quality_flags`, `quality_evidence`, `resource_metrics`, `quality_penalty_uncapped`, and `score_breakdown.quality`.\n\n## 6. Community Prior Score (`0.0-1.0`)\n\nTreat community data as external prior, not a local verdict.\n\nWeighted components:\n\n- `0.30`: normalized rating\n- `0.20`: current installs or downloads\n- `0.10`: all-time installs\n- `0.15`: trending metric\n- `0.10`: stars\n- `0.05`: comments\n- `0.10`: maintenance freshness from `last_updated`\n\nUse it to rank review priority and benchmark replacements. Emit `community_breakdown` in JSON so users can see which registry signals contributed.\n\n## 7. Static Risk Level\n\nRun static scans against runnable scripts and resource files.\nOnly fenced code blocks in `SKILL.md` and directly linked Markdown references are scanned as commands; prose outside fences and unlinked references are not command-scanned.\nCredential-like content checks still cover `SKILL.md`, scripts, assets, references, and root text files without echoing matched values.\nThis is lint-style evidence only. It cannot prove a skill is safe, because indirection, dynamic imports, encoded payloads, aliases, or external downloads can evade simple pattern matching.\n\nTypical flags: `curl-pipe-shell`, `dynamic-exec`, `protected-path-access`, `persistence-hook`, `external-post`, `shell-exec`, `network-download`, and `base64-payload`.\n\nStatic risk levels:\n\n- `none`: `0.0`\n- `low`: `0.0 < score < 2.0`\n- `medium`: `2.0-3.9`\n- `high`: `4.0+`\n\nIf static quality finds `private-content-artifact`, that evidence is promoted to `high` risk so credential-like bundled content receives `quarantine-review`.\n\n## Health Cap\n\nSome quality findings cap the final score even when usage or protected-capability signals are strong:\n\n- `script-syntax-error`: final score cap `4.0`\n- `empty-skill-contract`: final score cap `5.5`\n- `script-import-error`: final score cap `5.5`\n- `script-failure-burden`: final score cap `4.0` when the penalty is at least `0.45`\n\n## Verdict Bands\n\nUse `final_score` for verdict bands.\n\n- confidence `< 0.55` and `final_score < 4.5`: `insufficient-evidence`\n- `8.0-10.0`: `keep`\n- `6.0-7.9`: `keep-narrow`\n- `4.5-5.9`: `review`\n- `3.0-4.4`: `merge-delete`\n- `0.0-2.9`: `delete`\n\n## Action Rules\n\nEvaluate top to bottom; the first matching rule wins.\n\n| Condition | Action |\n| --- | --- |\n| system source, high risk / otherwise | `review-system` / `keep-system` |\n| high risk | `quarantine-review` |\n| medium risk and score `>= 6.0` | `keep-review-risk` |\n| quality penalty `>= 1.2` and score `>= 6.0` / `>= 4.5` | `keep-review-burden` / `review-burden` |\n| score `>= 8.0` / `>= 6.0` | `keep` / `keep-narrow` |\n| remaining medium risk | `review-risk` |\n| confidence `< 0.55` | `observe-30d` |\n| score `>= 4.5`: overlap `>= 0.65` / community prior `>= 0.6` / otherwise | `merge-or-review` / `review-vs-community` / `review` |\n| API/tool skill: zero calls and overlap `>= 0.75` / community prior `>= 0.6` / otherwise | `merge-delete` / `review-vs-community` / `merge-or-review` |\n| community prior `>= 0.6` | `review-vs-community` |\n| score `< 3.0` | `delete` |\n| otherwise, including overlap `>= 0.65` with calls `<= 1` | `merge-delete` |\n\n`delete`, `merge-delete`, and `quarantine-review` are report recommendations only. They are never permission for automatic deletion, isolation, or disabling without human review.\n\nFile v0.3.19:skill-card.md\n\n## Description: <br>\nReview your installed agent skills to see what you actually use, what overlaps, and what may no longer be worth keeping. <br>\n\nThis skill is ready for commercial/non-commercial use. <br>\n\n## Publisher: <br>\n[gongyu0918-debug](https://clawhub.ai/user/gongyu0918-debug) <br>\n\n### License/Terms of Use: <br>\nMIT No Attribution <br>\n\n\n## Use Case: <br>\nDevelopers and agent administrators use this skill to audit installed agent skills for actual usage, functional overlap, outcome impact, runtime burden, and cleanup recommendations before making manual retention decisions. <br>\n\n### Deployment Geography for Use: <br>\nGlobal <br>\n\n## Known Risks and Mitigations: <br>\nRisk: History and usage files may contain sensitive conversations, local paths, project names, or customer data. <br>\nMitigation: Pass explicit --skills-root paths and provide history or usage files only when comfortable with local scanning. <br>\nRisk: Cleanup recommendations could be mistaken for authorization to remove, merge, quarantine, isolate, or disable skills. <br>\nMitigation: Treat those recommendations as manual-review prompts and do not change installed skills automatically. <br>\n\n\n## Reference(s): <br>\n- [Scoring Rubric](references/scoring-rubric.md) <br>\n- [Ablation Protocol](references/ablation-protocol.md) <br>\n- [Report Delivery Contract](references/report-narration-prompt.md) <br>\n- [ClawHub Skill Page](https://clawhub.ai/gongyu0918-debug/skills/skill-usefulness-audit) <br>\n- [Project Homepage](https://github.com/gongyu0918-debug/skill-usefulness-audit) <br>\n\n\n## Skill Output: <br>\n**Output Type(s):** [text, markdown, JSON, shell commands, guidance] <br>\n**Output Format:** [Natural-language report with optional Markdown evidence, JSON evidence, and ablation-plan files] <br>\n**Output Parameters:** [1D] <br>\n**Other Properties Related to Output:** [Outputs can include local audit recommendations; deletion, merge, quarantine, or disable actions require manual review.] <br>\n\n## Skill Version(s): <br>\n0.3.19 (source: frontmatter and server release evidence) <br>\n\n## Ethical Considerations: <br>\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment. <br>\n\nFile v0.3.19:LICENSE\n\nMIT No Attribution\n\nCopyright 2026 gongyu0918-debug\n\nPermission is hereby granted, free of charge, to any person obtaining a copy of this software and associated documentation files (the \"Software\"), to deal in the Software without restriction, including without limitation the rights to use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the Software, and to permit persons to whom the Software is furnished to do so.\n\nTHE SOFTWARE IS PROVIDED \"AS IS\", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.\n\nArchive v0.3.18: 23 files, 71237 bytes\n\nFiles: LICENSE (911b), references/ablation-protocol.md (3995b), references/report-narration-prompt.md (944b), references/scoring-rubric.md (11776b), scripts/skill_usefulness_audit_lib/__init__.py (219b), scripts/skill_usefulness_audit_lib/ablation.py (5078b), scripts/skill_usefulness_audit_lib/cli.py (37159b), scripts/skill_usefulness_audit_lib/common.py (24079b), scripts/skill_usefulness_audit_lib/community.py (7317b), scripts/skill_usefulness_audit_lib/constants.py (11293b), scripts/skill_usefulness_audit_lib/reporting.py (58738b), scripts/skill_usefulness_audit_lib/risk_quality.py (46000b), scripts/skill_usefulness_audit_lib/risk_signatures_encoding.py (271b), scripts/skill_usefulness_audit_lib/risk_signatures_execution.py (687b), scripts/skill_usefulness_audit_lib/risk_signatures_network.py (581b), scripts/skill_usefulness_audit_lib/risk_signatures_sensitive.py (326b), scripts/skill_usefulness_audit_lib/risk_signatures.py (394b), scripts/skill_usefulness_audit_lib/scoring.py (18994b), scripts/skill_usefulness_audit_lib/usage_loader.py (11893b), scripts/skill_usefulness_audit.py (793b), skill-card.md (2576b), SKILL.md (6487b), _meta.json (142b)\n\nFile v0.3.18:SKILL.md\n\n---\nname: skill-usefulness-audit\nslug: skill-usefulness-audit\ndescription: Review your installed agent skills to see what you actually use, what overlaps, and what may no longer be worth keeping.\nversion: 0.3.18\ntags: [\"audit\",\"skills\",\"ablation\",\"openclaw\"]\nuser-invocable: true\ndisable-model-invocation: true\nargument-hint: --skills-root PATH --usage-file FILE\nhomepage: https://github.com/gongyu0918-debug/skill-usefulness-audit\nmetadata: {\"openclaw\":{\"skillKey\":\"skill-usefulness-audit\",\"requires\":{\"bins\":[\"python\"]},\"homepage\":\"https://github.com/gongyu0918-debug/skill-usefulness-audit\"}}\n---\n# Skill Usefulness Audit\n\n## Manual Trigger Only\n\nUse this skill only after a direct request to audit installed agent skills, their usage, overlap, cleanup options, or a structure-only inventory.\nDo not invoke it during normal tasks or use it for ordinary repository/source-code review, general security audit, or employee/human skill assessment.\n\n## Safety\n\nNever delete, merge, quarantine, isolate, or disable skills automatically.\nTreat `delete`, `merge-delete`, and `quarantine-review` as manual-review recommendations.\nDo not delete skills based only on a structure-only report.\nThis tool does not automatically replay historical conversations; it generates ablation plans and reads ablation result files that the user provides.\n\n## Audit Scope\n\nAudit these layers in order:\n\n1. Usage evidence, including recency and source quality.\n2. Installed metadata, instructions, and functional overlap.\n3. User-provided skill-on versus skill-off results for general skills.\n4. Runtime and bundle burden, including over-triggering, context cost, weak progressive disclosure, redundant resources, script failures, and private-looking files.\n5. Static health and risk hints.\n6. Optional offline community or registry metrics.\n\nTreat API and tool skills as protected capability skills during ablation.\nExamples: Excel, DOCX, PDF, browser automation, deployment, OCR, external API wrappers, MCP/API gateway helpers.\n\n## Workflow\n\n1. Collect user-provided roots before host-local defaults.\n2. Load only the usage, history, ablation, and community evidence that is available.\n3. Inspect each `SKILL.md` and its script/reference/asset metrics.\n4. Classify each skill as `api`, `tool`, or `general`, then score it with `{baseDir}/references/scoring-rubric.md`.\n5. Print the short usefulness report and, when requested, write Markdown evidence or an ablation plan.\n\n## Ablation Rules\n\nRead `{baseDir}/references/ablation-protocol.md` only when generating an ablation plan or evaluating ablation results.\nReplay only selected `general` candidates with identical prompts/artifacts and pairwise judging.\nDo not fake no-tool ablation for `api` or `tool` skills; use the rubric's protected-capability branch.\n\n## Run the Audit\n\nRun the audit after collecting available evidence:\n\n```bash\nREPORT_LANGUAGE=en  # use zh-CN when the current user invocation is Chinese\npython \"{baseDir}/scripts/skill_usefulness_audit.py\" audit \\\n  --skills-root ./skills \\\n  --report-language \"$REPORT_LANGUAGE\" \\\n  --markdown-out ./skill-audit-report.md\n```\n\nOpenClaw expands `{baseDir}` to the installed skill directory. Use it for bundled scripts and references.\n\nAdd evidence only when available:\n\n- `--usage-file`: JSON, JSONL, CSV, or TSV with per-skill usage.\n- `--history-file`: raw transcripts used only when direct usage is weak or missing; mentions remain `history_mentions` / `suspected_invocations`, not `calls`.\n- `--ablation-file`: normalized JSON or JSONL skill-on/skill-off results.\n- `--community-file`: offline JSON, JSONL, CSV, or TSV registry metrics.\n- `--ablation-plan-out`: a cost estimate and focused replay plan; its case counts can be overridden with the four `--ablation-*-cases` options documented by `--help`.\n- `--json-out`: machine-readable evidence only when requested or needed by another tool.\n\nPass `--report-language zh-CN` for a Chinese invocation and `--report-language en` for an English invocation. `auto` reads `SKILL_AUDIT_REPORT_LANGUAGE` or the process locale, then falls back to English.\n\nRun without extra files only when you need a structure-only audit.\nUsage, community, and ablation evidence become lower-confidence in that mode.\nHistory and usage files may contain sensitive conversations, local paths, project names, and customer data.\nMissing env means not configured in the current audit process, not proof that the skill is broken in every host.\n\n## Output Contract\n\nUse one run for both output layers; do not ask the user to choose a quick or full mode.\nStandard output is a short natural-language report. Its opening paragraph states the audited skill count and the total characters plus approximate tokens of loaded entry descriptions. Lead with actual usage, not static risk or bundle health, and keep scores, internal codes, risk flags, and tables out of this layer.\nWhen `--markdown-out` is provided, write the detailed evidence—with scores, action codes, missing evidence, burden, and risk notes—in the same run.\nMatch the user's language: clean Chinese for `zh-CN` and clean English for `en`, except for skill names and unavoidable paths or commands.\n\nBefore delivering the result, read `{baseDir}/references/report-narration-prompt.md`. Copy the short report to chat verbatim, apart from making its evidence path clickable. Do not paste raw JSON or the full Markdown evidence unless the user asks.\n\nJSON includes `report_mode`, per-skill `score_breakdown`, `quality_penalty`, `quality_penalty_uncapped`, `quality_evidence`, `community_breakdown`, `action_advice`, and `risk_review`. It includes `ablation_plan` only when `--ablation-plan-out` is used.\n\nKeep deletion advice conservative for system or host-core skills, and prefer narrowing or merging when overlapping skills still serve distinct host integrations.\n\n## Resources\n\n- `{baseDir}/scripts/skill_usefulness_audit.py`: compatibility wrapper for the modular audit package.\n- `{baseDir}/scripts/skill_usefulness_audit_lib/`: collect metadata, score skills, scan static risk hints, and render Markdown reports plus optional JSON artifacts.\n- `{baseDir}/references/report-narration-prompt.md`: concise prompt for turning the report into a user-facing conversational summary.\n- `{baseDir}/references/scoring-rubric.md`: 10-point scoring rules, confidence logic, community prior, and action thresholds.\n- `{baseDir}/references/ablation-protocol.md`: normalized replay method for historical conversations.\n\nFile v0.3.18:_meta.json\n\n{\n  \"ownerId\": \"kn7em0w89d0zac35fzt84qm2a182j54b\",\n  \"slug\": \"skill-usefulness-audit\",\n  \"version\": \"0.3.18\",\n  \"publishedAt\": 1785144769927\n}\n\nFile v0.3.18:references/ablation-protocol.md\n\n# Ablation Protocol\n\nUse this protocol for `general` skills selected by the ablation plan.\n\n## Contents\n\n- Goal\n- Cost-Efficient Triage\n- Sampling\n- Replay Method\n- Judge Method\n- Case Judgment\n- Judgment Rule\n- Early Stop Rules\n- Model Cost\n- Reporting\n\n## Goal\n\nMeasure whether the skill changes outcomes in a meaningful way.\n\nHigh consistency between skill-on and skill-off runs means the skill adds little value.\n\n## Cost-Efficient Triage\n\nGenerate a plan before replay by adding `--ablation-plan-out` to the audit command. Add `--json-out` only when machine-readable audit evidence is also needed.\n\nThe plan uses local evidence first:\n\n- final score\n- overlap\n- quality burden\n- activation volume\n- evidence confidence\n- missing or weak prior ablation\n\nIt then estimates model cost against a full protocol and writes an early-stop plan.\n\n## Sampling\n\nStart with `3` historical tasks per candidate skill.\nChoose tasks where the skill should plausibly matter.\nPrefer real user turns over synthetic prompts.\nExpand to `5` cases when the first batch is mixed.\nExpand to `10` cases only for high-impact or deletion-boundary decisions.\n\nOverride these defaults with:\n\n- `--ablation-baseline-cases`\n- `--ablation-initial-cases`\n- `--ablation-expand-cases`\n- `--ablation-max-cases`\n\n## Replay Method\n\nFor each selected case, run two isolated replays:\n\n1. `with_skill`\n2. `without_skill`\n\nKeep these constant:\n\n- same prompt\n- same files and artifacts\n- same model class when possible\n- same tool permissions\n- same success criteria\n\nUse a fresh thread or isolated run if the host supports it.\nSubagents are optional. They improve isolation and parallelism, but they increase total model spend when every branch runs full replay.\n\n## Judge Method\n\nUse pairwise comparison when judging open-ended outputs:\n\n1. Compare `with_skill` and `without_skill` side by side.\n2. Randomize A/B order.\n3. Spot-check reversed order on boundary cases.\n4. Prefer `pass/fail`, `same/better/worse`, and short reasons over long open-ended grading.\n\n## Case Judgment\n\nRecord:\n\n- `pass`: whether the run solved the task\n- `score`: optional `0.0-1.0` quality score\n- `tool_cost`: optional rough measure of tool calls, latency, or retries\n- `verdict`: `better`, `same`, or `worse`\n- `notes`: one short reason\n\n## Normalized JSON Example\n\n```json\n[\n  {\n    \"skill\": \"emotion-orchestrator\",\n    \"case_id\": \"case-001\",\n    \"with_skill\": {\"pass\": true, \"score\": 0.92},\n    \"without_skill\": {\"pass\": true, \"score\": 0.81},\n    \"verdict\": \"better\",\n    \"notes\": \"with-skill run adapted reply style and avoided a follow-up correction\"\n  }\n]\n```\n\n## Judgment Rule\n\nUse `same` when the final answer, correctness, and workflow remain materially equivalent.\nUse `better` when the skill improves correctness, speed, structure, or user-fit in a way the baseline did not.\nUse `worse` when the skill adds friction, drift, or errors.\nIgnore verdict-only cases with unsupported values. A case with an unknown or missing verdict is usable only when complete skill-on and skill-off pass/score fields allow the result to be inferred.\n\n## Early Stop Rules\n\n- Stop as low-value when `3/3` cases are `same` and `better_rate` is `0`.\n- Stop as useful when at least `2/3` cases are `better` and no case is `worse`.\n- Expand to `5` when the first batch is mixed.\n- Expand to `10` only for delete-boundary or high-impact decisions.\n\nDelete-boundary means \"needs stronger human review evidence\", not automatic deletion authority.\n\n## Model Cost\n\nThe audit script does not call an LLM during planning.\nThe plan estimates replay cost with three profiles:\n\n- `light`: about `6.2k` model-cost units per case\n- `realistic`: about `24k` model-cost units per case\n- `coding`: about `50k` model-cost units per case\n\nEach case assumes two replays plus one compact pairwise judge.\nThe JSON field `model_cost_estimates.unit` records this as `estimated_context_units_per_case`.\n\n## Reporting\n\nFeed the normalized result back into the audit with `--ablation-file`.\n\nFile v0.3.18:references/report-narration-prompt.md\n\n# Report Delivery Contract\n\nUse this after the audit command finishes.\n\n- Treat standard output as the final short report. Copy it verbatim into chat, apart from making the evidence path clickable when supported.\n- Do not add headings, bullets, extra counts, verification notes, or facts from the Markdown evidence unless requested.\n- Deliver both layers from the same run; do not offer separate short/full modes.\n- Keep the short report on actual use, missing use evidence, overlap, and verified outcome impact. Leave scores, internal codes, risk, bundle health, and tables in the Markdown evidence.\n- Match the user's language: clean Chinese for `zh-CN` and clean English for `en`, except for skill names and unavoidable paths or commands.\n- Do not paste raw JSON or read back the full Markdown report unless requested.\n- Treat removal results as manual-review recommendations. Never remove, merge, isolate, or disable a skill automatically.\n\nFile v0.3.18:references/scoring-rubric.md\n\n# Scoring Rubric\n\nScore each skill with one local 10-point score, one final score, and side signals.\n\n## Contents\n\n- Core Outputs\n- Usage Score\n- Uniqueness Score\n- Impact Score\n- Confidence Score\n- Quality Penalty\n- Community Prior Score\n- Static Risk Level\n- Verdict Bands\n- Action Rules\n\n## Core Outputs\n\n- `local_score = usage_score + uniqueness_score + impact_score`\n- `quality_penalty`: `0.0-2.5`\n- `quality_penalty_uncapped`: raw quality burden before the cap\n- `static_quality_penalty`: `0.0-1.4`\n- `final_score = clamp(local_score - quality_penalty, 0.0, 10.0)`\n- `usage_score`: `0.0-3.0`\n- `uniqueness_score`: `0.0-3.0`\n- `impact_score`: `0.0-4.0`\n- `confidence_score`: `0.0-1.0`\n- `community_prior_score`: `0.0-1.0`\n- `risk_level` / `static_risk_level`: `none / low / medium / high`\n\nKeep `community_prior_score` and static risk fields separate from `local_score`.\nUse quality burden, community prior, and static risk hints to shape review priority and final action.\n\n## 1. Usage Score (`0.0-3.0`)\n\nPrefer direct host usage logs.\nUse transcript mentions only as weaker fallback evidence.\n\n### Input Fields\n\n- `calls`\n- `recent_30d_calls`\n- `recent_90d_calls`\n- `last_used_at`\n- `active_days`\n\nHistory fallback fields:\n\n- `history_mentions`\n- `suspected_invocations`\n\nTranscript mentions are weak evidence only. They may influence the usage score through the history evidence weight, but they must not be reported as direct `calls`.\n- `usage_source`\n- `evidence_weight`\n- `executions`\n- `script_failures`\n- `repair_turns`\n- `reference_loads`\n- `false_triggers`\n\n### Base Usage Strength\n\n- When `recent_30d_calls` exists:\n  - `0.0`: `0`\n  - `1.0`: `1-2`\n  - `2.0`: `3-7`\n  - `3.0`: `8+`\n- When only `recent_90d_calls` exists:\n  - `0.0`: `0`\n  - `0.75`: `1-2`\n  - `1.5`: `3-9`\n  - `2.5`: `10+`\n- When only total `calls` exists:\n  - `0.0`: `0`\n  - `1.0`: `1-2`\n  - `2.0`: `3-9`\n  - `3.0`: `10+`\n\n### Recency Adjustments\n\n- add `0.5` when `last_used_at <= 7 days`\n- add `0.25` when `last_used_at <= 30 days`\n- subtract `0.5` when `last_used_at > 180 days`\n- add `0.25` when `active_days >= 10`\n- add `0.10` when `active_days >= 3`\n\n### Evidence Weight\n\n- `1.00`: direct usage file\n- `0.45`: transcript-history fallback based on `suspected_invocations`\n- `0.00`: missing usage evidence\n\nClamp the final usage score to `0.0-3.0`.\n\n## 2. Uniqueness Score (`0.0-3.0`)\n\nMeasure the highest functional-overlap similarity against any other installed skill.\nUse description, headings, and resource names as the comparison surface.\n\nBuckets:\n\n- `0.0`: highest overlap `>= 0.85`\n- `1.0`: highest overlap `0.65-0.84`\n- `2.0`: highest overlap `0.40-0.64`\n- `3.0`: highest overlap `< 0.40`\n\n## 3. Impact Score (`0.0-4.0`)\n\n### General skills\n\nUse ablation on historical conversations.\nCompute:\n\n- `consistency_rate`: skill-on and skill-off produce materially equivalent outcomes\n- `better_rate`: skill-on clearly improves the result\n- `worse_rate`: skill-on clearly harms the result\n\nBase score from consistency:\n\n- `0.0`: `consistency_rate >= 0.85`\n- `1.0`: `0.70-0.84`\n- `2.0`: `0.55-0.69`\n- `3.0`: `0.35-0.54`\n- `4.0`: `< 0.35`\n\nAdjustments:\n\n- add `1.0` when `better_rate - worse_rate >= 0.30`\n- subtract `1.0` when `worse_rate > better_rate`\n- clamp the final impact score to `0.0-4.0`\n\nWhen ablation is missing, use low-evidence score `1.0` for zero-call skills.\nFor skills with direct usage evidence but no ablation yet, keep temporary neutral score `2.0` and lower confidence.\n\n### API and tool skills\n\nSkip history ablation.\nUse protected-capability scoring instead:\n\n- start at `2.0`\n- add `1.0` when the skill ships executable scripts or hard capability resources\n- add `0.5` when highest overlap `< 0.35`\n- add `0.5` when calls `>= 3`\n- subtract `1.0` when highest overlap `>= 0.75`\n- subtract `0.5` when calls are `0`\n- clamp the final impact score to `0.0-4.0`\n\n## 4. Confidence Score (`0.0-1.0`)\n\nConfidence describes evidence quality, not usefulness.\n\nAdd:\n\n- `0.35` for direct usage files\n- `0.15` for history fallback\n- `0.20` when recent usage fields exist\n- `0.10` when only total direct calls exist\n- `0.25` for protected `api/tool` classification\n- `0.25` for `ge\n\nArchive v0.3.17: 23 files, 71540 bytes\n\nFiles: LICENSE (911b), references/ablation-protocol.md (3995b), references/report-narration-prompt.md (944b), references/scoring-rubric.md (11776b), scripts/skill_usefulness_audit_lib/__init__.py (219b), scripts/skill_usefulness_audit_lib/ablation.py (5078b), scripts/skill_usefulness_audit_lib/cli.py (37159b), scripts/skill_usefulness_audit_lib/common.py (24079b), scripts/skill_usefulness_audit_lib/community.py (7317b), scripts/skill_usefulness_audit_lib/constants.py (11293b), scripts/skill_usefulness_audit_lib/reporting.py (58738b), scripts/skill_usefulness_audit_lib/risk_quality.py (46000b), scripts/skill_usefulness_audit_lib/risk_signatures_encoding.py (271b), scripts/skill_usefulness_audit_lib/risk_signatures_execution.py (687b), scripts/skill_usefulness_audit_lib/risk_signatures_network.py (581b), scripts/skill_usefulness_audit_lib/risk_signatures_sensitive.py (326b), scripts/skill_usefulness_audit_lib/risk_signatures.py (394b), scripts/skill_usefulness_audit_lib/scoring.py (18994b), scripts/skill_usefulness_audit_lib/usage_loader.py (11893b), scripts/skill_usefulness_audit.py (793b), skill-card.md (2424b), SKILL.md (7726b), _meta.json (142b)\n\nArchive v0.3.16: 23 files, 71872 bytes\n\nFiles: LICENSE (911b), references/ablation-protocol.md (4955b), references/report-narration-prompt.md (1341b), references/scoring-rubric.md (11776b), scripts/skill_usefulness_audit_lib/__init__.py (219b), scripts/skill_usefulness_audit_lib/ablation.py (5078b), scripts/skill_usefulness_audit_lib/cli.py (32865b), scripts/skill_usefulness_audit_lib/common.py (24079b), scripts/skill_usefulness_audit_lib/community.py (7317b), scripts/skill_usefulness_audit_lib/constants.py (11293b), scripts/skill_usefulness_audit_lib/reporting.py (58738b), scripts/skill_usefulness_audit_lib/risk_quality.py (44424b), scripts/skill_usefulness_audit_lib/risk_signatures_encoding.py (271b), scripts/skill_usefulness_audit_lib/risk_signatures_execution.py (687b), scripts/skill_usefulness_audit_lib/risk_signatures_network.py (581b), scripts/skill_usefulness_audit_lib/risk_signatures_sensitive.py (326b), scripts/skill_usefulness_audit_lib/risk_signatures.py (394b), scripts/skill_usefulness_audit_lib/scoring.py (18994b), scripts/skill_usefulness_audit_lib/usage_loader.py (11893b), scripts/skill_usefulness_audit.py (793b), skill-card.md (2827b), SKILL.md (10152b), _meta.json (142b)\n\nArchive v0.3.15: 23 files, 69283 bytes\n\nFiles: LICENSE (911b), references/ablation-protocol.md (4955b), references/report-narration-prompt.md (1223b), references/scoring-rubric.md (11776b), scripts/skill_usefulness_audit_lib/__init__.py (219b), scripts/skill_usefulness_audit_lib/ablation.py (5078b), scripts/skill_usefulness_audit_lib/cli.py (32309b), scripts/skill_usefulness_audit_lib/common.py (24079b), scripts/skill_usefulness_audit_lib/community.py (7317b), scripts/skill_usefulness_audit_lib/constants.py (11293b), scripts/skill_usefulness_audit_lib/reporting.py (49794b), scripts/skill_usefulness_audit_lib/risk_quality.py (44368b), scripts/skill_usefulness_audit_lib/risk_signatures_encoding.py (271b), scripts/skill_usefulness_audit_lib/risk_signatures_execution.py (687b), scripts/skill_usefulness_audit_lib/risk_signatures_network.py (581b), scripts/skill_usefulness_audit_lib/risk_signatures_sensitive.py (326b), scripts/skill_usefulness_audit_lib/risk_signatures.py (394b), scripts/skill_usefulness_audit_lib/scoring.py (18994b), scripts/skill_usefulness_audit_lib/usage_loader.py (11893b), scripts/skill_usefulness_audit.py (793b), skill-card.md (2341b), SKILL.md (9381b), _meta.json (142b)\n\nArchive v0.3.14: 23 files, 67877 bytes\n\nFiles: LICENSE (911b), references/ablation-protocol.md (4759b), references/report-narration-prompt.md (1223b), references/scoring-rubric.md (11567b), scripts/skill_usefulness_audit_lib/__init__.py (219b), scripts/skill_usefulness_audit_lib/ablation.py (5005b), scripts/skill_usefulness_audit_lib/cli.py (32309b), scripts/skill_usefulness_audit_lib/common.py (24079b), scripts/skill_usefulness_audit_lib/community.py (7317b), scripts/skill_usefulness_audit_lib/constants.py (11293b), scripts/skill_usefulness_audit_lib/reporting.py (49794b), scripts/skill_usefulness_audit_lib/risk_quality.py (40509b), scripts/skill_usefulness_audit_lib/risk_signatures_encoding.py (271b), scripts/skill_usefulness_audit_lib/risk_signatures_execution.py (687b), scripts/skill_usefulness_audit_lib/risk_s...","readmeExcerpt":"Skill: skill-usefulness-audit Owner: gongyu0918-debug Summary: Review your installed agent skills to see what you actually use, what overlaps, and what may no longer be worth keeping. Tags: ablation:0.3.24, audit:0.3.24, latest:0.3.24, latest audit skills:0.2.7, latest audit skills openclaw:0.3.14, latest audit skills openclaw hermes claude-code:0.2.11, openclaw:0.3.24, skills:0.3.24 Version history: v0.3.24 | 2026-0","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"REPORT_LANGUAGE=en  # use zh-CN when the current user invocation is Chinese\npython \"{baseDir}/scripts/skill_usefulness_audit.py\" audit \\\n  --skills-root ./skills \\\n  --report-language \"$REPORT_LANGUAGE\" \\\n  --markdown-out ./skill-audit-report.md"},{"language":"json","snippet":"[\n  {\n    \"skill\": \"emotion-orchestrator\",\n    \"case_id\": \"case-001\",\n    \"with_skill\": {\"pass\": true, \"score\": 0.92},\n    \"without_skill\": {\"pass\": true, \"score\": 0.81},\n    \"verdict\": \"better\"\n  }\n]"},{"language":"bash","snippet":"REPORT_LANGUAGE=en  # use zh-CN when the current user invocation is Chinese\npython \"{baseDir}/scripts/skill_usefulness_audit.py\" audit \\\n  --skills-root ./skills \\\n  --report-language \"$REPORT_LANGUAGE\" \\\n  --markdown-out ./skill-audit-report.md"},{"language":"json","snippet":"[\n  {\n    \"skill\": \"emotion-orchestrator\",\n    \"case_id\": \"case-001\",\n    \"with_skill\": {\"pass\": true, \"score\": 0.92},\n    \"without_skill\": {\"pass\": true, \"score\": 0.81},\n    \"verdict\": \"better\"\n  }\n]"},{"language":"bash","snippet":"REPORT_LANGUAGE=en  # use zh-CN when the current user invocation is Chinese\npython \"{baseDir}/scripts/skill_usefulness_audit.py\" audit \\\n  --skills-root ./skills \\\n  --report-language \"$REPORT_LANGUAGE\" \\\n  --markdown-out ./skill-audit-report.md"},{"language":"json","snippet":"[\n  {\n    \"skill\": \"emotion-orchestrator\",\n    \"case_id\": \"case-001\",\n    \"with_skill\": {\"pass\": true, \"score\": 0.92},\n    \"without_skill\": {\"pass\": true, \"score\": 0.81},\n    \"verdict\": \"better\"\n  }\n]"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: skill-usefulness-audit\nslug: skill-usefulness-audit\ndescription: Review your installed agent skills to see what you actually use, what overlaps, and what may no longer be worth keeping.\nversion: 0.3.24\ntags: [\"audit\",\"skills\",\"ablation\",\"openclaw\"]\nuser-invocable: true\ndisable-model-invocation: true\nargument-hint: --skills-root PATH --usage-file FILE\nhomepage: https://github.com/gongyu0918-debug/skill-usefulness-audit\nmetadata: {\"openclaw\":{\"skillKey\":\"skill-usefulness-audit\",\"requires\":{\"bins\":[\"python\"]},\"homepage\":\"https://github.com/gongyu0918-debug/skill-usefulness-audit\"}}\n---\n# Skill Usefulness Audit\n\n## Manual Trigger Only\n\nUse this skill only after a direct request to audit installed agent skills, their usage, overlap, cleanup options, or a structure-only inventory.\nDo not invoke it during normal tasks or use it for ordinary repository/source-code review, general security audit, or employee/human skill assessment.\n\n## Safety\n\nNever delete, merge, quarantine, isolate, or disable skills automatically.\nTreat `delete`, `merge-delete`, and `quarantine-review` as manual-review recommendations.\nDo not delete skills based only on a structure-only report.\nThis tool does not automatically replay historical conversations; it generates ablation plans and reads ablation result files that the user provides.\n\n## Audit Scope\n\nAudit these layers in order:\n\n1. Usage evidence, including recency and source quality.\n2. Installed metadata, instructions, and functional overlap.\n3. User-provided skill-on versus skill-off results for general skills.\n4. Runtime and bundle burden, including over-triggering, context cost, weak progressive disclosure, redundant resources, script failures, and private-looking files.\n5. Static health and risk hints.\n6. Optional offline community or registry metrics.\n\nTreat API and tool skills as protected capability skills during ablation.\nExamples: Excel, DOCX, PDF, browser automation, deployment, OCR, external API wrappers, MCP/API gateway helpers.\n\n## Workflow\n\n1. Collect user-provided roots before host-local defaults.\n2. Load only the usage, history, ablation, and community evidence that is available.\n3. Inspect each `SKILL.md` and its script/reference/asset metrics.\n4. Let the bundled script classify each skill as `api`, `tool`, or `general` and calculate its score. Read `{baseDir}/references/scoring-rubric.md` only when checking or explaining a score, verdict, or action.\n5. Print the short usefulness report and, when requested, write Markdown evidence or an ablation plan.\n\n## Ablation Rules\n\nRead `{baseDir}/references/ablation-protocol.md` only when running replays, preparing normalized ablation records, or reviewing mixed or delete-boundary results. The script can generate an ablation plan without loading the protocol.\nReplay only selected `general` candidates with identical prompts/artifacts and pairwise judging.\nDo not fake no-tool ablation for `api` or `tool` skills; use the rubric's protected-capability branch.\n\n#"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn7em0w89d0zac35fzt84qm2a182j54b\",\n  \"slug\": \"skill-usefulness-audit\",\n  \"version\": \"0.3.24\",\n  \"publishedAt\": 1787566137284\n}"},{"path":"references/ablation-protocol.md","content":"# Ablation Protocol\n\nUse this protocol for `general` skills selected by the ablation plan.\n\n## Goal\n\nMeasure whether the skill changes outcomes in a meaningful way.\nHigh consistency between skill-on and skill-off runs means the skill adds little value.\n\n## Sampling\n\nStart with `3` historical tasks where the skill should plausibly matter. Prefer real user turns over synthetic prompts. Expand to `5` when results are mixed and to `10` only for high-impact or delete-boundary decisions.\n\n## Replay Method\n\nFor each selected case, run two isolated replays:\n\n1. `with_skill`\n2. `without_skill`\n\nKeep these constant:\n\n- same prompt\n- same files and artifacts\n- same model class when possible\n- same tool permissions\n- same success criteria\n\nUse a fresh thread or isolated run if the host supports it.\n\n## Judge Method\n\nFor open-ended outputs:\n\n1. Compare `with_skill` and `without_skill` side by side.\n2. Randomize A/B order.\n3. Spot-check reversed order on boundary cases.\n4. Prefer `pass/fail`, `same/better/worse`, and short reasons over long open-ended grading.\n\nRecord a standard `verdict` and one short `notes` reason. Each arm may also include optional `pass` and/or `score` from `0.0-1.0` for fallback inference. Optional `tool_cost` may describe calls, latency, or retries and currently does not affect the audit score.\n\n### Normalized JSON\n\n```json\n[\n  {\n    \"skill\": \"emotion-orchestrator\",\n    \"case_id\": \"case-001\",\n    \"with_skill\": {\"pass\": true, \"score\": 0.92},\n    \"without_skill\": {\"pass\": true, \"score\": 0.81},\n    \"verdict\": \"better\"\n  }\n]\n```\n\n## Judgment Rule\n\nUse `same` when the final answer, correctness, and workflow remain materially equivalent.\nUse `better` when the skill improves correctness, speed, structure, or user-fit in a way the baseline did not.\nUse `worse` when the skill adds friction, drift, or errors.\n\nIgnore verdict-only cases with unsupported values. A case with an unknown or missing verdict is usable only when both arms provide comparable `pass` and/or `score` fields for inference.\n\n## Early Stop Rules\n\n- Stop as low-value when `3/3` cases are `same` and `better_rate` is `0`.\n- Stop as useful when at least `2/3` cases are `better` and no case is `worse`.\n- Expand to `5` when the first batch is mixed.\n- Expand to `10` only for delete-boundary or high-impact decisions.\n\nDelete-boundary means \"needs stronger human review evidence\", not automatic deletion authority.\n\n## Planning and Model Cost\n\nCreate the replay plan with `--ablation-plan-out`. Planning uses local evidence and does not call an LLM.\n\nThe plan estimates replay cost for `light`, `realistic`, and `coding` profiles. Each case assumes two replays and one compact pairwise judge. `model_cost_estimates.unit` records `estimated_context_units_per_case`.\n\nFeed normalized results back with `--ablation-file`."},{"path":"references/report-narration-prompt.md","content":"# Report Delivery Contract\n\n- Treat standard output as the final short report. Copy it verbatim into chat, apart from making the evidence path clickable when supported.\n- Do not add headings, bullets, extra counts, verification notes, or facts from the Markdown evidence unless requested.\n- Keep the short report on actual use, missing use evidence, overlap, and verified outcome impact. Leave scores, internal codes, risk, bundle health, and tables in the Markdown evidence.\n- Match the user's language: clean Chinese for `zh-CN` and clean English for `en`, except for skill names and unavoidable paths or commands.\n- Do not paste raw JSON or read back the full Markdown report unless requested.\n- Treat removal results as manual-review recommendations. Never remove, merge, isolate, or disable a skill automatically."},{"path":"references/scoring-rubric.md","content":"# Scoring Rubric\n\n## Contents\n\n- Core Outputs\n- Usage Score\n- Uniqueness Score\n- Impact Score\n- Confidence Score\n- Quality Penalty\n- Community Prior Score\n- Static Risk Level\n- Verdict Bands\n- Action Rules\n\n## Core Outputs\n\n- `local_score = usage_score + uniqueness_score + impact_score`\n- `quality_penalty`: `0.0-2.5`\n- `quality_penalty_uncapped`: raw quality burden before the cap\n- `static_quality_penalty`: `0.0-1.4`\n- `final_score = clamp(local_score - quality_penalty, 0.0, 10.0)`\n- `risk_level` / `static_risk_level`: `none / low / medium / high`\n\n## 1. Usage Score (`0.0-3.0`)\n\nPrefer direct host usage logs.\nUse transcript mentions only as weaker fallback evidence.\n\n### Input Fields\n\n- Direct usage: `calls`, `recent_30d_calls`, `recent_90d_calls`, `last_used_at`, and `active_days`.\n- History fallback: `history_mentions` and `suspected_invocations`. These are weak evidence weighted through history and must not be reported as direct `calls`.\n- Evidence and runtime burden: `usage_source`, `evidence_weight`, `executions`, `script_failures`, `repair_turns`, `reference_loads`, and `false_triggers`.\n\n### Base Usage Strength\n\n- When `recent_30d_calls` exists:\n  - `0.0`: `0`\n  - `1.0`: `1-2`\n  - `2.0`: `3-7`\n  - `3.0`: `8+`\n- When only `recent_90d_calls` exists:\n  - `0.0`: `0`\n  - `0.75`: `1-2`\n  - `1.5`: `3-9`\n  - `2.5`: `10+`\n- When only total `calls` exists:\n  - `0.0`: `0`\n  - `1.0`: `1-2`\n  - `2.0`: `3-9`\n  - `3.0`: `10+`\n\n### Recency Adjustments\n\n- add `0.5` when `last_used_at <= 7 days`\n- add `0.25` when `last_used_at <= 30 days`\n- subtract `0.5` when `last_used_at > 180 days`\n- add `0.25` when `active_days >= 10`\n- add `0.10` when `active_days >= 3`\n\n### Evidence Weight\n\n- `1.00`: direct usage file\n- `0.45`: transcript-history fallback based on `suspected_invocations`\n- `0.00`: missing usage evidence\n\n## 2. Uniqueness Score (`0.0-3.0`)\n\nMeasure the highest functional-overlap similarity against any other installed skill using descriptions, headings, and resource names.\n\nBuckets:\n\n- `0.0`: highest overlap `>= 0.85`\n- `1.0`: highest overlap `0.65-0.84`\n- `2.0`: highest overlap `0.40-0.64`\n- `3.0`: highest overlap `< 0.40`\n\n## 3. Impact Score (`0.0-4.0`)\n\n### General skills\n\nUse ablation on historical conversations.\nCompute:\n\n- `consistency_rate`: skill-on and skill-off produce materially equivalent outcomes\n- `better_rate`: skill-on clearly improves the result\n- `worse_rate`: skill-on clearly harms the result\n\nBase score from consistency:\n\n- `0.0`: `consistency_rate >= 0.85`\n- `1.0`: `0.70-0.84`\n- `2.0`: `0.55-0.69`\n- `3.0`: `0.35-0.54`\n- `4.0`: `< 0.35`\n\nAdjustments:\n\n- add `1.0` when `better_rate - worse_rate >= 0.30`\n- subtract `1.0` when `worse_rate > better_rate`\n\nWhen ablation is missing, use low-evidence score `1.0` for zero-call skills.\nFor skills with direct usage evidence but no ablation yet, keep temporary neutral score `2.0` and lower confidence.\n\n### API and tool skills\n\nSkip history ablation.\nUse protected-capability scoring instead:\n\n-"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":null,"editorialQuality":{"score":100,"threshold":65,"status":"thin","wordCount":1847,"uniquenessScore":44,"reasons":["uniqueness-below-45"]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-09T06:14:25.202Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-09T06:14:25.202Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-09T15:19:05.433Z","emptyReason":null},"items":[{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-10T18:48:31.762Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}