{"id":"62253e86-3aec-4462-9ba5-1d0192be13c7","entityType":"agent","slug":"clawhub-voronindenis5-prompt-archaeology","name":"prompt-archaeology","canonicalUrl":"https://www.xpersona.co/agent/clawhub-voronindenis5-prompt-archaeology","canonicalPath":"/agent/clawhub-voronindenis5-prompt-archaeology","generatedAt":"2026-10-10T07:20:11.417Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-09T13:43:44.391Z","emptyReason":null},"description":"Excavate forgotten solutions, code snippets, and decisions from past conversation sessions. Use when the user is re-solving a problem you've likely solved before, hunting for a lost snippet, or wants to mine session history for buried knowledge instead of starting from scratch. Skill: prompt-archaeology Owner: voronindenis5 Summary: Excavate forgotten solutions, code snippets, and decisions from past conversation sessions. Use when the user is re-solving a problem you've likely solved before, hunting for a lost snippet, or wants to mine session history for buried knowledge instead of starting from scratch. Tags: latest:0.1.1 Version history: v0.1.1 | 2026-08-11T11:58:40.748Z | auto Version","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 2.5K downloads reported by the source. Last updated 10/9/2026.","installCommand":"clawhub skill install s17b6amkd3wzqgg640v03a9r1n83gxs1:prompt-archaeology","sourceUrl":"https://clawhub.ai/voronindenis5/prompt-archaeology","homepage":"https://clawhub.ai/voronindenis5/skills/prompt-archaeology","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/voronindenis5/prompt-archaeology","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/voronindenis5/skills/prompt-archaeology","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":68,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Excavate forgotten solutions, code snippets, and decisions from past conversation sessions. Use when the user is re-solving a problem you've likely solved befor"},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-09T13:43:44.391Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T13:43:44.391Z","emptyReason":null},"stars":null,"forks":null,"downloads":2540,"packageName":null,"latestVersion":"0.1.1","tractionLabel":"2.5K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T13:43:44.391Z","emptyReason":null},"lastUpdatedAt":"2026-10-09T13:43:44.391Z","lastCrawledAt":"2026-10-09T13:43:44.391Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-10T13:43:44.391Z","lastVerifiedAt":null,"highlights":[{"version":"0.1.1","createdAt":"2026-08-11T11:58:40.748Z","changelog":"Version 0.1.1 - Removed the sample file: skill-card.md - No other functional or documentation changes detected.","fileCount":13,"zipByteSize":30133},{"version":"0.1.0","createdAt":"2026-08-05T19:49:27.842Z","changelog":"Initial release summary: Introduces a new approach for mining past conversations for code, decisions, and solutions. - Search and excavate session history using keyword, semantic, temporal, and structural strategies. - Includes transparent relevance scoring and deduplication to surface the most useful prior artifacts. - Provides a standalone `excavate.py` script for searching, ranking, and extracting buried knowledge from session logs. - Designed to help users avoid re-solving problems, rediscover lost snippets, and learn from previous decisions and rejected approaches.","fileCount":13,"zipByteSize":30197}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s17b6amkd3wzqgg640v03a9r1n83gxs1:prompt-archaeology","setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-voronindenis5-prompt-archaeology/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-voronindenis5-prompt-archaeology/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-voronindenis5-prompt-archaeology/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-voronindenis5-prompt-archaeology/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-voronindenis5-prompt-archaeology/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-voronindenis5-prompt-archaeology/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-10T07:20:11.415Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-voronindenis5-prompt-archaeology/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-voronindenis5-prompt-archaeology/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-voronindenis5-prompt-archaeology/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-voronindenis5-prompt-archaeology/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-09T13:43:44.391Z","emptyReason":null},"readme":"Skill: prompt-archaeology\n\nOwner: voronindenis5\n\nSummary: Excavate forgotten solutions, code snippets, and decisions from past conversation sessions. Use when the user is re-solving a problem you've likely solved before, hunting for a lost snippet, or wants to mine session history for buried knowledge instead of starting from scratch.\n\nTags: latest:0.1.1\n\nVersion history:\n\nv0.1.1 | 2026-08-11T11:58:40.748Z | auto\n\nVersion 0.1.1\n\n- Removed the sample file: skill-card.md\n- No other functional or documentation changes detected.\n\nv0.1.0 | 2026-08-05T19:49:27.842Z | auto\n\nInitial release summary: Introduces a new approach for mining past conversations for code, decisions, and solutions.\n\n- Search and excavate session history using keyword, semantic, temporal, and structural strategies.\n- Includes transparent relevance scoring and deduplication to surface the most useful prior artifacts.\n- Provides a standalone `excavate.py` script for searching, ranking, and extracting buried knowledge from session logs.\n- Designed to help users avoid re-solving problems, rediscover lost snippets, and learn from previous decisions and rejected approaches.\n\nArchive index:\n\nArchive v0.1.1: 13 files, 30133 bytes\n\nFiles: LICENSE (1070b), README.md (4838b), references (0b), references/cli-reference.md (6262b), references/deduplication.md (4328b), references/extraction-patterns.md (4257b), references/relevance-scoring.md (5157b), references/search-strategies.md (4953b), scripts (0b), scripts/excavate.py (22802b), skill-card.md (2705b), SKILL.md (13010b), _meta.json (137b)\n\nFile v0.1.1:SKILL.md\n\n---\nname: prompt-archaeology\ndescription: \"Excavate forgotten solutions, code snippets, and decisions from past conversation sessions. Use when the user is re-solving a problem you've likely solved before, hunting for a lost snippet, or wants to mine session history for buried knowledge instead of starting from scratch.\"\nversion: 1.0.0\nauthor: Denis Voronin\nlicense: MIT\nmetadata:\n  hermes:\n    tags: [sessions, history, search, knowledge-mining, archaeology, recovery, hermes-agent]\n    related_skills: []\n---\n\n# Prompt Archaeology\n\n## Overview\n\n**Prompt Archaeology** is the practice of excavating your own conversation history instead of re-solving problems from scratch. Every AI session is a stratum — a sedimented layer of debugging, decision-making, and discovery. Over time, valuable artifacts sink below the surface: a one-liner that fixed a gnarly race condition, a config that satisfied a finicky build, the exact incantation that convinced a model to behave. Most agents never dig for these. They re-derive, re-guess, and re-fail.\n\nThis skill turns that history into a quarryable resource. It bundles:\n\n- **Search strategies** — keyword, semantic-adjacent, temporal, and structural queries tuned for session transcripts.\n- **Relevance scoring** — a transparent, composable ranking that surfaces the one session that actually matters.\n- **Knowledge extraction patterns** — recipes for pulling *decisions* and *solutions* out of a wall of chat, not just matching text.\n- **Deduplication** — collapse near-duplicate fixes across sessions into a single canonical answer.\n- **`excavate.py`** — a standalone Python script that crawls session logs and markdown files, ranks them, and prints the buried artifacts.\n\nThe metaphor is deliberate. An archaeologist does not grep the desert for \"pottery\" and ship the first hit. They survey, triangulate, carefully extract, and catalog. This skill teaches the agent to do the same with its own past.\n\n## When to Use\n\n- **The user is about to re-solve a known problem.** They describe a bug or task and you have a flicker of \"we've done this before.\" Excavate before answering.\n- **\"Didn't we figure out...?\" / \"What did we land on?\"** — retrieve the prior decision and its rationale, not just the outcome.\n- **Hunting for a lost code snippet, config value, or command** that worked months ago.\n- **Onboarding to a codebase you've touched before** — pull the architectural decisions out of old sessions.\n- **Avoiding repeated dead ends** — find the approaches that were *rejected* and why, so you don't walk back into them.\n- **Writing postmortems or ADRs** from scattered session evidence.\n\n### Don't use for\n\n- Fresh problems with no prior history — there's nothing to excavate; solve forward.\n- When you already hold the answer in active context — don't pad the turn with a search.\n- Sensitive retrieval across other users' private profiles unless explicitly authorized.\n\n## The Excavation Workflow\n\nA dig has five phases. Skipping any phase degrades result quality.\n\n### 1. Survey — frame the query\n\nBefore searching, state **what artifact you want** and **what shape it takes**:\n\n| Artifact you want | Query shape | Example seeds |\n|---|---|---|\n| A fix for a bug | error string + symptom words | the exception text, \"traceback\", the failing assertion |\n| A decision + rationale | the option names + \"decided\" / \"chose\" / \"went with\" | the two libraries you were weighing |\n| A config value | the key name + surrounding file | `\"max_connections\"`, `nginx.conf` |\n| A rejected approach | the approach + \"didn't work\" / \"gave up\" / \"abandoned\" | the tool you tried first |\n| A command / incantation | the tool + the goal verb | `ffmpeg`, \"concatenate\" |\n\nWrite the query down. A vague survey yields a vague dig.\n\n### 2. Locate — run the searches\n\nRun **multiple passes**, not one. Different phrasings live in different sessions.\n\n- **Exact / keyword pass** — the literal error string, function name, or filename. Highest precision.\n- **Semantic-adjacent pass** — paraphrase the intent. If the exact term misses, the concept might be filed under different words.\n- **Temporal pass** — constrain to the window when the work happened (\"sessions from the week we shipped v2\").\n- **Structural pass** — look for *code blocks*, *file diffs*, or *command outputs* near the topic, not just prose. Solutions often hide in fenced blocks.\n\n`excavate.py` runs the keyword and structural passes directly; use `session_search` or a semantic tool for the semantic-adjacent pass.\n\n### 3. Score — rank the finds\n\nNot every hit is an artifact. Rank each located session against four signals (this is the scoring baked into `excavate.py`, in `--explain` mode):\n\n| Signal | What it measures | Weight |\n|---|---|---|\n| **density** | match count relative to session length | high — a session densely packed with the term is probably *about* it |\n| **recency** | newer sessions score higher (configurable) | medium — recent fixes are more likely still valid |\n| **code presence** | does the session contain runnable code/commands? | high — a fix with code beats a fix with prose |\n| **resolution markers** | phrases like \"that fixed it\", \"works now\", \"merged\" | highest — explicit success is gold |\n\nThe composite score is `density·0.25 + recency·0.15 + code·0.25 + resolution·0.35` (weights live in `excavate.py` and are tunable). Relevance is **not** raw match count — a 200-message session with one mention ranks below a 12-message session built around the topic.\n\n### 4. Extract — pull the artifact out\n\nOnce you've found the winning session, don't dump the whole transcript. Extract the **minimal artifact**:\n\n- **For a fix:** the failing state → the change → the success marker. Three quotes, nothing more.\n- **For a decision:** the options considered → the chosen option → the stated rationale.\n- **For a command:** the exact command + the one line of context that says what it does.\n- **For a rejected approach:** what was tried → the observed failure → the inferred lesson.\n\nQuote the session (`> ...`) and cite it. Extraction patterns are detailed in `references/extraction-patterns.md`.\n\n### 5. Deduplicate — collapse the finds\n\nThe same fix often appears in three sessions (the first attempt, the retry, the \"oh and also\" follow-up). Deduplicate before reporting:\n\n- **Exact-code dedup** — identical fenced blocks collapse to one.\n- **Near-dup detection** — normalized text similarity > 0.85 merges into a cluster; keep the highest-scoring member as the canonical answer.\n- **Variant detection** — if two snippets differ only in a version number or path, treat as the same artifact and note the latest variant.\n\n`excavate.py --dedup` runs all three. See `references/deduplication.md`.\n\n## Using `excavate.py`\n\nThe script lives at `scripts/excavate.py`. It has no third-party dependencies — stdlib only — so it runs anywhere Python 3.8+ does.\n\n```bash\n# Basic keyword dig over a directory of .md / .txt / .json session logs\npython3 scripts/excavate.py dig ./sessions --query \"kafka consumer rebalance\"\n\n# Multiple terms (AND'd within a session), show top 5 with per-session scores\npython3 scripts/excavate.py dig ./sessions --query \"rebalance retry backoff\" --top 5 --explain\n\n# Add a date window (ISO dates), dedup near-identical results\npython3 scripts/excavate.py dig ./sessions \\\n  --query \"connection pool exhaustion\" \\\n  --after 2024-01-01 --before 2024-06-01 \\\n  --dedup\n\n# Dump the extracted code blocks across all matching sessions\npython3 scripts/excavate.py dig ./sessions --query \"ffmpeg concatenate\" --extract code\n\n# Index a directory once, then query the index repeatedly (faster for large corpora)\npython3 scripts/excavate.py index ./sessions --out sessions.idx\npython3 scripts/excavate.py query sessions.idx --query \"oauth refresh token\" --top 3 --explain\n```\n\n`--explain` prints the per-signal score breakdown so you can see *why* a session ranked where it did. Full CLI reference: `references/cli-reference.md`.\n\n### Programmatic use\n\n```python\nfrom excavate import ArchaeologyIndex\n\nidx = ArchaeologyIndex()\nidx.scan(\"./sessions\")            # walk the directory once\nfor hit in idx.search(\"kafka rebalance\", top=5, explain=True):\n    print(hit.score, hit.path, hit.extraction)\n```\n\nThe `ArchaeologyIndex` class is the stable surface; the CLI is a thin wrapper over it.\n\n## Relevance Scoring in Depth\n\nScoring details, the math, and how to retune weights for your corpus are in `references/relevance-scoring.md`. Key points:\n\n- Scores are normalized to **[0, 1]** per signal before weighting, so a corpus change doesn't silently inflate one signal.\n- **Resolution markers dominate** by default — a session that explicitly says \"that fixed it\" beats a longer, denser session that merely mentions the term. Tune the weight down if your logs lack success markers.\n- Recency is **configurable**, not gospel. For stable domains (algorithms, math) weight it low; for fast-moving domains (frontend deps) weight it high.\n- The scoring is **transparent**, not learned. Every ranking is explainable; nothing is a black-box embedding. This matters when the user asks \"why did you trust that session?\"\n\n## Integration with `session_search`\n\nIf you're running inside Hermes, the native `session_search` tool is your semantic-adjacent pass — it has FTS5 over the session DB. Use this skill's workflow to *decide what to search for and how to rank the results*, then let `excavate.py` handle corpora that aren't in the session DB (exported logs, markdown notes, JSONL exports, another agent's transcripts).\n\n```text\nsemantic-adjacent pass  →  session_search(query=\"...\")        # Hermes session DB\nkeyword + structural    →  excavate.py dig ./exported-logs    # file-based corpora\n```\n\n## Common Pitfalls\n\n1. **Searching one query and giving up.** The single biggest failure mode. Run at least three passes (exact, semantic-adjacent, structural). Artifacts are rarely filed under the first word you reach for.\n\n2. **Trusting match count as relevance.** A session that mentions \"docker\" forty times while setting up a CI pipeline is not the answer to \"how did we fix the docker permissions bug.\" Use the composite score, not raw hits.\n\n3. **Skipping dedup and reporting three copies of the same fix.** Always run `--dedup` when `top > 1`. The user asked for the answer, not the archaeology of the answer.\n\n4. **Extracting the whole session.** The transcript is the *site*, not the *artifact*. Quote minimally.\n\n5. **Retrieving a decision without its rationale.** \"We chose Postgres\" is useless without \"because we needed strong consistency for the ledger.\" Resolution and rationale travel together — extract both or neither.\n\n6. **Assuming recency equals correctness.** A two-year-old session that solved the exact algorithmic problem beats yesterday's near-miss. Recency is a *tiebreaker*, weighted low by default for a reason.\n\n7. **Digging without a survey.** If you can't state what artifact you want and what shape it takes, your query will be too vague to rank well. Spend ten seconds on the table in Phase 1.\n\n8. **Forgetting structural pass.** The fix is often inside a fenced code block that doesn't repeat the keyword in prose. `--extract code` exists for this reason.\n\n9. **Treating near-dups as separate finds.** Three sessions with the same stack trace are one problem, not three. Normalize before you count.\n\n10. **Not citing the source session.** Always cite. The user may want to open the original; future you will want to re-excavate.\n\n## Verification Checklist\n\n- [ ] Survey written: the artifact type and query shape are stated before searching.\n- [ ] At least two query passes run (one exact/keyword, one semantic-adjacent or structural).\n- [ ] Results ranked with the composite score, not raw match count (`--explain` if using the script).\n- [ ] Deduplication run when multiple results are returned.\n- [ ] Extraction is minimal — failing state, change, success marker — not the whole transcript.\n- [ ] Source session cited (path or session id).\n- [ ] Decision artifacts include rationale, not just the chosen option.\n- [ ] Recency weighted appropriately for the domain (low for stable, high for volatile).\n\n## Further Reading\n\n- `references/search-strategies.md` — the full playbook for the Locate phase: query expansion, negation, temporal constraints, and how to pick pass order.\n- `references/relevance-scoring.md` — the math behind the composite score, normalization, and how to retune weights.\n- `references/extraction-patterns.md` — extraction templates for fix, decision, command, and rejection artifacts.\n- `references/deduplication.md` — exact, near-dup, and variant detection in depth.\n- `references/cli-reference.md` — every `excavate.py` flag and subcommand.\n- `scripts/excavate.py` — the implementation. Stdlib only, single file, importable.\n\n---\n\n*Prompt Archaeology: don't re-derive what you've already discovered. Excavate it.*\n\nFile v0.1.1:README.md\n\n# Prompt Archaeology\n\n> An AI agent skill for excavating forgotten solutions, code snippets, and decisions from past conversation sessions — instead of re-solving problems from scratch.\n\n[![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](LICENSE)\n[![Python 3.8+](https://img.shields.io/badge/python-3.8+-blue.svg)](https://www.python.org/downloads/)\n\nEvery AI session is a stratum — a sedimented layer of debugging, decision-making, and discovery. Over time, valuable artifacts sink below the surface: a one-liner that fixed a gnarly race condition, a config that satisfied a finicky build, the exact incantation that convinced a model to behave. Most agents never dig for these. **Prompt Archaeology** turns that history into a quarryable resource.\n\n## What it gives you\n\n- **Search strategies** — keyword, semantic-adjacent, temporal, and structural queries tuned for session transcripts.\n- **Relevance scoring** — a transparent, composable ranking that surfaces the one session that actually matters (not just the one that mentions the term most).\n- **Knowledge extraction patterns** — recipes for pulling *decisions* and *solutions* out of a wall of chat, not just matching text.\n- **Deduplication** — collapse near-duplicate fixes across sessions into a single canonical answer.\n- **`excavate.py`** — a standalone Python script that crawls session logs and markdown files, ranks them, and prints the buried artifacts. **Zero third-party dependencies** (stdlib only).\n\n## The workflow in one breath\n\n**Survey** what artifact you want → **Locate** it with multiple query passes → **Score** the finds with the composite ranker → **Extract** the minimal artifact (with citation) → **Deduplicate** before reporting.\n\n## Quick start\n\n```bash\ngit clone https://github.com/voronindenis5/prompt-archaeology.git\ncd prompt-archaeology\n\n# Basic keyword dig over a directory of session logs (.md/.txt/.json/.jsonl)\npython3 scripts/excavate.py dig ./my-sessions --query \"kafka consumer rebalance\"\n\n# Multiple terms, top 5, with per-signal score breakdown\npython3 scripts/excavate.py dig ./my-sessions --query \"rebalance retry backoff\" --top 5 --explain\n\n# Date window + dedup near-identical results\npython3 scripts/excavate.py dig ./my-sessions \\\n  --query \"connection pool exhaustion\" \\\n  --after 2024-01-01 --before 2024-06-01 --dedup\n\n# Dump just the extracted code blocks across matches\npython3 scripts/excavate.py dig ./my-sessions --query \"ffmpeg concatenate\" --extract code\n```\n\n### Index once, query many\n\nFor large corpora, build an index and query it repeatedly:\n\n```bash\npython3 scripts/excavate.py index ./my-sessions --out sessions.idx\npython3 scripts/excavate.py query sessions.idx --query \"oauth refresh token\" --top 3 --explain\n```\n\n### Programmatic use\n\n```python\nfrom excavate import ArchaeologyIndex\n\nidx = ArchaeologyIndex()\nidx.scan(\"./my-sessions\")\nfor hit in idx.search(\"kafka rebalance\", top=5, explain=True):\n    print(hit.score, hit.path)\n    print(hit.extraction)\n```\n\n## Why \"archaeology\"?\n\nAn archaeologist does not grep the desert for \"pottery\" and ship the first hit. They survey, triangulate, carefully extract, and catalog. This skill teaches the agent to do the same with its own past — because re-deriving a solution you already found is the most expensive way to be wrong.\n\n## Repository layout\n\n```\nprompt-archaeology/\n├── SKILL.md                          # The skill itself (Hermes/OpenClaw format)\n├── README.md                         # You are here\n├── LICENSE                           # MIT\n├── references/\n│   ├── search-strategies.md          # Locate-phase playbook\n│   ├── relevance-scoring.md          # The math + how to retune weights\n│   ├── extraction-patterns.md        # Fix/decision/command/rejection templates\n│   ├── deduplication.md              # Exact, near-dup, and variant detection\n│   └── cli-reference.md              # Every excavate.py flag\n└── scripts/\n    └── excavate.py                   # Stdlib-only searcher + ranker\n```\n\n## Reference docs\n\n| Doc | What it covers |\n|---|---|\n| [`search-strategies.md`](references/search-strategies.md) | Query expansion, negation, temporal constraints, pass ordering |\n| [`relevance-scoring.md`](references/relevance-scoring.md) | Composite score math, normalization, weight tuning |\n| [`extraction-patterns.md`](references/extraction-patterns.md) | Minimal-artifact templates by artifact type |\n| [`deduplication.md`](references/deduplication.md) | Exact / near-dup / variant detection in depth |\n| [`cli-reference.md`](references/cli-reference.md) | Full `excavate.py` CLI reference |\n\n## Requirements\n\n- Python **3.8+** (uses only the standard library — no `pip install` needed)\n\n## License\n\nMIT © Denis Voronin\n\nFile v0.1.1:_meta.json\n\n{\n  \"ownerId\": \"kn75wwn4x6djaf28jbykeamazd81gtdp\",\n  \"slug\": \"prompt-archaeology\",\n  \"version\": \"0.1.1\",\n  \"publishedAt\": 1786449520748\n}\n\nFile v0.1.1:references/cli-reference.md\n\n# `excavate.py` CLI Reference\n\n`excavate.py` is a stdlib-only Python script for excavating session logs and markdown files. It supports three subcommands: `dig`, `index`, and `query`.\n\n## Global behavior\n\n- **No third-party dependencies.** Runs on Python 3.8+ with the standard library only.\n- **Reads** `.md`, `.txt`, `.json`, and `.jsonl` files (see [File formats](#file-formats)).\n- **Writes** nothing unless `--out` is given (index subcommand).\n- **Exits non-zero** on argument errors; zero on successful search (even with no hits).\n\n## Subcommands\n\n### `dig` — search a directory directly\n\n```bash\npython3 scripts/excavate.py dig <directory> --query <terms> [options]\n```\n\nScans `<directory>` recursively, scores every file against the query, prints the top matches. Use `dig` for one-off searches; use `index` + `query` for repeated searches over the same corpus.\n\n**Options:**\n\n| Flag | Type | Default | Description |\n|---|---|---|---|\n| `--query`, `-q` | str (required) | — | Search terms. Multiple terms are AND'd within a file (all must appear). Use `--query \"a b c\"` or repeat `--query` for OR semantics is not supported; pass a single string. |\n| `--top`, `-n` | int | 5 | Number of top results to print. |\n| `--after` | date (ISO) | — | Only include files modified on/after this date (`YYYY-MM-DD`). |\n| `--before` | date (ISO) | — | Only include files modified on/before this date (`YYYY-MM-DD`). |\n| `--not` | str | — | Exclude files containing this term. Repeatable. |\n| `--explain` | flag | off | Print per-signal score breakdown for each result. |\n| `--extract` | `code` \\| `all` \\| `none` | `none` | Print extracted code blocks (`code`), full extraction (`all`), or just scores (`none`). |\n| `--dedup` | flag | off | Collapse near-duplicate results into clusters. |\n| `--dedup-threshold` | float | 0.85 | Jaccard threshold for near-dup (0–1). |\n| `--keep-variants` | flag | off | With `--dedup`, report all version/path variants instead of collapsing. |\n\n**Examples:**\n\n```bash\n# Basic dig\npython3 scripts/excavate.py dig ./sessions --query \"kafka rebalance\"\n\n# Top 10 with score breakdown\npython3 scripts/excavate.py dig ./sessions --query \"pool exhaustion\" --top 10 --explain\n\n# Date window, exclude sidekiq mentions, dedup\npython3 scripts/excavate.py dig ./sessions \\\n  --query \"redis cache\" \\\n  --after 2024-01-01 --before 2024-06-30 \\\n  --not sidekiq \\\n  --dedup\n\n# Extract code blocks only\npython3 scripts/excavate.py dig ./sessions --query \"ffmpeg concat\" --extract code\n```\n\n### `index` — build a reusable index\n\n```bash\npython3 scripts/excavate.py index <directory> --out <index-file> [options]\n```\n\nScans `<directory>` once and serializes the `ArchaeologyIndex` to `<index-file>` (pickle format). Subsequent `query` calls load the index instead of re-scanning.\n\n**Options:**\n\n| Flag | Type | Default | Description |\n|---|---|---|---|\n| `--out`, `-o` | path (required) | — | Output index file path. |\n| `--after` / `--before` | date | — | Pre-filter by mtime at index time (cannot be relaxed later). |\n\n**Example:**\n\n```bash\npython3 scripts/excavate.py index ./sessions --out sessions.idx\n```\n\n### `query` — search an existing index\n\n```bash\npython3 scripts/excavate.py query <index-file> --query <terms> [options]\n```\n\nLoads `<index-file>` and runs a search. Accepts the same options as `dig` (`--top`, `--not`, `--explain`, `--extract`, `--dedup`, etc.), except date filters (those were applied at index time).\n\n**Example:**\n\n```bash\npython3 scripts/excavate.py query sessions.idx --query \"oauth refresh\" --top 3 --explain --dedup\n```\n\n## File formats\n\n`excavate.py` auto-detects format by extension:\n\n| Extension | Parsing |\n|---|---|\n| `.md`, `.markdown` | Raw text. Honors YAML frontmatter (between `---` fences) for date extraction. Fenced code blocks (``` ``` ```) extracted for structural pass. |\n| `.txt` | Raw text. No frontmatter support. |\n| `.json` | Expected to be a single object or array of message objects with `role` and `content` fields. Content fields are concatenated. |\n| `.jsonl` | Each line is a JSON object with `role` and `content`. Concatenated in order. |\n\nFor `.json` / `.jsonl`, the script looks for common field names: `content`, `text`, `message`, `body`. If your format differs, pre-process to `.md` or `.txt`.\n\n## Scoring flags\n\nSee `references/relevance-scoring.md` for the scoring model. The weights are constants in the script; retune by editing:\n\n```python\nWEIGHT_DENSITY    = 0.25\nWEIGHT_RECENCY    = 0.15\nWEIGHT_CODE       = 0.25\nWEIGHT_RESOLUTION = 0.35\n```\n\n## Programmatic API\n\n```python\nfrom excavate import ArchaeologyIndex, SearchHit\n\nidx = ArchaeologyIndex()\nidx.scan(\"./sessions\")                    # walk directory, parse files\nidx.save(\"sessions.idx\")                  # serialize\nidx2 = ArchaeologyIndex.load(\"sessions.idx\")\n\nhits: list[SearchHit] = idx.search(\n    query=\"kafka rebalance\",\n    top=5,\n    explain=True,                         # populate hit.explanation\n    exclude=(\"sidekiq\",),                 # --not terms\n)\n\nfor hit in hits:\n    print(hit.path, hit.score, hit.explanation)\n    print(hit.extraction)                 # extracted code/text\n```\n\n### `SearchHit` fields\n\n| Field | Type | Description |\n|---|---|---|\n| `path` | str | File path. |\n| `score` | float | Composite score in [0, 1]. |\n| `explanation` | dict \\| None | Per-signal breakdown (populated when `explain=True`). |\n| `extraction` | str | Extracted artifact (code block or relevant excerpt). |\n| `matches` | list[str] | Matched line snippets. |\n| `mtime` | float | File modification time. |\n\n## Exit codes\n\n| Code | Meaning |\n|---|---|\n| 0 | Search completed (including no hits). |\n| 2 | Argument error (missing required flag, bad value). |\n| 3 | I/O error (directory not found, unreadable file). |\n\n## Performance\n\n- **Scan speed:** ~1,000 files/sec on markdown, ~3,000/sec on plain text (single-threaded, SSD).\n- **Index size:** roughly 30–50% of the source corpus (pickle-serialized parsed text + metadata).\n- **Query speed:** <100ms for corpora under 10k files (in-memory).\n- **Memory:** proportional to corpus size; the index holds parsed text in memory for fast repeated queries.\n\nFor corpora above ~50k files, prefer `session_search` (FTS5-backed) over `excavate.py`.\n\nFile v0.1.1:references/deduplication.md\n\n# Deduplication\n\nThe same fix often appears in three sessions: the first attempt, the retry, and the \"oh and also\" follow-up. Deduplication collapses these into a single canonical answer before you report.\n\n`excavate.py --dedup` runs three layers of dedup in order: exact → near-dup → variant.\n\n## Layer 1: Exact dedup\n\n**Rule:** identical fenced code blocks collapse to one.\n\nTwo sessions with byte-identical ` ```bash ... ``` ` blocks are the same artifact. Keep the higher-scoring session as the canonical instance; drop the others from the result set but note them as duplicates.\n\n```\nsessions/fix-v1.md        score 0.71    [canonical]\n  ≡ sessions/fix-v1-retry.md             (exact dup of fix-v1.md)\n  ≡ sessions/fix-v1-followup.md          (exact dup of fix-v1.md)\n```\n\nExact dedup is cheap (a hash) and catches the easy cases.\n\n## Layer 2: Near-dup detection\n\n**Rule:** normalized text similarity > 0.85 merges into a cluster.\n\nTwo sessions that describe the same fix in slightly different words are the same artifact. Near-dup uses **normalized token overlap**:\n\n1. Lowercase, strip punctuation, remove stopwords.\n2. Tokenize on whitespace.\n3. Compute Jaccard similarity: `|A ∩ B| / |A ∪ B|`.\n4. If Jaccard > 0.85, merge.\n\nKeep the highest-scoring session in the cluster as canonical.\n\n```\nsessions/2024-03-12-a.md  score 0.82    [canonical]\n  ≈ sessions/2024-03-13-b.md            (near-dup, Jaccard 0.91)\n  ≈ sessions/2024-03-14-c.md            (near-dup, Jaccard 0.88)\n```\n\n### Tuning the threshold\n\n- **0.85 (default):** conservative. Only clearly-the-same fixes merge.\n- **0.75:** aggressive. Catches more dups but risks merging distinct fixes that share boilerplate (e.g., two different nginx fixes with similar config structure).\n- **0.95:** paranoid. Almost only exact dups merge.\n\nSet via `--dedup-threshold` on the CLI, or `DEDUP_THRESHOLD` in code.\n\n### What gets normalized away\n\n- Case (`Error` ≡ `error`)\n- Punctuation (`failed.` ≡ `failed`)\n- Stopwords (`the connection pool` ≡ `connection pool`)\n- Whitespace runs\n\n### What does NOT get normalized away\n\n- Numbers and identifiers (`pool_size=20` ≢ `pool_size=50`)\n- Code structure (two fenced blocks with different commands stay distinct)\n- File paths\n\n## Layer 3: Variant detection\n\n**Rule:** if two snippets differ only in a version number, path, or date, treat as the same artifact and note the latest variant.\n\n```\nsessions/2023-09-01.md:  pip install fastapi==0.68.0\nsessions/2024-03-12.md:  pip install fastapi==0.110.0\n```\n\nThese are the same artifact (\"install fastapi\") with a version variant. Collapse to one, note the latest (0.110.0).\n\nVariant detection works by:\n\n1. Masking version-like tokens (`\\d+\\.\\d+\\.\\d++`), date-like tokens, and path-like tokens.\n2. Re-running near-dup (Layer 2) on the masked text.\n3. If they merge, they're variants.\n\n### When variants matter\n\nSometimes the variant *is* the artifact — e.g., \"the exact version that worked with Python 3.9.\" Variant detection has a `--keep-variants` flag that reports all variants instead of collapsing, for these cases.\n\n## Output format\n\nWith `--dedup`, results print as clusters:\n\n```\n=== Cluster 1 (3 sessions, canonical score 0.82) ===\n  [canonical] sessions/2024-03-12-a.md        score 0.82\n  [dup]       sessions/2024-03-13-b.md        near-dup (Jaccard 0.91)\n  [dup]       sessions/2024-03-14-c.md        near-dup (Jaccard 0.88)\n\n=== Cluster 2 (1 session, canonical score 0.64) ===\n  [canonical] sessions/2024-02-20-x.md        score 0.64\n```\n\nReport only the canonical session per cluster to the user; mention duplicate counts inline if relevant (\"found in 3 sessions\").\n\n## Anti-patterns\n\n- **Dedup before scoring.** Always score first, then dedup — the highest-scoring session must be the canonical one.\n- **Aggressive near-dup on small corpora.** With <50 sessions, a 0.75 threshold will over-merge. Stick to 0.85+.\n- **Collapsing variants when versions matter.** If the user asked \"which version worked?\", `--keep-variants` is mandatory.\n- **Reporting every member of a cluster.** The user wants the answer, not the cluster graph. Report canonical; mention dup count.\n- **Dedup across different artifact types.** A fix and a rejection of the same approach are *different artifacts* even if textually similar. Dedup within artifact type.\n\nFile v0.1.1:references/extraction-patterns.md\n\n# Extraction Patterns\n\nThe transcript is the **site**, not the **artifact**. Once you've located and scored the winning session, extract the minimal artifact — never dump the whole transcript. This doc gives templates for each artifact type.\n\n## Universal extraction rules\n\n1. **Quote, don't paraphrase.** Use `> ...` blockquotes. The artifact must be the session's words, not your reconstruction.\n2. **Three quotes max** (failing state, change, success marker) unless the artifact is inherently longer (a config file).\n3. **Cite the session** — path or session id — on every extraction.\n4. **Drop the journey.** The eight wrong turns before the fix are not the artifact. They may be relevant for a *rejection* extraction (below), but not for a *fix*.\n\n## 1. Fix extraction\n\nThe most common artifact. Three quotes: failing state → change → success marker.\n\n**Template:**\n\n```markdown\n> **Failing:** `ConnectionPool exceeded max_connections (50)`\n>\n> **Change:** set `SQLALCHEMY_POOL_SIZE=20` and `SQLALCHEMY_MAX_OVERFLOW=5` in `.env`\n>\n> **Resolved:** \"spun up the app, hit it with 200 concurrent requests, pool stable — that fixed it\"\n\n— `sessions/2024-03-12-db-pool.md`\n```\n\n**What to drop:** the forty messages of hypothesis-testing before the config change. They're noise once the fix is known.\n\n## 2. Decision extraction\n\nTwo flavors: a decision *with* rationale is valuable; a decision *without* rationale is nearly useless. Always extract both.\n\n**Template:**\n\n```markdown\n> **Options considered:** Postgres vs. MongoDB for the ledger\n>\n> **Chose:** Postgres\n>\n> **Rationale:** \"we need strong consistency for the ledger — eventual consistency\n> would mean we could double-spend on a network partition, and that's a\n> correctness bug, not a performance bug\"\n\n— `sessions/2024-01-08-db-choice.md`\n```\n\n**If the rationale isn't in the session,** say so explicitly:\n\n```markdown\n> **Chose:** Postgres (rationale not recorded in this session; see sessions/2024-01-09*.md for follow-up)\n```\n\nDon't invent a rationale to fill the gap.\n\n## 3. Command / incantation extraction\n\nThe exact command plus one line of context. Nothing else.\n\n**Template:**\n\n```markdown\n> Concatenate two MP4s without re-encoding:\n>\n> ```bash\n> ffmpeg -i \"concat:in1.mp4|in2.mp4\" -c copy out.mp4\n> ```\n\n— `sessions/2024-05-30-video.md`\n```\n\nIf the command has prerequisites (a `dep install`, an env var), include them in the same block. If the session showed a *wrong* invocation first, include it only as a `# NOT this:` comment.\n\n## 4. Rejected-approach extraction\n\nOften more valuable than the fix — it saves you from re-walking a dead end.\n\n**Template:**\n\n```markdown\n> **Tried:** rewriting the parser as a single regex\n>\n> **Failed:** \"catastrophic backtracking on inputs > 4KB, CPU pinned at 100%\"\n>\n> **Lesson:** nested quantifiers on unbounded input; stick with the recursive-descent parser\n\n— `sessions/2024-02-14-parser.md`\n```\n\n**When to extract a rejection:** the session explicitly records *what was tried*, *how it failed*, and (ideally) *why*. If only the failure is recorded with no diagnosis, extract the failure and mark the lesson as inferred.\n\n## 5. Config / architecture extraction\n\nFor durable artifacts (config files, architecture diagrams described in prose). Quote the whole relevant block, not a snippet.\n\n**Template:**\n\n```markdown\n> nginx config that fixed the 502 on long-polling:\n>\n> ```nginx\n> location /events {\n>     proxy_pass http://backend;\n>     proxy_read_timeout 3600s;\n>     proxy_buffering off;\n> }\n> ```\n\n— `sessions/2024-04-01-nginx.md`\n```\n\n## Anti-patterns\n\n- **The whole transcript.** The user asked for the fix, not the dig diary.\n- **Paraphrased fix.** \"We changed the pool size\" is useless; `SQLALCHEMY_POOL_SIZE=20` is the artifact.\n- **Decision without rationale.** \"We chose Postgres\" answers nothing. Always pair with the *why*.\n- **Fix without success marker.** If the session never confirmed the fix worked, say so: \"applied but not confirmed in this session.\"\n- **Extracting the first mention.** The first mention of a term is often the *problem statement*, not the fix. Score and read before extracting.\n- **Inventing rationale.** If the session doesn't say why, don't fill it in. Mark it missing.\n\nFile v0.1.1:references/relevance-scoring.md\n\n# Relevance Scoring in Depth\n\n`excavate.py` ranks sessions with a **transparent, composite score** — not a black-box embedding. Every ranking is explainable. This doc covers the math, normalization, and how to retune for your corpus.\n\n## The four signals\n\n| Signal | What it measures | Default weight | Range |\n|---|---|---|---|\n| `density` | match count relative to session length | 0.25 | [0, 1] |\n| `recency` | newer sessions score higher | 0.15 | [0, 1] |\n| `code` | does the session contain runnable code/commands? | 0.25 | [0, 1] |\n| `resolution` | success markers like \"that fixed it\", \"works now\", \"merged\" | 0.35 | [0, 1] |\n\nWeights sum to 1.0. Composite score is the weighted sum, also in [0, 1].\n\n## Why these weights\n\n- **Resolution dominates (0.35).** A session that explicitly says \"that fixed it\" is the strongest possible signal. Prose that merely mentions the term is weak evidence by comparison.\n- **Density and code tie (0.25).** A session densely packed with the term is probably *about* it; a session with code is probably *solving* it. Both matter; neither alone is decisive.\n- **Recency is the tiebreaker (0.15).** Useful, but not authoritative — a two-year-old fix to an algorithmic problem is still correct.\n\n## Normalization\n\nEach signal is normalized to **[0, 1]** before weighting, so a corpus change doesn't silently inflate one signal.\n\n### density\n\n```\ndensity(session) = match_count / max(match_count across corpus)\n```\n\nA session with 12 matches ranks at density 1.0 *only if* 12 is the max in the corpus; if another session has 50 matches, the 12-match session ranks 0.24. This is why raw match count is a bad proxy — density is relative.\n\nThe implementation also applies a **log saturation** so that one very long session doesn't dominate:\n\n```\ndensity(session) = log(1 + match_count) / log(1 + max_match_count)\n```\n\n### recency\n\n```\nrecency(session) = (session_mtime - corpus_min_time) / (corpus_max_time - corpus_min_time)\n```\n\nLinear interpolation between the oldest and newest session in the corpus. If the corpus spans 2022–2025, a mid-2024 session scores ~0.6.\n\n### code\n\n```\ncode(session) = 1.0 if session contains ≥1 fenced code block or shell command\n              = 0.5 if session contains inline `code` only\n              = 0.0 otherwise\n```\n\nBinary-ish. A session with code is qualitatively different from one without.\n\n### resolution\n\n```\nresolution(session) = (count of resolution markers) / max(count of resolution markers across corpus)\n```\n\nCapped at 1.0. Resolution markers (see below) are weighted by type.\n\n## Resolution markers\n\nThe marker lexicon lives in `excavate.py` as `RESOLUTION_MARKERS`. It groups phrases by strength:\n\n| Strength | Examples |\n|---|---|\n| Strong (×1.0) | \"that fixed it\", \"works now\", \"merged\", \"deployed\", \"shipped\" |\n| Medium (×0.7) | \"fixed\", \"resolved\", \"solved\", \"working\" |\n| Weak (×0.4) | \"seems to work\", \"might be it\", \"I think that's it\" |\n\nA session with one strong marker outscores a session with three weak ones.\n\n## Retuning weights\n\nWeights are constants at the top of `excavate.py`:\n\n```python\nWEIGHT_DENSITY    = 0.25\nWEIGHT_RECENCY    = 0.15\nWEIGHT_CODE       = 0.25\nWEIGHT_RESOLUTION = 0.35\n```\n\n### When to retune\n\n| Your corpus | Change |\n|---|---|\n| Fast-moving domain (frontend deps, infra) | Bump `RECENCY` to 0.25, drop `DENSITY` to 0.15 |\n| Stable domain (algorithms, math, CS theory) | Drop `RECENCY` to 0.05, bump `RESOLUTION` to 0.45 |\n| Mostly code (debugging transcripts) | Bump `CODE` to 0.35, drop `DENSITY` to 0.15 |\n| Prose-heavy (design discussions, few markers) | Drop `RESOLUTION` to 0.20, bump `DENSITY` to 0.40 |\n| Logs lack success markers entirely | Set `RESOLUTION` to 0.0 — it'll only add noise |\n\n### How to retune\n\nEdit the constants, then validate on a small labeled set (sessions where you *know* the right answer). The top-1 hit rate should climb as you tune toward your corpus's character.\n\n## Why not embeddings?\n\nEmbeddings would raise recall on the semantic-adjacent pass, but:\n\n1. **They're opaque.** The user asks \"why did you trust that session?\" and you can't answer. The composite score answers in one line.\n2. **They need a model.** `excavate.py` is stdlib-only and runs anywhere. Adding a model breaks that.\n3. **They need a corpus big enough to matter.** For most personal session corpora (hundreds to low thousands of sessions), keyword + structural + the composite ranker is competitive with embedding search and far cheaper.\n\nIf you have a large corpus and want semantic recall, run `session_search` (FTS5) or an embedding index for the semantic pass, then feed those candidates into `excavate.py`'s scorer. The signals compose.\n\n## The `--explain` output\n\n`excavate.py --explain` prints, per result:\n\n```\nsessions/2024-03-12-kafka-fix.md\n  score: 0.82\n    density    0.71   (12 matches, log-saturated)\n    recency    0.64   (2024-03-12, within corpus range)\n    code       1.00   (3 fenced blocks)\n    resolution 1.00   (1 strong marker: \"that fixed it\")\n```\n\nThis is the explainability contract. If a ranking looks wrong, `--explain` shows you exactly which signal is off and whether to retune.\n\nFile v0.1.1:references/search-strategies.md\n\n# Search Strategies (the Locate Phase)\n\nThe Locate phase runs **multiple query passes**, because artifacts are rarely filed under the first word you reach for. This doc is the full playbook.\n\n## Pass order\n\nRun passes in this order. Each later pass only runs if the earlier ones didn't surface a high-confidence hit.\n\n1. **Exact / keyword** — highest precision, lowest recall. The literal error string, function name, filename, or config key.\n2. **Semantic-adjacent** — paraphrase the intent. If the exact term misses, the concept may be filed under different words.\n3. **Structural** — look for *code blocks*, *diffs*, or *command outputs* near the topic, not just prose. Solutions hide inside fenced blocks.\n4. **Temporal** — constrain to the window when the work happened, then re-run keyword. Narrows a noisy corpus to the relevant stratum.\n\n## 1. Exact / keyword\n\nUse the **literal token** the artifact would contain:\n\n- An exception → the exact exception class and message fragment.\n- A function → its name.\n- A config value → the key name.\n- A CLI failure → the exact stderr fragment.\n\n```\n# exact, high-precision\nquery: \"connectionpool value too many connections\"\nquery: \"AttributeError: 'NoneType' object has no attribute 'split'\"\n```\n\n**Pitfall:** exact pass misses when the user paraphrased the error in their message but the stack trace was elided. Always follow with semantic-adjacent.\n\n## 2. Semantic-adjacent\n\nParaphrase the **intent**, not the token. Think: \"if I didn't remember the exact word, how would I describe this?\"\n\n| Concept | Exact miss | Semantic seed |\n|---|---|---|\n| OOM kill | `\"OOMKilled\"` | `\"memory limit pod restarted\"` |\n| Race condition | `\"concurrent modification\"` | `\"intermittent wrong order flaky\"` |\n| Cert expiry | `\"x509: certificate has expired\"` | `\"tls handshake failed clock skew\"` |\n| Port conflict | `\"Address already in use\"` | `\"cannot bind port already listening\"` |\n\n`excavate.py` does keyword matching; for the semantic pass use `session_search` (FTS5) or any embedding search over the corpus.\n\n## 3. Structural (code-block) pass\n\nThe fix is often inside a fenced code block whose prose doesn't repeat the keyword. `excavate.py --extract code` pulls every fenced block from matching sessions and re-ranks by proximity to the query terms.\n\nWhen to prioritize this pass:\n\n- You remember the *shape* of the answer (a shell one-liner, a YAML snippet) but not the words around it.\n- The keyword appears only in code comments or command output, never in prose.\n- The session is a wall of debugging chat with the fix buried in one block.\n\n## 4. Temporal pass\n\nConstrain to a date window, then re-run keyword. Two modes:\n\n- **`--after` / `--before`** on `excavate.py` filters by file mtime or, if the session has frontmatter, the session date.\n- **Relative** (\"sessions from the week we shipped v2\") — convert to absolute dates first.\n\nTemporal narrowing is how you separate the *original* fix from the *five times someone re-asked about it later*. The earliest high-scoring session in the window is usually the source.\n\n## Query expansion\n\nWhen keyword + semantic both miss, expand the query:\n\n- **Synonyms:** `rebalance` → `rebalancing`, `rebalance`, `coordinator`, `consumer group`.\n- **Hyponyms:** `kafka` → `producer`, `consumer`, `broker`, `topic`, `partition`.\n- **Surrounding nouns:** the files, services, or libraries adjacent to the problem.\n\nExpansion trades precision for recall. Run it last, and require a resolution marker (Phase 3 signal) to trust any expanded hit.\n\n## Negation\n\nTo find a session that discusses X but **not** Y (e.g., \"redis caching\" but not \"sidekiq\"):\n\n```bash\npython3 scripts/excavate.py dig ./sessions --query \"redis cache\" --not \"sidekiq\"\n```\n\nNegation is useful for disambiguating overloaded terms (e.g., \"migration\" the DB step vs. \"migration\" the framework).\n\n## Picking the number of passes\n\n| Confidence after pass N | Action |\n|---|---|\n| Pass 1 returns a session with a resolution marker | Stop. Extract. |\n| Pass 1 returns dense hits but no resolution marker | Run pass 2 to confirm. |\n| Pass 1 misses or is sparse | Run pass 2, then 3. |\n| Passes 1–3 all miss | Run pass 4 (temporal), then query expansion. |\n| All passes miss | The artifact may not exist. Say so — don't fabricate. |\n\n## Anti-patterns\n\n- **Single-query dig.** One search, one answer. This is grep, not archaeology. Always run ≥ 2 passes unless pass 1 returns a resolution.\n- **Keyword worship.** Refusing to paraphrase because \"the error message is the error message.\" The error message may not be in the transcript.\n- **Ignoring structure.** Reading only prose and missing the fenced block that contains the actual fix.\n- **No temporal filter on noisy corpora.** A term that appears in 200 sessions needs narrowing; use the date window.\n- **Treating expansion hits as gospel.** Expanded queries have low precision. Require a resolution marker before trusting.\n\nFile v0.1.1:skill-card.md\n\n## Description:\n\nExcavate forgotten solutions, code snippets, and decisions from past conversation sessions when the user is re-solving known problems, hunting for lost snippets, or mining session history instead of starting from scratch.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[voronindenis5](https://clawhub.ai/user/voronindenis5)\n\n### License/Terms of Use:\n\nMIT\n\n## Use Case:\n\nDevelopers and agent users use this skill to search authorized prior session logs, rank likely matches, and extract minimal cited artifacts such as fixes, decisions, commands, and rejected approaches. It is also useful for deduplicating repeated findings before reporting canonical answers.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The skill searches private session history and may expose sensitive transcript content.\n\nMitigation: Run it only against explicitly authorized session or export directories; avoid broad home directories, shared profiles, credential folders, and other users' data.\n\nRisk: Generated index files persist copied transcript contents.\n\nMitigation: Treat .idx files as sensitive local artifacts; do not share or commit them, and delete them when they are no longer needed.\n\nRisk: Existing index files are loaded with unsafe pickle deserialization.\n\nMitigation: Only load index files that were generated locally and are trusted; do not load third-party .idx files unless the implementation is changed away from pickle.\n\n## Reference(s):\n\n- [Server-resolved source repository](https://github.com/voronindenis5/prompt-archaeology)\n- [ClawHub skill page](https://clawhub.ai/voronindenis5/skills/prompt-archaeology)\n- [Search Strategies](references/search-strategies.md)\n- [Relevance Scoring](references/relevance-scoring.md)\n- [Extraction Patterns](references/extraction-patterns.md)\n- [Deduplication](references/deduplication.md)\n- [CLI Reference](references/cli-reference.md)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, code, shell commands, guidance]\n\n**Output Format:** [Markdown guidance with cited excerpts and optional shell or Python commands; the bundled script prints ranked text results and extracted artifacts.]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [The index subcommand can create local .idx index files when explicitly invoked with --out.]\n\n## Skill Version(s):\n\n0.1.1 (source: server release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v0.1.1:LICENSE\n\nMIT License\n\nCopyright (c) 2026 Denis Voronin\n\nPermission is hereby granted, free of charge, to any person obtaining a copy\nof this software and associated documentation files (the \"Software\"), to deal\nin the Software without restriction, including without limitation the rights\nto use, copy, modify, merge, publish, distribute, sublicense, and/or sell\ncopies of the Software, and to permit persons to whom the Software is\nfurnished to do so, subject to the following conditions:\n\nThe above copyright notice and this permission notice shall be included in all\ncopies or substantial portions of the Software.\n\nTHE SOFTWARE IS PROVIDED \"AS IS\", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR\nIMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,\nFITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE\nAUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER\nLIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,\nOUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE\nSOFTWARE.\n\nArchive v0.1.0: 13 files, 30197 bytes\n\nFiles: LICENSE (1070b), README.md (4838b), references (0b), references/cli-reference.md (6262b), references/deduplication.md (4328b), references/extraction-patterns.md (4257b), references/relevance-scoring.md (5157b), references/search-strategies.md (4953b), scripts (0b), scripts/excavate.py (22802b), skill-card.md (2876b), SKILL.md (13010b), _meta.json (137b)\n\nFile v0.1.0:SKILL.md\n\n---\nname: prompt-archaeology\ndescription: \"Excavate forgotten solutions, code snippets, and decisions from past conversation sessions. Use when the user is re-solving a problem you've likely solved before, hunting for a lost snippet, or wants to mine session history for buried knowledge instead of starting from scratch.\"\nversion: 1.0.0\nauthor: Denis Voronin\nlicense: MIT\nmetadata:\n  hermes:\n    tags: [sessions, history, search, knowledge-mining, archaeology, recovery, hermes-agent]\n    related_skills: []\n---\n\n# Prompt Archaeology\n\n## Overview\n\n**Prompt Archaeology** is the practice of excavating your own conversation history instead of re-solving problems from scratch. Every AI session is a stratum — a sedimented layer of debugging, decision-making, and discovery. Over time, valuable artifacts sink below the surface: a one-liner that fixed a gnarly race condition, a config that satisfied a finicky build, the exact incantation that convinced a model to behave. Most agents never dig for these. They re-derive, re-guess, and re-fail.\n\nThis skill turns that history into a quarryable resource. It bundles:\n\n- **Search strategies** — keyword, semantic-adjacent, temporal, and structural queries tuned for session transcripts.\n- **Relevance scoring** — a transparent, composable ranking that surfaces the one session that actually matters.\n- **Knowledge extraction patterns** — recipes for pulling *decisions* and *solutions* out of a wall of chat, not just matching text.\n- **Deduplication** — collapse near-duplicate fixes across sessions into a single canonical answer.\n- **`excavate.py`** — a standalone Python script that crawls session logs and markdown files, ranks them, and prints the buried artifacts.\n\nThe metaphor is deliberate. An archaeologist does not grep the desert for \"pottery\" and ship the first hit. They survey, triangulate, carefully extract, and catalog. This skill teaches the agent to do the same with its own past.\n\n## When to Use\n\n- **The user is about to re-solve a known problem.** They describe a bug or task and you have a flicker of \"we've done this before.\" Excavate before answering.\n- **\"Didn't we figure out...?\" / \"What did we land on?\"** — retrieve the prior decision and its rationale, not just the outcome.\n- **Hunting for a lost code snippet, config value, or command** that worked months ago.\n- **Onboarding to a codebase you've touched before** — pull the architectural decisions out of old sessions.\n- **Avoiding repeated dead ends** — find the approaches that were *rejected* and why, so you don't walk back into them.\n- **Writing postmortems or ADRs** from scattered session evidence.\n\n### Don't use for\n\n- Fresh problems with no prior history — there's nothing to excavate; solve forward.\n- When you already hold the answer in active context — don't pad the turn with a search.\n- Sensitive retrieval across other users' private profiles unless explicitly authorized.\n\n## The Excavation Workflow\n\nA dig has five phases. Skipping any phase degrades result quality.\n\n### 1. Survey — frame the query\n\nBefore searching, state **what artifact you want** and **what shape it takes**:\n\n| Artifact you want | Query shape | Example seeds |\n|---|---|---|\n| A fix for a bug | error string + symptom words | the exception text, \"traceback\", the failing assertion |\n| A decision + rationale | the option names + \"decided\" / \"chose\" / \"went with\" | the two libraries you were weighing |\n| A config value | the key name + surrounding file | `\"max_connections\"`, `nginx.conf` |\n| A rejected approach | the approach + \"didn't work\" / \"gave up\" / \"abandoned\" | the tool you tried first |\n| A command / incantation | the tool + the goal verb | `ffmpeg`, \"concatenate\" |\n\nWrite the query down. A vague survey yields a vague dig.\n\n### 2. Locate — run the searches\n\nRun **multiple passes**, not one. Different phrasings live in different sessions.\n\n- **Exact / keyword pass** — the literal error string, function name, or filename. Highest precision.\n- **Semantic-adjacent pass** — paraphrase the intent. If the exact term misses, the concept might be filed under different words.\n- **Temporal pass** — constrain to the window when the work happened (\"sessions from the week we shipped v2\").\n- **Structural pass** — look for *code blocks*, *file diffs*, or *command outputs* near the topic, not just prose. Solutions often hide in fenced blocks.\n\n`excavate.py` runs the keyword and structural passes directly; use `session_search` or a semantic tool for the semantic-adjacent pass.\n\n### 3. Score — rank the finds\n\nNot every hit is an artifact. Rank each located session against four signals (this is the scoring baked into `excavate.py`, in `--explain` mode):\n\n| Signal | What it measures | Weight |\n|---|---|---|\n| **density** | match count relative to session length | high — a session densely packed with the term is probably *about* it |\n| **recency** | newer sessions score higher (configurable) | medium — recent fixes are more likely still valid |\n| **code presence** | does the session contain runnable code/commands? | high — a fix with code beats a fix with prose |\n| **resolution markers** | phrases like \"that fixed it\", \"works now\", \"merged\" | highest — explicit success is gold |\n\nThe composite score is `density·0.25 + recency·0.15 + code·0.25 + resolution·0.35` (weights live in `excavate.py` and are tunable). Relevance is **not** raw match count — a 200-message session with one mention ranks below a 12-message session built around the topic.\n\n### 4. Extract — pull the artifact out\n\nOnce you've found the winning session, don't dump the whole transcript. Extract the **minimal artifact**:\n\n- **For a fix:** the failing state → the change → the success marker. Three quotes, nothing more.\n- **For a decision:** the options considered → the chosen option → the stated rationale.\n- **For a command:** the exact command + the one line of context that says what it does.\n- **For a rejected approach:** what was tried → the observed failure → the inferred lesson.\n\nQuote the session (`> ...`) and cite it. Extraction patterns are detailed in `references/extraction-patterns.md`.\n\n### 5. Deduplicate — collapse the finds\n\nThe same fix often appears in three sessions (the first attempt, the retry, the \"oh and also\" follow-up). Deduplicate before reporting:\n\n- **Exact-code dedup** — identical fenced blocks collapse to one.\n- **Near-dup detection** — normalized text similarity > 0.85 merges into a cluster; keep the highest-scoring member as the canonical answer.\n- **Variant detection** — if two snippets differ only in a version number or path, treat as the same artifact and note the latest variant.\n\n`excavate.py --dedup` runs all three. See `references/deduplication.md`.\n\n## Using `excavate.py`\n\nThe script lives at `scripts/excavate.py`. It has no third-party dependencies — stdlib only — so it runs anywhere Python 3.8+ does.\n\n```bash\n# Basic keyword dig over a directory of .md / .txt / .json session logs\npython3 scripts/excavate.py dig ./sessions --query \"kafka consumer rebalance\"\n\n# Multiple terms (AND'd within a session), show top 5 with per-session scores\npython3 scripts/excavate.py dig ./sessions --query \"rebalance retry backoff\" --top 5 --explain\n\n# Add a date window (ISO dates), dedup near-identical results\npython3 scripts/excavate.py dig ./sessions \\\n  --query \"connection pool exhaustion\" \\\n  --after 2024-01-01 --before 2024-06-01 \\\n  --dedup\n\n# Dump the extracted code blocks across all matching sessions\npython3 scripts/excavate.py dig ./sessions --query \"ffmpeg concatenate\" --extract code\n\n# Index a directory once, then query the index repeatedly (faster for large corpora)\npython3 scripts/excavate.py index ./sessions --out sessions.idx\npython3 scripts/excavate.py query sessions.idx --query \"oauth refresh token\" --top 3 --explain\n```\n\n`--explain` prints the per-signal score breakdown so you can see *why* a session ranked where it did. Full CLI reference: `references/cli-reference.md`.\n\n### Programmatic use\n\n```python\nfrom excavate import ArchaeologyIndex\n\nidx = ArchaeologyIndex()\nidx.scan(\"./sessions\")            # walk the directory once\nfor hit in idx.search(\"kafka rebalance\", top=5, explain=True):\n    print(hit.score, hit.path, hit.extraction)\n```\n\nThe `ArchaeologyIndex` class is the stable surface; the CLI is a thin wrapper over it.\n\n## Relevance Scoring in Depth\n\nScoring details, the math, and how to retune weights for your corpus are in `references/relevance-scoring.md`. Key points:\n\n- Scores are normalized to **[0, 1]** per signal before weighting, so a corpus change doesn't silently inflate one signal.\n- **Resolution markers dominate** by default — a session that explicitly says \"that fixed it\" beats a longer, denser session that merely mentions the term. Tune the weight down if your logs lack success markers.\n- Recency is **configurable**, not gospel. For stable domains (algorithms, math) weight it low; for fast-moving domains (frontend deps) weight it high.\n- The scoring is **transparent**, not learned. Every ranking is explainable; nothing is a black-box embedding. This matters when the user asks \"why did you trust that session?\"\n\n## Integration with `session_search`\n\nIf you're running inside Hermes, the native `session_search` tool is your semantic-adjacent pass — it has FTS5 over the session DB. Use this skill's workflow to *decide what to search for and how to rank the results*, then let `excavate.py` handle corpora that aren't in the session DB (exported logs, markdown notes, JSONL exports, another agent's transcripts).\n\n```text\nsemantic-adjacent pass  →  session_search(query=\"...\")        # Hermes session DB\nkeyword + structural    →  excavate.py dig ./exported-logs    # file-based corpora\n```\n\n## Common Pitfalls\n\n1. **Searching one query and giving up.** The single biggest failure mode. Run at least three passes (exact, semantic-adjacent, structural). Artifacts are rarely filed under the first word you reach for.\n\n2. **Trusting match count as relevance.** A session that mentions \"docker\" forty times while setting up a CI pipeline is not the answer to \"how did we fix the docker permissions bug.\" Use the composite score, not raw hits.\n\n3. **Skipping dedup and reporting three copies of the same fix.** Always run `--dedup` when `top > 1`. The user asked for the answer, not the archaeology of the answer.\n\n4. **Extracting the whole session.** The transcript is the *site*, not the *artifact*. Quote minimally.\n\n5. **Retrieving a decision without its rationale.** \"We chose Postgres\" is useless without \"because we needed strong consistency for the ledger.\" Resolution and rationale travel together — extract both or neither.\n\n6. **Assuming recency equals correctness.** A two-year-old session that solved the exact algorithmic problem beats yesterday's near-miss. Recency is a *tiebreaker*, weighted low by default for a reason.\n\n7. **Digging without a survey.** If you can't state what artifact you want and what shape it takes, your query will be too vague to rank well. Spend ten seconds on the table in Phase 1.\n\n8. **Forgetting structural pass.** The fix is often inside a fenced code block that doesn't repeat the keyword in prose. `--extract code` exists for this reason.\n\n9. **Treating near-dups as separate finds.** Three sessions with the same stack trace are one problem, not three. Normalize before you count.\n\n10. **Not citing the source session.** Always cite. The user may want to open the original; future you will want to re-excavate.\n\n## Verification Checklist\n\n- [ ] Survey written: the artifact type and query shape are stated before searching.\n- [ ] At least two query passes run (one exact/keyword, one semantic-adjacent or structural).\n- [ ] Results ranked with the composite score, not raw match count (`--explain` if using the script).\n- [ ] Deduplication run when multiple results are returned.\n- [ ] Extraction is minimal — failing state, change, success marker — not the whole transcript.\n- [ ] Source session cited (path or session id).\n- [ ] Decision artifacts include rationale, not just the chosen option.\n- [ ] Recency weighted appropriately for the domain (low for stable, high for volatile).\n\n## Further Reading\n\n- `references/search-strategies.md` — the full playbook for the Locate phase: query expansion, negation, temporal constraints, and how to pick pass order.\n- `references/relevance-scoring.md` — the math behind the composite score, normalization, and how to retune weights.\n- `references/extraction-patterns.md` — extraction templates for fix, decision, command, and rejection artifacts.\n- `references/deduplication.md` — exact, near-dup, and variant detection in depth.\n- `references/cli-reference.md` — every `excavate.py` flag and subcommand.\n- `scripts/excavate.py` — the implementation. Stdlib only, single file, importable.\n\n---\n\n*Prompt Archaeology: don't re-derive what you've already discovered. Excavate it.*\n\nFile v0.1.0:README.md\n\n# Prompt Archaeology\n\n> An AI agent skill for excavating forgotten solutions, code snippets, and decisions from past conversation sessions — instead of re-solving problems from scratch.\n\n[![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](LICENSE)\n[![Python 3.8+](https://img.shields.io/badge/python-3.8+-blue.svg)](https://www.python.org/downloads/)\n\nEvery AI session is a stratum — a sedimented layer of debugging, decision-making, and discovery. Over time, valuable artifacts sink below the surface: a one-liner that fixed a gnarly race condition, a config that satisfied a finicky build, the exact incantation that convinced a model to behave. Most agents never dig for these. **Prompt Archaeology** turns that history into a quarryable resource.\n\n## What it gives you\n\n- **Search strategies** — keyword, semantic-adjacent, temporal, and structural queries tuned for session transcripts.\n- **Relevance scoring** — a transparent, composable ranking that surfaces the one session that actually matters (not just the one that mentions the term most).\n- **Knowledge extraction patterns** — recipes for pulling *decisions* and *solutions* out of a wall of chat, not just matching text.\n- **Deduplication** — collapse near-duplicate fixes across sessions into a single canonical answer.\n- **`excavate.py`** — a standalone Python script that crawls session logs and markdown files, ranks them, and prints the buried artifacts. **Zero third-party dependencies** (stdlib only).\n\n## The workflow in one breath\n\n**Survey** what artifact you want → **Locate** it with multiple query passes → **Score** the finds with the composite ranker → **Extract** the minimal artifact (with citation) → **Deduplicate** before reporting.\n\n## Quick start\n\n```bash\ngit clone https://github.com/voronindenis5/prompt-archaeology.git\ncd prompt-archaeology\n\n# Basic keyword dig over a directory of session logs (.md/.txt/.json/.jsonl)\npython3 scripts/excavate.py dig ./my-sessions --query \"kafka consumer rebalance\"\n\n# Multiple terms, top 5, with per-signal score breakdown\npython3 scripts/excavate.py dig ./my-sessions --query \"rebalance retry backoff\" --top 5 --explain\n\n# Date window + dedup near-identical results\npython3 scripts/excavate.py dig ./my-sessions \\\n  --query \"connection pool exhaustion\" \\\n  --after 2024-01-01 --before 2024-06-01 --dedup\n\n# Dump just the extracted code blocks across matches\npython3 scripts/excavate.py dig ./my-sessions --query \"ffmpeg concatenate\" --extract code\n```\n\n### Index once, query many\n\nFor large corpora, build an index and query it repeatedly:\n\n```bash\npython3 scripts/excavate.py index ./my-sessions --out sessions.idx\npython3 scripts/excavate.py query sessions.idx --query \"oauth refresh token\" --top 3 --explain\n```\n\n### Programmatic use\n\n```python\nfrom excavate import ArchaeologyIndex\n\nidx = ArchaeologyIndex()\nidx.scan(\"./my-sessions\")\nfor hit in idx.search(\"kafka rebalance\", top=5, explain=True):\n    print(hit.score, hit.path)\n    print(hit.extraction)\n```\n\n## Why \"archaeology\"?\n\nAn archaeologist does not grep the desert for \"pottery\" and ship the first hit. They survey, triangulate, carefully extract, and catalog. This skill teaches the agent to do the same with its own past — because re-deriving a solution you already found is the most expensive way to be wrong.\n\n## Repository layout\n\n```\nprompt-archaeology/\n├── SKILL.md                          # The skill itself (Hermes/OpenClaw format)\n├── README.md                         # You are here\n├── LICENSE                           # MIT\n├── references/\n│   ├── search-strategies.md          # Locate-phase playbook\n│   ├── relevance-scoring.md          # The math + how to retune weights\n│   ├── extraction-patterns.md        # Fix/decision/command/rejection templates\n│   ├── deduplication.md              # Exact, near-dup, and variant detection\n│   └── cli-reference.md              # Every excavate.py flag\n└── scripts/\n    └── excavate.py                   # Stdlib-only searcher + ranker\n```\n\n## Reference docs\n\n| Doc | What it covers |\n|---|---|\n| [`search-strategies.md`](references/search-strategies.md) | Query expansion, negation, temporal constraints, pass ordering |\n| [`relevance-scoring.md`](references/relevance-scoring.md) | Composite score math, normalization, weight tuning |\n| [`extraction-patterns.md`](references/extraction-patterns.md) | Minimal-artifact templates by artifact type |\n| [`deduplication.md`](references/deduplication.md) | Exact / near-dup / variant detection in depth |\n| [`cli-reference.md`](references/cli-reference.md) | Full `excavate.py` CLI reference |\n\n## Requirements\n\n- Python **3.8+** (uses only the standard library — no `pip install` needed)\n\n## License\n\nMIT © Denis Voronin\n\nFile v0.1.0:_meta.json\n\n{\n  \"ownerId\": \"kn75wwn4x6djaf28jbykeamazd81gtdp\",\n  \"slug\": \"prompt-archaeology\",\n  \"version\": \"0.1.0\",\n  \"publishedAt\": 1785959367842\n}\n\nFile v0.1.0:references/cli-reference.md\n\n# `excavate.py` CLI Reference\n\n`excavate.py` is a stdlib-only Python script for excavating session logs and markdown files. It supports three subcommands: `dig`, `index`, and `query`.\n\n## Global behavior\n\n- **No third-party dependencies.** Runs on Python 3.8+ with the standard library only.\n- **Reads** `.md`, `.txt`, `.json`, and `.jsonl` files (see [File formats](#file-formats)).\n- **Writes** nothing unless `--out` is given (index subcommand).\n- **Exits non-zero** on argument errors; zero on successful search (even with no hits).\n\n## Subcommands\n\n### `dig` — search a directory directly\n\n```bash\npython3 scripts/excavate.py dig <directory> --query <terms> [options]\n```\n\nScans `<directory>` recursively, scores every file against the query, prints the top matches. Use `dig` for one-off searches; use `index` + `query` for repeated searches over the same corpus.\n\n**Options:**\n\n| Flag | Type | Default | Description |\n|---|---|---|---|\n| `--query`, `-q` | str (required) | — | Search terms. Multiple terms are AND'd within a file (all must appear). Use `--query \"a b c\"` or repeat `--query` for OR semantics is not supported; pass a single string. |\n| `--top`, `-n` | int | 5 | Number of top results to print. |\n| `--after` | date (ISO) | — | Only include files modified on/after this date (`YYYY-MM-DD`). |\n| `--before` | date (ISO) | — | Only include files modified on/before this date (`YYYY-MM-DD`). |\n| `--not` | str | — | Exclude files containing this term. Repeatable. |\n| `--explain` | flag | off | Print per-signal score breakdown for each result. |\n| `--extract` | `code` \\| `all` \\| `none` | `none` | Print extracted code blocks (`code`), full extraction (`all`), or just scores (`none`). |\n| `--dedup` | flag | off | Collapse near-duplicate results into clusters. |\n| `--dedup-threshold` | float | 0.85 | Jaccard threshold for near-dup (0–1). |\n| `--keep-variants` | flag | off | With `--dedup`, report all version/path variants instead of collapsing. |\n\n**Examples:**\n\n```bash\n# Basic dig\npython3 scripts/excavate.py dig ./sessions --query \"kafka rebalance\"\n\n# Top 10 with score breakdown\npython3 scripts/excavate.py dig ./sessions --query \"pool exhaustion\" --top 10 --explain\n\n# Date window, exclude sidekiq mentions, dedup\npython3 scripts/excavate.py dig ./sessions \\\n  --query \"redis cache\" \\\n  --after 2024-01-01 --before 2024-06-30 \\\n  --not sidekiq \\\n  --dedup\n\n# Extract code blocks only\npython3 scripts/excavate.py dig ./sessions --query \"ffmpeg concat\" --extract code\n```\n\n### `index` — build a reusable index\n\n```bash\npython3 scripts/excavate.py index <directory> --out <index-file> [options]\n```\n\nScans `<directory>` once and serializes the `ArchaeologyIndex` to `<index-file>` (pickle format). Subsequent `query` calls load the index instead of re-scanning.\n\n**Options:**\n\n| Flag | Type | Default | Description |\n|---|---|---|---|\n| `--out`, `-o` | path (required) | — | Output index file path. |\n| `--after` / `--before` | date | — | Pre-filter by mtime at index time (cannot be relaxed later). |\n\n**Example:**\n\n```bash\npython3 scripts/excavate.py index ./sessions --out sessions.idx\n```\n\n### `query` — search an existing index\n\n```bash\npython3 scripts/excavate.py query <index-file> --query <terms> [options]\n```\n\nLoads `<index-file>` and runs a search. Accepts the same options as `dig` (`--top`, `--not`, `--explain`, `--extract`, `--dedup`, etc.), except date filters (those were applied at index time).\n\n**Example:**\n\n```bash\npython3 scripts/excavate.py query sessions.idx --query \"oauth refresh\" --top 3 --explain --dedup\n```\n\n## File formats\n\n`excavate.py` auto-detects format by extension:\n\n| Extension | Parsing |\n|---|---|\n| `.md`, `.markdown` | Raw text. Honors YAML frontmatter (between `---` fences) for date extraction. Fenced code blocks (``` ``` ```) extracted for structural pass. |\n| `.txt` | Raw text. No frontmatter support. |\n| `.json` | Expected to be a single object or array of message objects with `role` and `content` fields. Content fields are concatenated. |\n| `.jsonl` | Each line is a JSON object with `role` and `content`. Concatenated in order. |\n\nFor `.json` / `.jsonl`, the script looks for common field names: `content`, `text`, `message`, `body`. If your format differs, pre-process to `.md` or `.txt`.\n\n## Scoring flags\n\nSee `references/relevance-scoring.md` for the scoring model. The weights are constants in the script; retune by editing:\n\n```python\nWEIGHT_DENSITY    = 0.25\nWEIGHT_RECENCY    = 0.15\nWEIGHT_CODE       = 0.25\nWEIGHT_RESOLUTION = 0.35\n```\n\n## Programmatic API\n\n```python\nfrom excavate import ArchaeologyIndex, SearchHit\n\nidx = ArchaeologyIndex()\nidx.scan(\"./sessions\")                    # walk directory, parse files\nidx.save(\"sessions.idx\")                  # serialize\nidx2 = ArchaeologyIndex.load(\"sessions.idx\")\n\nhits: list[SearchHit] = idx.search(\n    query=\"kafka rebalance\",\n    top=5,\n    explain=True,                         # populate hit.explanation\n    exclude=(\"sidekiq\",),                 # --not terms\n)\n\nfor hit in hits:\n    print(hit.path, hit.score, hit.explanation)\n    print(hit.extraction)                 # extracted code/text\n```\n\n### `SearchHit` fields\n\n| Field | Type | Description |\n|---|---|---|\n| `path` | str | File path. |\n| `score` | float | Composite score in [0, 1]. |\n| `explanation` | dict \\| None | Per-signal breakdown (populated when `explain=True`). |\n| `extraction` | str | Extracted artifact (code block or relevant excerpt). |\n| `matches` | list[str] | Matched line snippets. |\n| `mtime` | float | File modification time. |\n\n## Exit codes\n\n| Code | Meaning |\n|---|---|\n| 0 | Search completed (including no hits). |\n| 2 | Argument error (missing required flag, bad value). |\n| 3 | I/O error (directory not found, unreadable file). |\n\n## Performance\n\n- **Scan speed:** ~1,000 files/sec on markdown, ~3,000/sec on plain text (single-threaded, SSD).\n- **Index size:** roughly 30–50% of the source corpus (pickle-serialized parsed text + metadata).\n- **Query speed:** <100ms for corpora under 10k files (in-memory).\n- **Memory:** proportional to corpus size; the index holds parsed text in memory for fast repeated queries.\n\nFor corpora above ~50k files, prefer `session_search` (FTS5-backed) over `excavate.py`.\n\nFile v0.1.0:references/deduplication.md\n\n# Deduplication\n\nThe same fix often appears in three sessions: the first attempt, the retry, and the \"oh and also\" follow-up. Deduplication collapses these into a single canonical answer before you report.\n\n`excavate.py --dedup` runs three layers of dedup in order: exact → near-dup → variant.\n\n## Layer 1: Exact dedup\n\n**Rule:** identical fenced code blocks collapse to one.\n\nTwo sessions with byte-identical ` ```bash ... ``` ` blocks are the same artifact. Keep the higher-scoring session as the canonical instance; drop the others from the result set but note them as duplicates.\n\n```\nsessions/fix-v1.md        score 0.71    [canonical]\n  ≡ sessions/fix-v1-retry.md             (exact dup of fix-v1.md)\n  ≡ sessions/fix-v1-followup.md          (exact dup of fix-v1.md)\n```\n\nExact dedup is cheap (a hash) and catches the easy cases.\n\n## Layer 2: Near-dup detection\n\n**Rule:** normalized text similarity > 0.85 merges into a cluster.\n\nTwo sessions that describe the same fix in slightly different words are the same artifact. Near-dup uses **normalized token overlap**:\n\n1. Lowercase, strip punctuation, remove stopwords.\n2. Tokenize on whitespace.\n3. Compute Jaccard similarity: `|A ∩ B| / |A ∪ B|`.\n4. If Jaccard > 0.85, merge.\n\nKeep the highest-scoring session in the cluster as canonical.\n\n```\nsessions/2024-03-12-a.md  score 0.82    [canonical]\n  ≈ sessions/2024-03-13-b.md            (near-dup, Jaccard 0.91)\n  ≈ sessions/2024-03-14-c.md            (near-dup, Jaccard 0.88)\n```\n\n### Tuning the threshold\n\n- **0.85 (default):** conservative. Only clearly-the-same fixes merge.\n- **0.75:** aggressive. Catches more dups but risks merging distinct fixes that share boilerplate (e.g., two different nginx fixes with similar config structure).\n- **0.95:** paranoid. Almost only exact dups merge.\n\nSet via `--dedup-threshold` on the CLI, or `DEDUP_THRESHOLD` in code.\n\n### What gets normalized away\n\n- Case (`Error` ≡ `error`)\n- Punctuation (`failed.` ≡ `failed`)\n- Stopwords (`the connection pool` ≡ `connection pool`)\n- Whitespace runs\n\n### What does NOT get normalized away\n\n- Numbers and identifiers (`pool_size=20` ≢ `pool_size=50`)\n- Code structure (two fenced blocks with different commands stay distinct)\n- File paths\n\n## Layer 3: Variant detection\n\n**Rule:** if two snippets differ only in a version number, path, or date, treat as the same artifact and note the latest variant.\n\n```\nsessions/2023-09-01.md:  pip install fastapi==0.68.0\nsessions/2024-03-12.md:  pip install fastapi==0.110.0\n```\n\nThese are the same artifact (\"install fastapi\") with a version variant. Collapse to one, note the latest (0.110.0).\n\nVariant detection works by:\n\n1. Masking version-like tokens (`\\d+\\.\\d+\\.\\d++`), date-like tokens, and path-like tokens.\n2. Re-running near-dup (Layer 2) on the masked text.\n3. If they merge, they're variants.\n\n### When variants matter\n\nSometimes the variant *is* the artifact — e.g., \"the exact version that worked with Python 3.9.\" Variant detection has a `--keep-variants` flag that reports all variants instead of collapsing, for these cases.\n\n## Output format\n\nWith `--dedup`, results print as clusters:\n\n```\n=== Cluster 1 (3 sessions, canonical score 0.82) ===\n  [canonical] sessions/2024-03-12-a.md        score 0.82\n  [dup]       sessions/2024-03-13-b.md        near-dup (Jaccard 0.91)\n  [dup]       sessions/2024-03-14-c.md        near-dup (Jaccard 0.88)\n\n=== Cluster 2 (1 session, canonical score 0.64) ===\n  [canonical] sessions/2024-02-20-x.md        score 0.64\n```\n\nReport only the canonical session per cluster to the user; mention duplicate counts inline if relevant (\"found in 3 sessions\").\n\n## Anti-patterns\n\n- **Dedup before scoring.** Always score first, then dedup — the highest-scoring session must be the canonical one.\n- **Aggressive near-dup on small corpora.** With <50 sessions, a 0.75 threshold will over-merge. Stick to 0.85+.\n- **Collapsing variants when versions matter.** If the user asked \"which version worked?\", `--keep-variants` is mandatory.\n- **Reporting every member of a cluster.** The user wants the answer, not the cluster graph. Report canonical; mention dup count.\n- **Dedup across different artifact types.** A fix and a rejection of the same approach are *different artifacts* even if textually similar. Dedup within artifact type.\n\nFile v0.1.0:references/extraction-patterns.md\n\n# Extraction Patterns\n\nThe transcript is the **site**, not the **artifact**. Once you've located and scored the winning session, extract the minimal artifact — never dump the whole transcript. This doc gives templates for each artifact type.\n\n## Universal extraction rules\n\n1. **Quote, don't paraphrase.** Use `> ...` blockquotes. The artifact must be the session's words, not your reconstruction.\n2. **Three quotes max** (failing state, change, success marker) unless the artifact is inherently longer (a config file).\n3. **Cite the session** — path or session id — on every extraction.\n4. **Drop the journey.** The eight wrong turns before the fix are not the artifact. They may be relevant for a *rejection* extraction (below), but not for a *fix*.\n\n## 1. Fix extraction\n\nThe most common artifact. Three quotes: failing state → change → success marker.\n\n**Template:**\n\n```markdown\n> **Failing:** `ConnectionPool exceeded max_connections (50)`\n>\n> **Change:** set `SQLALCHEMY_POOL_SIZE=20` and `SQLALCHEMY_MAX_OVERFLOW=5` in `.env`\n>\n> **Resolved:** \"spun up the app, hit it with 200 concurrent requests, pool stable — that fixed it\"\n\n— `sessions/2024-03-12-db-pool.md`\n```\n\n**What to drop:** the forty messages of hypothesis-testing before the config change. They're noise once the fix is known.\n\n## 2. Decision extraction\n\nTwo flavors: a decision *with* rationale is valuable; a decision *without* rationale is nearly useless. Always extract both.\n\n**Template:**\n\n```markdown\n> **Options considered:** Postgres vs. MongoDB for the ledger\n>\n> **Chose:** Postgres\n>\n> **Rationale:** \"we need strong consistency for the ledger — eventual consistency\n> would mean we could double-spend on a network partition, and that's a\n> correctness bug, not a performance bug\"\n\n— `sessions/2024-01-08-db-choice.md`\n```\n\n**If the rationale isn't in the session,** say so explicitly:\n\n```markdown\n> **Chose:** Postgres (rationale not recorded in this session; see sessions/2024-01-09*.md for follow-up)\n```\n\nDon't invent a rationale to fill the gap.\n\n## 3. Command / incantation extraction\n\nThe exact command plus one line of context. Nothing else.\n\n**Template:**\n\n```markdown\n> Concatenate two MP4s without re-encoding:\n>\n> ```bash\n> ffmpeg -i \"concat:in1.mp4|in2.mp4\" -c copy out.mp4\n> ```\n\n— `sessions/2024-05-30-video.md`\n```\n\nIf the command has prerequisites (a `dep install`, an env var), include them in the same block. If the session showed a *wrong* invocation first, include it only as a `# NOT this:` comment.\n\n## 4. Rejected-approach extraction\n\nOften more valuable than the fix — it saves you from re-walking a dead end.\n\n**Template:**\n\n```markdown\n> **Tried:** rewriting the parser as a single regex\n>\n> **Failed:** \"catastrophic backtracking on inputs > 4KB, CPU pinned at 100%\"\n>\n> **Lesson:** nested quantifiers on unbounded input; stick with the recursive-descent parser\n\n— `sessions/2024-02-14-parser.md`\n```\n\n**When to extract a rejection:** the session explicitly records *what was tried*, *how it failed*, and (ideally) *why*. If only the failure is recorded with no diagnosis, extract the failure and mark the lesson as inferred.\n\n## 5. Config / architecture extraction\n\nFor durable artifacts (config files, architecture diagrams described in prose). Quote the whole relevant block, not a snippet.\n\n**Template:**\n\n```markdown\n> nginx config that fixed the 502 on long-polling:\n>\n> ```nginx\n> location /events {\n>     proxy_pass http://backend;\n>     proxy_read_timeout 3600s;\n>     proxy_buffering off;\n> }\n> ```\n\n— `sessions/2024-04-01-nginx.md`\n```\n\n## Anti-patterns\n\n- **The whole transcript.** The user asked for the fix, not the dig diary.\n- **Paraphrased fix.** \"We changed the pool size\" is useless; `SQLALCHEMY_POOL_SIZE=20` is the artifact.\n- **Decision without rationale.** \"We chose Postgres\" answers nothing. Always pair with the *why*.\n- **Fix without success marker.** If the session never confirmed the fix worked, say so: \"applied but not confirmed in this session.\"\n- **Extracting the first mention.** The first mention of a term is often the *problem statement*, not the fix. Score and read before extracting.\n- **Inventing rationale.** If the session doesn't say why, don't fill it in. Mark it missing.\n\nFile v0.1.0:references/relevance-scoring.md\n\n# Relevance Scoring in Depth\n\n`excavate.py` ranks sessions with a **transparent, composite score** — not a black-box embedding. Every ranking is explainable. This doc covers the math, normalization, and how to retune for your corpus.\n\n## The four signals\n\n| Signal | What it measures | Default weight | Range |\n|---|---|---|---|\n| `density` | match count relative to session length | 0.25 | [0, 1] |\n| `recency` | newer sessions score higher | 0.15 | [0, 1] |\n| `code` | does the session contain runnable code/commands? | 0.25 | [0, 1] |\n| `resolution` | success markers like \"that fixed it\", \"works now\", \"merged\" | 0.35 | [0, 1] |\n\nWeights sum to 1.0. Composite score is the weighted sum, also in [0, 1].\n\n## Why these weights\n\n- **Resolution dominates (0.35).** A session that explicitly says \"that fixed it\" is the strongest possible signal. Prose that merely mentions the term is weak evidence by comparison.\n- **Density and code tie (0.25).** A session densely packed with the term is probably *about* it; a session with code is probably *solving* it. Both matter; neither alone is decisive.\n- **Recency is the tiebreaker (0.15).** Useful, but not authoritative — a two-year-old fix to an algorithmic problem is still correct.\n\n## Normalization\n\nEach signal is normalized to **[0, 1]** before weighting, so a corpus change doesn't silently inflate one signal.\n\n### density\n\n```\ndensity(session) = match_count / max(match_count across corpus)\n```\n\nA session with 12 matches ranks at density 1.0 *only if* 12 is the max in the corpus; if another session has 50 matches, the 12-match session ranks 0.24. This is why raw match count is a bad proxy — density is relative.\n\nThe implementation also applies a **log saturation** so that one very long session doesn't dominate:\n\n```\ndensity(session) = log(1 + match_count) / log(1 + max_match_count)\n```\n\n### recency\n\n```\nrecency(session) = (session_mtime - corpus_min_time) / (corpus_max_time - corpus_min_time)\n```\n\nLinear interpolation between the oldest and newest session in the corpus. If the corpus spans 2022–2025, a mid-2024 session scores ~0.6.\n\n### code\n\n```\ncode(session) = 1.0 if session contains ≥1 fenced code block or shell command\n              = 0.5 if session contains inline `code` only\n              = 0.0 otherwise\n```\n\nBinary-ish. A session with code is qualitatively different from one without.\n\n### resolution\n\n```\nresolution(session) = (count of resolution markers) / max(count of resolution markers across corpus)\n```\n\nCapped at 1.0. Resolution markers (see below) are weighted by type.\n\n## Resolution markers\n\nThe marker lexicon lives in `excavate.py` as `RESOLUTION_MARKERS`. It groups phrases by strength:\n\n| Strength | Examples |\n|---|---|\n| Strong (×1.0) | \"that fixed it\", \"works now\", \"merged\", \"deployed\", \"shipped\" |\n| Medium (×0.7) | \"fixed\", \"resolved\", \"solved\", \"working\" |\n| Weak (×0.4) | \"seems to work\", \"might be it\", \"I think that's it\" |\n\nA session with one strong marker outscores a session with three weak ones.\n\n## Retuning weights\n\nWeights are constants at the top of `excavate.py`:\n\n```python\nWEIGHT_DENSITY    = 0.25\nWEIGHT_RECENCY    = 0.15\nWEIGHT_CODE       = 0.25\nWEIGHT_RESOLUTION = 0.35\n```\n\n### When to retune\n\n| Your corpus | Change |\n|---|---|\n| Fast-moving domain (frontend deps, infra) | Bump `RECENCY` to 0.25, drop `DENSITY` to 0.15 |\n| Stable domain (algorithms, math, CS theory) | Drop `RECENCY` to 0.05, bump `RESOLUTION` to 0.45 |\n| Mostly code (debugging transcripts) | Bump `CODE` to 0.35, drop `DENSITY` to 0.15 |\n| Prose-heavy (design discussions, few markers) | Drop `RESOLUTION` to 0.20, bump `DENSITY` to 0.40 |\n| Logs lack success markers entirely | Set `RESOLUTION` to 0.0 — it'll only add noise |\n\n### How to retune\n\nEdit the constants, then validate on a small labeled set (sessions where you *know* the right answer). The top-1 hit rate should climb as you tune toward your corpus's character.\n\n## Why not embeddings?\n\nEmbeddings would raise recall on the semantic-adjacent pass, but:\n\n1. **They're opaque.** The user asks \"why did you trust that session?\" and you can't answer. The composite score answers in one line.\n2. **They need a model.** `excavate.py` is stdlib-only and runs anywhere. Adding a model breaks that.\n3. **They need a corpus big enough to matter.** For most personal session corpora (hundreds to low thousands of sessions), keyword + structural + the composite ranker is competitive with embedding search and far cheaper.\n\nIf you have a large corpus and want semantic recall, run `session_search` (FTS5) or an embedding index for the semantic pass, then feed those candidates into `excavate.py`'s scorer. The signals compose.\n\n## The `--explain` output\n\n`excavate.py --explain` prints, per result:\n\n```\nsessions/2024-03-12-kafka-fix.md\n  score: 0.82\n    density    0.71   (12 matches, log-saturated)\n    recency    0.64   (2024-03-12, within corpus range)\n    code       1.00   (3 fenced blocks)\n    resolution 1.00   (1 strong marker: \"that fixed it\")\n```\n\nThis is the explainability contract. If a ranking looks wrong, `--explain` shows you exactly which signal is off and whether to retune.\n\nFile v0.1.0:references/search-strategies.md\n\n# Search Strategies (the Locate Phase)\n\nThe Locate phase runs **multiple query passes**, because artifacts are rarely filed under the first word you reach for. This doc is the full playbook.\n\n## Pass order\n\nRun passes in this order. Each later pass only runs if the earlier ones didn't surface a high-confidence hit.\n\n1. **Exact / keyword** — highest precision, lowest recall. The literal error string, function name, filename, or config key.\n2. **Semantic-adjacent** — paraphrase the intent. If the exact term misses, the concept may be filed under different words.\n3. **Structural** — look for *code blocks*, *diffs*, or *command outputs* near the topic, not just prose. Solutions hide inside fenced blocks.\n4. **Temporal** — constrain to the window when the work happened, then re-run keyword. Narrows a noisy corpus to the relevant stratum.\n\n## 1. Exact / keyword\n\nUse the **literal token** the artifact would contain:\n\n- An exception → the exact exception class and message fragment.\n- A function → its name.\n- A config value → the key name.\n- A CLI failure → the exact stderr fragment.\n\n```\n# exact, high-precision\nquery: \"connectionpool value too many connections\"\nquery: \"AttributeError: 'NoneType' object has no attribute 'split'\"\n```\n\n**Pitfall:** exact pass misses when the user paraphrased the error in their message but the stack trace was elided. Always follow with semantic-adjacent.\n\n## 2. Semantic-adjacent\n\nParaphrase the **intent**, not the token. Think: \"if I didn't remember the exact word, how would I describe this?\"\n\n| Concept | Exact miss | Semantic seed |\n|---|---|---|\n| OOM kill | `\"OOMKilled\"` | `\"memory limit pod restarted\"` |\n| Race condition | `\"concurrent modification\"` | `\"intermittent wrong order flaky\"` |\n| Cert expiry | `\"x509: certificate has expired\"` | `\"tls handshake failed clock skew\"` |\n| Port conflict | `\"Address already in use\"` | `\"cannot bind port already listening\"` |\n\n`excavate.py` does keyword matching; for the semantic pass use `session_search` (FTS5) or any embedding search over the corpus.\n\n## 3. Structural (code-block) pass\n\nThe fix is often inside a fenced code block whose prose doesn't repeat the keyword. `excavate.py --extract code` pulls every fenced block from matching sessions and re-ranks by proximity to the query terms.\n\nWhen to prioritize this pass:\n\n- You remember the *shape* of the answer (a shell one-liner, a YAML snippet) but not the words around it.\n- The keyword appears only in code comments or command output, never in prose.\n- The session is a wall of debugging chat with the fix buried in one block.\n\n## 4. Temporal pass\n\nConstrain to a date window, then re-run keyword. Two modes:\n\n- **`--after` / `--before`** on `excavate.py` filters by file mtime or, if the session has frontmatter, the session date.\n- **Relative** (\"sessions from the week we shipped v2\") — convert to absolute dates first.\n\nTemporal narrowing is how you separate the *original* fix from the *five times someone re-asked about it later*. The earliest high-scoring session in the window is usually the source.\n\n## Query expansion\n\nWhen keyword + semantic both miss, expand the query:\n\n- **Synonyms:** `rebalance` → `rebalancing`, `rebalance`, `coordinator`, `consumer group`.\n- **Hyponyms:** `kafka` → `producer`, `consumer`, `broker`, `topic`, `partition`.\n- **Surrounding nouns:** the files, services, or libraries adjacent to the problem.\n\nExpansion trades precision for recall. Run it last, and require a resolution marker (Phase 3 signal) to trust any expanded hit.\n\n## Negation\n\nTo find a session that discusses X but **not** Y (e.g., \"redis caching\" but not \"sidekiq\"):\n\n```bash\npython3 scripts/excavate.py dig ./sessions --query \"redis cache\" --not \"sidekiq\"\n```\n\nNegation is useful for disambiguating overloaded terms (e.g., \"migration\" the DB step vs. \"migration\" the framework).\n\n## Picking the number of passes\n\n| Confidence after pass N | Action |\n|---|---|\n| Pass 1 returns a session with a resolution marker | Stop. Extract. |\n| Pass 1 returns dense hits but no resolution marker | Run pass 2 to confirm. |\n| Pass 1 misses or is sparse | Run pass 2, then 3. |\n| Passes 1–3 all miss | Run pass 4 (temporal), then query expansion. |\n| All passes miss | The artifact may not exist. Say so — don't fabricate. |\n\n## Anti-patterns\n\n- **Single-query dig.** One search, one answer. This is grep, not archaeology. Always run ≥ 2 passes unless pass 1 returns a resolution.\n- **Keyword worship.** Refusing to paraphrase because \"the error message is the error message.\" The error message may not be in the transcript.\n- **Ignoring structure.** Reading only prose and missing the fenced block that contains the actual fix.\n- **No temporal filter on noisy corpora.** A term that appears in 200 sessions needs narrowing; use the date window.\n- **Treating expansion hits as gospel.** Expanded queries have low precision. Require a resolution marker before trusting.\n\nFile v0.1.0:skill-card.md\n\n## Description:\n\nPrompt Archaeology helps agents recover prior fixes, code snippets, decisions, commands, and rejected approaches from authorized past conversation sessions instead of re-solving known problems.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[voronindenis5](https://clawhub.ai/user/voronindenis5)\n\n### License/Terms of Use:\n\nMIT\n\n## Use Case:\n\nDevelopers and engineers use this skill to search authorized session logs, notes, or exported conversation history for prior fixes, decisions, commands, and rejected approaches. It is intended for cases where a user suspects the answer was already worked out in an earlier session and wants a cited, minimal artifact.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Searching session logs or notes can expose private, credential-bearing, or otherwise sensitive content.\n\nMitigation: Search only directories you are authorized to process, avoid credential or personal-data locations, and review extracted snippets before sharing them.\n\nRisk: Existing index files are loaded with Python pickle, which is unsafe for untrusted input.\n\nMitigation: Create indexes locally from trusted corpora and do not open index files from other people unless the index format is changed away from pickle or an explicit trusted-only warning and opt-in unsafe path are added.\n\nRisk: Recovered historical fixes or decisions may be stale or incomplete.\n\nMitigation: Require source-session citation and validate any recovered command, code, or decision against the current repository, dependencies, and user requirements before applying it.\n\n## Reference(s):\n\n- [Source repository](https://github.com/voronindenis5/prompt-archaeology)\n- [ClawHub skill page](https://clawhub.ai/voronindenis5/skills/prompt-archaeology)\n- [Search strategies](references/search-strategies.md)\n- [Relevance scoring](references/relevance-scoring.md)\n- [Extraction patterns](references/extraction-patterns.md)\n- [Deduplication](references/deduplication.md)\n- [CLI reference](references/cli-reference.md)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, code, shell commands, guidance]\n\n**Output Format:** [Markdown guidance with optional quoted excerpts, code blocks, and shell commands; the bundled CLI prints plain-text ranked search results.]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [May cite source session paths, include score explanations, and deduplicate near-identical results when requested.]\n\n## Skill Version(s):\n\n0.1.0 (source: ClawHub release metadata; artifact frontmatter says 1.0.0)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v0.1.0:LICENSE\n\nMIT License\n\nCopyright (c) 2026 Denis Voronin\n\nPermission is hereby granted, free of charge, to any person obtaining a copy\nof this software and associated documentation files (the \"Software\"), to deal\nin the Software without restriction, including without limitation the rights\nto use, copy, modify, merge, publish, distribute, sublicense, and/or sell\ncopies of the Software, and to permit persons to whom the Software is\nfurnished to do so, subject to the following conditions:\n\nThe above copyright notice and this permission notice shall be included in all\ncopies or substantial portions of the Software.\n\nTHE SOFTWARE IS PROVIDED \"AS IS\", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR\nIMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,\nFITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE\nAUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER\nLIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,\nOUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE\nSOFTWARE.","readmeExcerpt":"Skill: prompt-archaeology Owner: voronindenis5 Summary: Excavate forgotten solutions, code snippets, and decisions from past conversation sessions. Use when the user is re-solving a problem you've likely solved before, hunting for a lost snippet, or wants to mine session history for buried knowledge instead of starting from scratch. Tags: latest:0.1.1 Version history: v0.1.1 | 2026-08-11T11:58:40.748Z | auto Version ","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"# Basic keyword dig over a directory of .md / .txt / .json session logs\npython3 scripts/excavate.py dig ./sessions --query \"kafka consumer rebalance\"\n\n# Multiple terms (AND'd within a session), show top 5 with per-session scores\npython3 scripts/excavate.py dig ./sessions --query \"rebalance retry backoff\" --top 5 --explain\n\n# Add a date window (ISO dates), dedup near-identical results\npython3 scripts/excavate.py dig ./sessions \\\n  --query \"connection pool exhaustion\" \\\n  --after 2024-01-01 --before 2024-06-01 \\\n  --dedup\n\n# Dump the extracted code blocks across all matching sessions\npython3 scripts/excavate.py dig ./sessions --query \"ffmpeg concatenate\" --extract code\n\n# Index a directory once, then query the index repeatedly (faster for large corpora)\npython3 scripts/excavate.py index ./sessions --out sessions.idx\npython3 scripts/excavate.py query sessions.idx --query \"oauth refresh token\" --top 3 --explain"},{"language":"python","snippet":"from excavate import ArchaeologyIndex\n\nidx = ArchaeologyIndex()\nidx.scan(\"./sessions\")            # walk the directory once\nfor hit in idx.search(\"kafka rebalance\", top=5, explain=True):\n    print(hit.score, hit.path, hit.extraction)"},{"language":"text","snippet":"semantic-adjacent pass  →  session_search(query=\"...\")        # Hermes session DB\nkeyword + structural    →  excavate.py dig ./exported-logs    # file-based corpora"},{"language":"bash","snippet":"git clone https://github.com/voronindenis5/prompt-archaeology.git\ncd prompt-archaeology\n\n# Basic keyword dig over a directory of session logs (.md/.txt/.json/.jsonl)\npython3 scripts/excavate.py dig ./my-sessions --query \"kafka consumer rebalance\"\n\n# Multiple terms, top 5, with per-signal score breakdown\npython3 scripts/excavate.py dig ./my-sessions --query \"rebalance retry backoff\" --top 5 --explain\n\n# Date window + dedup near-identical results\npython3 scripts/excavate.py dig ./my-sessions \\\n  --query \"connection pool exhaustion\" \\\n  --after 2024-01-01 --before 2024-06-01 --dedup\n\n# Dump just the extracted code blocks across matches\npython3 scripts/excavate.py dig ./my-sessions --query \"ffmpeg concatenate\" --extract code"},{"language":"bash","snippet":"python3 scripts/excavate.py index ./my-sessions --out sessions.idx\npython3 scripts/excavate.py query sessions.idx --query \"oauth refresh token\" --top 3 --explain"},{"language":"python","snippet":"from excavate import ArchaeologyIndex\n\nidx = ArchaeologyIndex()\nidx.scan(\"./my-sessions\")\nfor hit in idx.search(\"kafka rebalance\", top=5, explain=True):\n    print(hit.score, hit.path)\n    print(hit.extraction)"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: prompt-archaeology\ndescription: \"Excavate forgotten solutions, code snippets, and decisions from past conversation sessions. Use when the user is re-solving a problem you've likely solved before, hunting for a lost snippet, or wants to mine session history for buried knowledge instead of starting from scratch.\"\nversion: 1.0.0\nauthor: Denis Voronin\nlicense: MIT\nmetadata:\n  hermes:\n    tags: [sessions, history, search, knowledge-mining, archaeology, recovery, hermes-agent]\n    related_skills: []\n---\n\n# Prompt Archaeology\n\n## Overview\n\n**Prompt Archaeology** is the practice of excavating your own conversation history instead of re-solving problems from scratch. Every AI session is a stratum — a sedimented layer of debugging, decision-making, and discovery. Over time, valuable artifacts sink below the surface: a one-liner that fixed a gnarly race condition, a config that satisfied a finicky build, the exact incantation that convinced a model to behave. Most agents never dig for these. They re-derive, re-guess, and re-fail.\n\nThis skill turns that history into a quarryable resource. It bundles:\n\n- **Search strategies** — keyword, semantic-adjacent, temporal, and structural queries tuned for session transcripts.\n- **Relevance scoring** — a transparent, composable ranking that surfaces the one session that actually matters.\n- **Knowledge extraction patterns** — recipes for pulling *decisions* and *solutions* out of a wall of chat, not just matching text.\n- **Deduplication** — collapse near-duplicate fixes across sessions into a single canonical answer.\n- **`excavate.py`** — a standalone Python script that crawls session logs and markdown files, ranks them, and prints the buried artifacts.\n\nThe metaphor is deliberate. An archaeologist does not grep the desert for \"pottery\" and ship the first hit. They survey, triangulate, carefully extract, and catalog. This skill teaches the agent to do the same with its own past.\n\n## When to Use\n\n- **The user is about to re-solve a known problem.** They describe a bug or task and you have a flicker of \"we've done this before.\" Excavate before answering.\n- **\"Didn't we figure out...?\" / \"What did we land on?\"** — retrieve the prior decision and its rationale, not just the outcome.\n- **Hunting for a lost code snippet, config value, or command** that worked months ago.\n- **Onboarding to a codebase you've touched before** — pull the architectural decisions out of old sessions.\n- **Avoiding repeated dead ends** — find the approaches that were *rejected* and why, so you don't walk back into them.\n- **Writing postmortems or ADRs** from scattered session evidence.\n\n### Don't use for\n\n- Fresh problems with no prior history — there's nothing to excavate; solve forward.\n- When you already hold the answer in active context — don't pad the turn with a search.\n- Sensitive retrieval across other users' private profiles unless explicitly authorized.\n\n## The Excavation Workflow\n\nA dig has five phases. Skipping any phase degrad"},{"path":"README.md","content":"# Prompt Archaeology\n\n> An AI agent skill for excavating forgotten solutions, code snippets, and decisions from past conversation sessions — instead of re-solving problems from scratch.\n\n[![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](LICENSE)\n[![Python 3.8+](https://img.shields.io/badge/python-3.8+-blue.svg)](https://www.python.org/downloads/)\n\nEvery AI session is a stratum — a sedimented layer of debugging, decision-making, and discovery. Over time, valuable artifacts sink below the surface: a one-liner that fixed a gnarly race condition, a config that satisfied a finicky build, the exact incantation that convinced a model to behave. Most agents never dig for these. **Prompt Archaeology** turns that history into a quarryable resource.\n\n## What it gives you\n\n- **Search strategies** — keyword, semantic-adjacent, temporal, and structural queries tuned for session transcripts.\n- **Relevance scoring** — a transparent, composable ranking that surfaces the one session that actually matters (not just the one that mentions the term most).\n- **Knowledge extraction patterns** — recipes for pulling *decisions* and *solutions* out of a wall of chat, not just matching text.\n- **Deduplication** — collapse near-duplicate fixes across sessions into a single canonical answer.\n- **`excavate.py`** — a standalone Python script that crawls session logs and markdown files, ranks them, and prints the buried artifacts. **Zero third-party dependencies** (stdlib only).\n\n## The workflow in one breath\n\n**Survey** what artifact you want → **Locate** it with multiple query passes → **Score** the finds with the composite ranker → **Extract** the minimal artifact (with citation) → **Deduplicate** before reporting.\n\n## Quick start\n\n```bash\ngit clone https://github.com/voronindenis5/prompt-archaeology.git\ncd prompt-archaeology\n\n# Basic keyword dig over a directory of session logs (.md/.txt/.json/.jsonl)\npython3 scripts/excavate.py dig ./my-sessions --query \"kafka consumer rebalance\"\n\n# Multiple terms, top 5, with per-signal score breakdown\npython3 scripts/excavate.py dig ./my-sessions --query \"rebalance retry backoff\" --top 5 --explain\n\n# Date window + dedup near-identical results\npython3 scripts/excavate.py dig ./my-sessions \\\n  --query \"connection pool exhaustion\" \\\n  --after 2024-01-01 --before 2024-06-01 --dedup\n\n# Dump just the extracted code blocks across matches\npython3 scripts/excavate.py dig ./my-sessions --query \"ffmpeg concatenate\" --extract code\n```\n\n### Index once, query many\n\nFor large corpora, build an index and query it repeatedly:\n\n```bash\npython3 scripts/excavate.py index ./my-sessions --out sessions.idx\npython3 scripts/excavate.py query sessions.idx --query \"oauth refresh token\" --top 3 --explain\n```\n\n### Programmatic use\n\n```python\nfrom excavate import ArchaeologyIndex\n\nidx = ArchaeologyIndex()\nidx.scan(\"./my-sessions\")\nfor hit in idx.search(\"kafka rebalance\", top=5, explain=True):\n    print(hit.score, hit.path)\n    print(hit.extraction"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn75wwn4x6djaf28jbykeamazd81gtdp\",\n  \"slug\": \"prompt-archaeology\",\n  \"version\": \"0.1.1\",\n  \"publishedAt\": 1786449520748\n}"},{"path":"references/cli-reference.md","content":"# `excavate.py` CLI Reference\n\n`excavate.py` is a stdlib-only Python script for excavating session logs and markdown files. It supports three subcommands: `dig`, `index`, and `query`.\n\n## Global behavior\n\n- **No third-party dependencies.** Runs on Python 3.8+ with the standard library only.\n- **Reads** `.md`, `.txt`, `.json`, and `.jsonl` files (see [File formats](#file-formats)).\n- **Writes** nothing unless `--out` is given (index subcommand).\n- **Exits non-zero** on argument errors; zero on successful search (even with no hits).\n\n## Subcommands\n\n### `dig` — search a directory directly\n\n```bash\npython3 scripts/excavate.py dig <directory> --query <terms> [options]\n```\n\nScans `<directory>` recursively, scores every file against the query, prints the top matches. Use `dig` for one-off searches; use `index` + `query` for repeated searches over the same corpus.\n\n**Options:**\n\n| Flag | Type | Default | Description |\n|---|---|---|---|\n| `--query`, `-q` | str (required) | — | Search terms. Multiple terms are AND'd within a file (all must appear). Use `--query \"a b c\"` or repeat `--query` for OR semantics is not supported; pass a single string. |\n| `--top`, `-n` | int | 5 | Number of top results to print. |\n| `--after` | date (ISO) | — | Only include files modified on/after this date (`YYYY-MM-DD`). |\n| `--before` | date (ISO) | — | Only include files modified on/before this date (`YYYY-MM-DD`). |\n| `--not` | str | — | Exclude files containing this term. Repeatable. |\n| `--explain` | flag | off | Print per-signal score breakdown for each result. |\n| `--extract` | `code` \\| `all` \\| `none` | `none` | Print extracted code blocks (`code`), full extraction (`all`), or just scores (`none`). |\n| `--dedup` | flag | off | Collapse near-duplicate results into clusters. |\n| `--dedup-threshold` | float | 0.85 | Jaccard threshold for near-dup (0–1). |\n| `--keep-variants` | flag | off | With `--dedup`, report all version/path variants instead of collapsing. |\n\n**Examples:**\n\n```bash\n# Basic dig\npython3 scripts/excavate.py dig ./sessions --query \"kafka rebalance\"\n\n# Top 10 with score breakdown\npython3 scripts/excavate.py dig ./sessions --query \"pool exhaustion\" --top 10 --explain\n\n# Date window, exclude sidekiq mentions, dedup\npython3 scripts/excavate.py dig ./sessions \\\n  --query \"redis cache\" \\\n  --after 2024-01-01 --before 2024-06-30 \\\n  --not sidekiq \\\n  --dedup\n\n# Extract code blocks only\npython3 scripts/excavate.py dig ./sessions --query \"ffmpeg concat\" --extract code\n```\n\n### `index` — build a reusable index\n\n```bash\npython3 scripts/excavate.py index <directory> --out <index-file> [options]\n```\n\nScans `<directory>` once and serializes the `ArchaeologyIndex` to `<index-file>` (pickle format). Subsequent `query` calls load the index instead of re-scanning.\n\n**Options:**\n\n| Flag | Type | Default | Description |\n|---|---|---|---|\n| `--out`, `-o` | path (required) | — | Output index file path. |\n| `--after` / `--before` | date | — | Pre-filter by mtime at index time"},{"path":"references/deduplication.md","content":"# Deduplication\n\nThe same fix often appears in three sessions: the first attempt, the retry, and the \"oh and also\" follow-up. Deduplication collapses these into a single canonical answer before you report.\n\n`excavate.py --dedup` runs three layers of dedup in order: exact → near-dup → variant.\n\n## Layer 1: Exact dedup\n\n**Rule:** identical fenced code blocks collapse to one.\n\nTwo sessions with byte-identical ` ```bash ... ``` ` blocks are the same artifact. Keep the higher-scoring session as the canonical instance; drop the others from the result set but note them as duplicates.\n\n```\nsessions/fix-v1.md        score 0.71    [canonical]\n  ≡ sessions/fix-v1-retry.md             (exact dup of fix-v1.md)\n  ≡ sessions/fix-v1-followup.md          (exact dup of fix-v1.md)\n```\n\nExact dedup is cheap (a hash) and catches the easy cases.\n\n## Layer 2: Near-dup detection\n\n**Rule:** normalized text similarity > 0.85 merges into a cluster.\n\nTwo sessions that describe the same fix in slightly different words are the same artifact. Near-dup uses **normalized token overlap**:\n\n1. Lowercase, strip punctuation, remove stopwords.\n2. Tokenize on whitespace.\n3. Compute Jaccard similarity: `|A ∩ B| / |A ∪ B|`.\n4. If Jaccard > 0.85, merge.\n\nKeep the highest-scoring session in the cluster as canonical.\n\n```\nsessions/2024-03-12-a.md  score 0.82    [canonical]\n  ≈ sessions/2024-03-13-b.md            (near-dup, Jaccard 0.91)\n  ≈ sessions/2024-03-14-c.md            (near-dup, Jaccard 0.88)\n```\n\n### Tuning the threshold\n\n- **0.85 (default):** conservative. Only clearly-the-same fixes merge.\n- **0.75:** aggressive. Catches more dups but risks merging distinct fixes that share boilerplate (e.g., two different nginx fixes with similar config structure).\n- **0.95:** paranoid. Almost only exact dups merge.\n\nSet via `--dedup-threshold` on the CLI, or `DEDUP_THRESHOLD` in code.\n\n### What gets normalized away\n\n- Case (`Error` ≡ `error`)\n- Punctuation (`failed.` ≡ `failed`)\n- Stopwords (`the connection pool` ≡ `connection pool`)\n- Whitespace runs\n\n### What does NOT get normalized away\n\n- Numbers and identifiers (`pool_size=20` ≢ `pool_size=50`)\n- Code structure (two fenced blocks with different commands stay distinct)\n- File paths\n\n## Layer 3: Variant detection\n\n**Rule:** if two snippets differ only in a version number, path, or date, treat as the same artifact and note the latest variant.\n\n```\nsessions/2023-09-01.md:  pip install fastapi==0.68.0\nsessions/2024-03-12.md:  pip install fastapi==0.110.0\n```\n\nThese are the same artifact (\"install fastapi\") with a version variant. Collapse to one, note the latest (0.110.0).\n\nVariant detection works by:\n\n1. Masking version-like tokens (`\\d+\\.\\d+\\.\\d++`), date-like tokens, and path-like tokens.\n2. Re-running near-dup (Layer 2) on the masked text.\n3. If they merge, they're variants.\n\n### When variants matter\n\nSometimes the variant *is* the artifact — e.g., \"the exact version that worked with Python 3.9.\" Variant detection has a `--keep-variants` f"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Excavate forgotten solutions, code snippets, and decisions from past conversation sessions. Use when the user is re-solving a problem you've likely solved before, hunting for a lost snippet, or wants to mine session history for buried knowledge instead of starting from scratch. Skill: prompt-archaeology Owner: voronindenis5 Summary: Excavate forgotten solutions, code snippets, and decisions from past conversation sessions. Use when the user is re-solving a problem you've likely solved before, hunting for a lost snippet, or wants to mine session history for buried knowledge instead of starting from scratch. Tags: latest:0.1.1 Version history: v0.1.1 | 2026-08-11T11:58:40.748Z | auto Version","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1372,"uniquenessScore":47,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-09T13:43:44.391Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-09T13:43:44.391Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T07:20:11.417Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}