{"id":"8029e64d-591c-4ef6-a2f4-9d2fb739abde","entityType":"agent","slug":"clawhub-wmantly-openclaw-rag-skill","name":"Rag","canonicalUrl":"https://www.xpersona.co/agent/clawhub-wmantly-openclaw-rag-skill","canonicalPath":"/agent/clawhub-wmantly-openclaw-rag-skill","generatedAt":"2026-10-10T21:50:59.040Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T16:41:52.721Z","emptyReason":null},"description":"Complete RAG (Retrieval-Augmented Generation) system for OpenClaw. Indexes chat sessions, workspace code, documentation, and skills into local ChromaDB for s...","descriptionLabel":"Source description","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.3K downloads reported by the source. Last updated 10/10/2026.","installCommand":"clawhub skill install s1799acy8a391927skb4d19q3h885hmb:openclaw-rag-skill","sourceUrl":"https://clawhub.ai/wmantly/openclaw-rag-skill","homepage":"https://clawhub.ai/wmantly/skills/openclaw-rag-skill","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/wmantly/openclaw-rag-skill","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/wmantly/skills/openclaw-rag-skill","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":63,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Rag technical dossier on Xpersona with agent coverage, OPENCLEW support, and live trust metadata."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-10T16:41:52.721Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T16:41:52.721Z","emptyReason":null},"stars":null,"forks":null,"downloads":1335,"packageName":null,"latestVersion":"1.0.6","tractionLabel":"1.3K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T16:41:52.604Z","emptyReason":null},"lastUpdatedAt":"2026-10-10T16:41:52.721Z","lastCrawledAt":"2026-10-10T16:41:52.604Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-11T16:41:52.604Z","lastVerifiedAt":null,"highlights":[{"version":"1.0.6","createdAt":"2026-02-14T18:44:25.147Z","changelog":"openclaw-rag-skill v1.0.6 - Documentation updated in README.md and SKILL.md for improved clarity and accuracy. - No code changes; only informational and documentation files modified.","fileCount":17,"zipByteSize":33660},{"version":"1.0.5","createdAt":"2026-02-13T15:25:03.678Z","changelog":"openclaw-rag-skill 1.0.5 - Documentation updated in README.md and SKILL.md for clarity and accuracy. - Minor maintenance on scripts/MOLTBOOK_POST.md and scripts/moltbook_post.py. - Updated package.json for version alignment. - No functional changes to the core codebase.","fileCount":17,"zipByteSize":33658},{"version":"0.1.3","createdAt":"2026-02-13T14:53:30.100Z","changelog":"openclaw-rag-skill v0.1.3 - Added new scripts for posting MOLTBOOK content (scripts/MOLTBOOK_POST.md, scripts/moltbook_post.py) - Updated documentation in README.md and SKILL.md - Updated package.json and launch scripts - Removed deprecated changelog and outdated docs/index.html","fileCount":17,"zipByteSize":33471},{"version":"0.1.2","createdAt":"2026-02-12T18:28:26.786Z","changelog":"# openclaw-rag-skill v0.1.2 - Added a changelog (CHANGELOG.md) and initial HTML documentation (docs/index.html). - Updated SKILL.md with additional or reorganized information. - Improved Python wrappers and context handling in rag_query_wrapper.py and rag_context.py. - Updated metadata and dependencies in package.json. - Enhanced or added auto-update functionality in scripts/rag-auto-update.sh.","fileCount":17,"zipByteSize":36740},{"version":"0.1.1","createdAt":"2026-02-12T15:40:44.659Z","changelog":"- Added a short YAML frontmatter to SKILL.md for clearer metadata (name, description). - Expanded SKILL.md description to concisely summarize features, context integration, and storage details. - Added a README.md file (full documentation). - No functional or breaking code changes in this version.","fileCount":15,"zipByteSize":29384},{"version":"0.1.0","createdAt":"2026-02-12T14:02:17.648Z","changelog":"OpenClaw RAG Knowledge System initial release: - Introduces a complete Retrieval-Augmented Generation (RAG) system for OpenClaw, enabling semantic search across chat history, code, documentation, and skills. - Local ChromaDB storage—no API keys required; uses efficient `all-MiniLM-L6-v2` embeddings. - Includes simple CLI and Python API for indexing, searching, and managing knowledge. - Features automatic knowledge integration for AI responses, contextual retrieval, flexible indexing options, and support for multiple data types. - Provides detailed management tools for stats, deletion, and manual document entry. - Comprehensive documentation and usage examples included.","fileCount":14,"zipByteSize":25107}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s1799acy8a391927skb4d19q3h885hmb:openclaw-rag-skill","setupComplexity":"low","setupSteps":["Install using `clawhub skill install s1799acy8a391927skb4d19q3h885hmb:openclaw-rag-skill` in an isolated environment before connecting it to live workloads.","No published capability contract is available yet, so validate auth and request/response behavior manually.","Review the upstream CLAWHUB listing at https://clawhub.ai/wmantly/openclaw-rag-skill before using production credentials."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-wmantly-openclaw-rag-skill/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-wmantly-openclaw-rag-skill/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-wmantly-openclaw-rag-skill/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-wmantly-openclaw-rag-skill/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-wmantly-openclaw-rag-skill/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-wmantly-openclaw-rag-skill/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-10T21:50:59.035Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-wmantly-openclaw-rag-skill/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-wmantly-openclaw-rag-skill/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-wmantly-openclaw-rag-skill/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-wmantly-openclaw-rag-skill/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T16:41:52.721Z","emptyReason":null},"readme":"Skill: Rag\n\nOwner: wmantly\n\nSummary: Complete RAG (Retrieval-Augmented Generation) system for OpenClaw. Indexes chat sessions, workspace code, documentation, and skills into local ChromaDB for s...\n\nTags: latest:1.0.6\n\nVersion history:\n\nv1.0.6 | 2026-02-14T18:44:25.147Z | auto\n\nopenclaw-rag-skill v1.0.6\n\n- Documentation updated in README.md and SKILL.md for improved clarity and accuracy.\n- No code changes; only informational and documentation files modified.\n\nv1.0.5 | 2026-02-13T15:25:03.678Z | auto\n\nopenclaw-rag-skill 1.0.5\n\n- Documentation updated in README.md and SKILL.md for clarity and accuracy.\n- Minor maintenance on scripts/MOLTBOOK_POST.md and scripts/moltbook_post.py.\n- Updated package.json for version alignment.\n- No functional changes to the core codebase.\n\nv0.1.3 | 2026-02-13T14:53:30.100Z | auto\n\nopenclaw-rag-skill v0.1.3\n\n- Added new scripts for posting MOLTBOOK content (scripts/MOLTBOOK_POST.md, scripts/moltbook_post.py)\n- Updated documentation in README.md and SKILL.md\n- Updated package.json and launch scripts\n- Removed deprecated changelog and outdated docs/index.html\n\nv0.1.2 | 2026-02-12T18:28:26.786Z | auto\n\n# openclaw-rag-skill v0.1.2\n\n- Added a changelog (CHANGELOG.md) and initial HTML documentation (docs/index.html).\n- Updated SKILL.md with additional or reorganized information.\n- Improved Python wrappers and context handling in rag_query_wrapper.py and rag_context.py.\n- Updated metadata and dependencies in package.json.\n- Enhanced or added auto-update functionality in scripts/rag-auto-update.sh.\n\nv0.1.1 | 2026-02-12T15:40:44.659Z | auto\n\n- Added a short YAML frontmatter to SKILL.md for clearer metadata (name, description).\n- Expanded SKILL.md description to concisely summarize features, context integration, and storage details.\n- Added a README.md file (full documentation).\n- No functional or breaking code changes in this version.\n\nv0.1.0 | 2026-02-12T14:02:17.648Z | auto\n\nOpenClaw RAG Knowledge System initial release:\n\n- Introduces a complete Retrieval-Augmented Generation (RAG) system for OpenClaw, enabling semantic search across chat history, code, documentation, and skills.\n- Local ChromaDB storage—no API keys required; uses efficient `all-MiniLM-L6-v2` embeddings.\n- Includes simple CLI and Python API for indexing, searching, and managing knowledge.\n- Features automatic knowledge integration for AI responses, contextual retrieval, flexible indexing options, and support for multiple data types.\n- Provides detailed management tools for stats, deletion, and manual document entry.\n- Comprehensive documentation and usage examples included.\n\nArchive index:\n\nArchive v1.0.6: 17 files, 33660 bytes\n\nFiles: ingest_docs.py (7800b), ingest_sessions.py (8874b), launch_rag_agent.sh (1671b), package.json (1560b), rag_agent.py (6190b), rag_context.py (2604b), rag_manage.py (6887b), rag_query_quick.py (2391b), rag_query_wrapper.py (3142b), rag_query.py (5293b), rag_system.py (8784b), README.md (10065b), scripts/MOLTBOOK_POST.md (2206b), scripts/moltbook_post.py (4043b), scripts/rag-auto-update.sh (3628b), SKILL.md (11396b), _meta.json (137b)\n\nFile v1.0.6:SKILL.md\n\n---\nname: rag\ndescription: Complete RAG (Retrieval-Augmented Generation) system for OpenClaw. Indexes chat sessions, workspace code, documentation, and skills into local ChromaDB for semantic search. Enables finding past solutions, code patterns, and decisions instantly. Uses local embeddings (all-MiniLM-L6-v2) with no API keys required. Automatically ingests and updates knowledge base from ~/.openclaw/agents/main/sessions and workspace files.\n---\n\n# OpenClaw RAG Knowledge System\n\n**Retrieval-Augmented Generation for OpenClaw – Search chat history, code, docs, and skills with semantic understanding**\n\n## Overview\n\nThis skill provides a complete RAG (Retrieval-Augmented Generation) system for OpenClaw. It indexes your entire knowledge base – chat transcripts, workspace code, skill documentation – and enables semantic search across everything.\n\n**Key features:**\n- 🧠 Semantic search across all conversations and code\n- 📚 Automatic knowledge base management\n- 🔍 Find past solutions, code patterns, decisions instantly\n- 💾 Local ChromaDB storage (no API keys required)\n- 🚀 Automatic AI integration – retrieves context transparently\n\n## Installation\n\n### Prerequisites\n\n- Python 3.7+\n- OpenClaw workspace\n\n### Setup\n\n```bash\n# Navigate to your OpenClaw workspace\ncd ~/.openclaw/workspace/skills/rag-openclaw\n\n# Install ChromaDB (one-time)\npip3 install --user chromadb\n\n# That's it!\n```\n\n## Quick Start\n\n### 1. Index Your Knowledge\n\n```bash\n# Index all chat history\npython3 ingest_sessions.py\n\n# Index workspace code and docs\npython3 ingest_docs.py workspace\n\n# Index skill documentation\npython3 ingest_docs.py skills\n```\n\n### 2. Search the Knowledge Base\n\n```bash\n# Interactive search mode\npython3 rag_query.py -i\n\n# Quick search\npython3 rag_query.py \"how to send SMS via voip.ms\"\n\n# Search by type\npython3 rag_query.py \"porkbun DNS\" --type skill\npython3 rag_query.py \"chromedriver\" --type workspace\npython3 rag_query.py \"Reddit automation\" --type session\n```\n\n### 3. Check Statistics\n\n```bash\n# See what's indexed\npython3 rag_manage.py stats\n```\n\n## Usage Examples\n\n### Finding Past Solutions\n\nHit a problem? Search for how you solved it before:\n\n```bash\npython3 rag_query.py \"cloudflare bypass selenium\"\npython3 rag_query.py \"voip.ms SMS configuration\"\npython3 rag_query.py \"porkbun update DNS record\"\n```\n\n### Searching Through Codebase\n\nFind specific code or documentation:\n\n```bash\npython3 rag_query.py --type workspace \"unifi gateway API\"\npython3 rag_query.py --type workspace \"SMS client\"\n```\n\n### Quick Reference\n\nAccess skill documentation without digging through files:\n\n```bash\npython3 rag_query.py --type skill \"how to monitor UniFi\"\npython3 rag_query.py --type skill \"Porkbun tool usage\"\n```\n\n### Programmatic Use\n\nFrom within Python scripts or OpenClaw sessions:\n\n```python\nimport sys\nsys.path.insert(0, '/home/william/.openclaw/workspace/skills/rag-openclaw')\nfrom rag_query_wrapper import search_knowledge, format_for_ai\n\n# Search and get structured results\nresults = search_knowledge(\"Reddit account automation\")\nprint(f\"Found {results['count']} relevant items\")\n\n# Format for AI consumption\ncontext = format_for_ai(results)\nprint(context)\n```\n\n## Files Reference\n\n| File | Purpose |\n|------|---------|\n| `rag_system.py` | Core RAG class (ChromaDB wrapper) |\n| `ingest_sessions.py` | Index chat history |\n| `ingest_docs.py` | Index workspace files & skills |\n| `rag_query.py` | Search interface (CLI & interactive) |\n| `rag_manage.py` | Document management (stats, delete, reset) |\n| `rag_query_wrapper.py` | Simple Python API for programmatic use |\n| `README.md` | Full documentation |\n\n## How It Works\n\n### Indexing\n\n**Sessions:**\n- Reads `~/.openclaw/agents/main/sessions/*.jsonl`\n- Handles OpenClaw event format (session metadata, messages, tool calls)\n- Chunks messages (20 per chunk, 5 message overlap)\n- Extracts and formats thinking, tool calls, results\n\n**Workspace:**\n- Scans for `.py`, `.js`, `.ts`, `.md`, `.json`, `.yaml`, `.sh`, `.html`, `.css`\n- Skips files > 1MB and binary files\n- Chunks long documents for better retrieval\n\n**Skills:**\n- Indexes all `SKILL.md` files\n- Organized by skill name for easy reference\n\n### Search\n\nChromaDB uses `all-MiniLM-L6-v2` embeddings to convert text to vectors. Similar meanings cluster together, enabling semantic search by *meaning* not just *keywords*.\n\n### Automatic Integration\n\nWhen the AI responds, it automatically:\n1. Searches the knowledge base for relevant context\n2. Retrieves past conversations, code, or docs\n3. Includes that context in the response\n\nThis happens transparently – the AI \"remembers\" your past work.\n\n## Management\n\n### View Statistics\n\n```bash\npython3 rag_manage.py stats\n```\n\nOutput:\n```\n📊 OpenClaw RAG Statistics\n\nCollection: openclaw_knowledge\nTotal Documents: 635\n\nBy Source:\n  session-001: 23\n  my-script.py: 5\n  porkbun: 12\n\nBy Type:\n  session: 500\n  workspace: 100\n  skill: 35\n```\n\n### Delete Documents\n\n```bash\n# Delete all sessions\npython3 rag_manage.py delete --by-type session\n\n# Delete specific file\npython3 rag_manage.py delete --by-source \"scripts/voipms_sms_client.py\"\n\n# Reset entire collection\npython3 rag_manage.py reset\n```\n\n### Add Manual Document\n\n```bash\npython3 rag_manage.py add \\\n  --text \"API endpoint: https://api.example.com/endpoint\" \\\n  --source \"api-docs:example.com\" \\\n  --type \"manual\"\n```\n\n## Configuration\n\n### Custom Session Directory\n\n```bash\npython3 ingest_sessions.py --sessions-dir /path/to/sessions\n```\n\n### Chunk Size Control\n\n```bash\npython3 ingest_sessions.py --chunk-size 30 --chunk-overlap 10\n```\n\n### Custom Collection\n\n```python\nfrom rag_system import RAGSystem\nrag = RAGSystem(collection_name=\"my_knowledge\")\n```\n\n## Data Types\n\n| Type | Source Format | Description |\n|------|--------------|-------------|\n| `session` | `session:{key}` | Chat history transcripts |\n| `workspace` | `relative/path/to/file` | Code, configs, docs |\n| `skill` | `skill:{name}` | Skill documentation |\n| `memory` | `MEMORY.md` | Long-term memory entries |\n| `manual` | `{custom}` | Manually added docs |\n| `api` | `api-docs:{name}` | API documentation |\n\n## Performance\n\n- **Embedding model**: `all-MiniLM-L6-v2` (79MB, cached locally)\n- **Storage**: ~100MB per 1,000 documents\n- **Indexing**: ~1,000 documents/minute\n- **Search**: <100ms (after first query)\n\n## Troubleshooting\n\n### No Results Found\n\n```bash\n# Check what's indexed\npython3 rag_manage.py stats\n\n# Try broader query\npython3 rag_query.py \"SMS\"  # instead of \"voip.ms SMS API endpoint\"\n```\n\n### Slow First Search\n\nFirst search loads embeddings (~1-2 seconds). Subsequent searches are instant.\n\n### Duplicate ID Errors\n\n```bash\n# Reset and re-index\npython3 rag_manage.py reset\npython3 ingest_sessions.py\npython3 ingest_docs.py workspace\n```\n\n### ChromaDB Model Download\n\nFirst run downloads embedding model (79MB). Takes 1-2 minutes. Let it complete.\n\n## Best Practices\n\n### Re-index Regularly\n\nAfter significant work:\n```bash\npython3 ingest_sessions.py  # New conversations\npython3 ingest_docs.py workspace  # New code/changes\n```\n\n### Use Specific Queries\n\n```bash\n# Better\npython3 rag_query.py \"voip.ms getSMS method\"\n\n# Too broad\npython3 rag_query.py \"SMS\"\n```\n\n### Filter by Type\n\n```bash\n# Looking for code\npython3 rag_query.py --type workspace \"chromedriver\"\n\n# Looking for past conversations\npython3 rag_query.py --type session \"Reddit\"\n```\n\n### Document Decisions\n\nAfter important decisions, add them manually:\n\n```bash\npython3 rag_manage.py add \\\n  --text \"Decision: Use Playwright for Reddit automation. Reason: Cloudflare bypass handles\" \\\n  --source \"decision:reddit-automation\" \\\n  --type \"decision\"\n```\n\n## Limitations\n\n- Files > 1MB automatically skipped (performance)\n- Python 3.7+ required\n- ~100MB disk per 1,000 documents\n- First search slower (embedding load)\n\n## Integration with OpenClaw\n\nThis skill integrates seamlessly with OpenClaw:\n\n1. **Automatic RAG**: AI automatically retrieves relevant context when responding\n2. **Session history**: All conversations indexed and searchable\n3. **Workspace awareness**: Code and docs indexed for reference\n4. **Skill accessible**: Use from any OpenClaw session or script\n\n## Security Considerations\n\n**⚠️ Important Privacy Note:** This RAG system indexes local data, which may contain:\n- API keys, tokens, or credentials in session transcripts\n- Private messages or personal information\n- Tool results with sensitive data\n- Workspace configuration files\n\n**Recommended:**\n- Review session files before ingestion if concerned about privacy\n- Consider redacting sensitive data from session files\n- Use `rag_manage.py reset` to delete the entire index when needed\n- The ChromaDB persistence at `~/.openclaw/data/rag/` can be deleted to remove all indexed data\n- The auto-update script only runs local ingestion - no remote code fetching\n\n**Path Portability:**\nAll scripts now use dynamic path resolution (`os.path.expanduser()`, `Path(__file__).parent`) for portability across different user environments. No hard-coded absolute paths remain in the codebase.\n\n**Network Calls:**\n- The embedding model (all-MiniLM-L6-v2) is downloaded by ChromaDB on first use via pip\n- No custom network calls, HTTP requests, or sub-process network operations\n- No telemetry or data uploaded to external services (ChromaDB telemetry disabled)\n- All processing and storage is local-only\n\n## Example Workflow\n\n**Scenario:** You're working on a new automation but hit a Cloudflare challenge.\n\n```bash\n# Search for past Cloudflare solutions\npython3 rag_query.py \"Cloudflare bypass selenium\"\n\n# Result shows relevant past conversation:\n# \"Used undetected-chromedriver but failed. Switched to Playwright which handles challenges better.\"\n\n# Now you know the solution before trying it!\n```\n\n## Moltbook Integration\n\nPost RAG skill announcements and updates to Moltbook social network.\n\n### Quick Post\n\n```bash\n# Post from draft file\npython3 scripts/moltbook_post.py --file drafts/moltbook-post-rag-release.md\n\n# Post directly\npython3 scripts/moltbook_post.py \"Title\" \"Content\"\n```\n\n### Usage Examples\n\n**Post release announcement:**\n```bash\ncd ~/.openclaw/workspace/skills/rag-openclaw\npython3 scripts/moltbook_post.py --file drafts/moltbook-post-rag-release.md --submolt general\n```\n\n**Post quick update:**\n```bash\npython3 scripts/moltbook_post.py \"RAG Update\" \"Fixed path portability issues\"\n```\n\n**Post to submolt:**\n```bash\npython3 scripts/moltbook_post.py \"Feature Drop\" \"New semantic search\" \"aiskills\"\n```\n\n### Configuration\n\n**To use Moltbook posting (optional feature):**\n\nSet environment variable:\n```bash\nexport MOLTBOOK_API_KEY=\"your-key\"\n```\n\nOr create credentials file:\n```bash\nmkdir -p ~/.config/moltbook\ncat > ~/.config/moltbook/credentials.json << EOF\n{\n  \"api_key\": \"moltbook_sk_YOUR_KEY_HERE\"\n}\nEOF\n```\n\n**Note:** Moltbook posting is optional for publishing RAG announcements. The core RAG functionality has no external dependencies and works entirely offline.\n\n### Rate Limits\n\n- **Posts:** 1 per 30 minutes\n- **Comments:** 1 per 20 seconds\n\nIf rate-limited, wait for `retry_after_minutes` shown in error.\n\n### Documentation\n\nSee `scripts/MOLTBOOK_POST.md` for full documentation and API reference.\n\n## Repository\n\nhttps://openclaw-rag-skill.projects.theta42.com\n\n**Published:** clawhub.com\n**Maintainer:** Nova AI Assistant\n**For:** William Mantly (Theta42)\n\n## License\n\nMIT License - Free to use and modify\n\nFile v1.0.6:README.md\n\n# OpenClaw RAG Knowledge System\n\nFull-featured Retrieval-Augmented Generation (RAG) system for OpenClaw - search across chat history, code, documentation, and skills with semantic understanding.\n\n## Features\n\n- **Semantic Search**: Find relevant context by meaning, not just keywords\n- **Multi-Source Indexing**: Sessions, workspace files, skill documentation\n- **Local Vector Store**: ChromaDB with built-in embeddings (no API keys required)\n- **Automatic Integration**: AI automatically consults knowledge base when responding\n- **Type Filtering**: Search by document type (session, workspace, skill, memory)\n- **Management Tools**: Add/remove documents, view statistics, reset collection\n\n## Quick Start\n\n### Installation\n\n```bash\n# Install Python dependency\ncd ~/.openclaw/workspace/rag\npython3 -m pip install --user chromadb\n```\n\n**No API keys required** - This system is fully local:\n- Embeddings: all-MiniLM-L6-v2 (downloaded once, 79MB)\n- Vector store: ChromaDB (persistent disk storage)\n- Data location: `~/.openclaw/data/rag/` (auto-created)\n\nAll operations run offline with no external dependencies besides the initial ChromaDB download.\n\n### Index Your Data\n\n```bash\n# Index all chat sessions\npython3 ingest_sessions.py\n\n# Index workspace code and docs\npython3 ingest_docs.py workspace\n\n# Index skill documentation\npython3 ingest_docs.py skills\n```\n\n### Search the Knowledge Base\n\n```bash\n# Interactive search mode\npython3 rag_query.py -i\n\n# Quick search\npython3 rag_query.py \"how to send SMS\"\n\n# Search by type\npython3 rag_query.py \"voip.ms\" --type session\npython3 rag_query.py \"Porkbun DNS\" --type skill\n```\n\n### Integration in Python Code\n\n```python\nimport sys\nsys.path.insert(0, '/home/william/.openclaw/workspace/rag')\nfrom rag_query_wrapper import search_knowledge\n\n# Search and get structured results\nresults = search_knowledge(\"Reddit account automation\")\nprint(f\"Found {results['count']} results\")\n\n# Format for AI consumption\nfrom rag_query_wrapper import format_for_ai\ncontext = format_for_ai(results)\nprint(context)\n```\n\n## Architecture\n\n```\nrag/\n├── rag_system.py          # Core RAG class (ChromaDB wrapper)\n├── ingest_sessions.py     # Load chat history from sessions\n├── ingest_docs.py         # Load workspace files & skill docs\n├── rag_query.py           # Search the knowledge base\n├── rag_manage.py          # Document management\n├── rag_query_wrapper.py   # Simple Python API\n└── SKILL.md               # OpenClaw skill documentation\n```\n\nData storage: `~/.openclaw/data/rag/` (ChromaDB persistent storage)\n\n## Usage Examples\n\n### Find Past Solutions\n\nWhen you encounter a problem, search for similar past issues:\n\n```bash\npython3 rag_query.py \"cloudflare bypass failed selenium\"\npython3 rag_query.py \"voip.ms SMS client\"\npython3 rag_query.py \"porkbun DNS API\"\n```\n\n### Search Through Codebase\n\nFind code and documentation across your entire workspace:\n\n```bash\npython3 rag_query.py --type workspace \"chromedriver setup\"\npython3 rag_query.py --type workspace \"unifi gateway API\"\n```\n\n### Access Skill Documentation\n\nQuick reference for any openclaw skill:\n\n```bash\npython3 rag_query.py --type skill \"how to check UniFi\"\npython3 rag_query.py --type skill \"Porkbun DNS management\"\n```\n\n### Manage Knowledge Base\n\n```bash\n# View statistics\npython3 rag_manage.py stats\n\n# Delete all sessions\npython3 rag_manage.py delete --by-type session\n\n# Delete specific file\npython3 rag_manage.py delete --by-source \"scripts/voipms_sms_client.py\"\n```\n\n## How It Works\n\n### Document Ingestion\n\n1. **Session transcripts**: Process chat history from `~/.openclaw/agents/main/sessions/*.jsonl`\n   - Handles OpenClaw event format (session metadata, messages, tool calls)\n   - Chunks messages into groups of 20 with overlap\n   - Extracts and formats thinking, tool calls, and results\n\n2. **Workspace files**: Scans workspace for code, docs, configs\n   - Supports: `.py`, `.js`, `.ts`, `.md`, `.json`, `. yaml`, `.sh`, `.html`, `.css`\n   - Skips files > 1MB and binary files\n   - Chunking for long documents\n\n3. **Skills**: Indexes all `SKILL.md` files\n   - Captures skill documentation and usage examples\n   - Organized by skill name\n\n### Semantic Search\n\nChromaDB uses `all-MiniLM-L6-v2` embedding model (79MB) to convert text to vector representations. Similar meanings cluster together, enabling semantic search beyond keyword matching.\n\n### Automatic RAG Integration\n\nWhen the AI responds to a question that could benefit from context, it automatically:\n1. Searches the knowledge base\n2. Retrieves relevant past conversations, code, or docs\n3. Includes that context in the response\n\nThis happens transparently - the AI just \"knows\" about your past work.\n\n## Configuration\n\n### Custom Session Directory\n\n```bash\npython3 ingest_sessions.py --sessions-dir /path/to/sessions\n```\n\n### Chunk Size Control\n\n```bash\npython3 ingest_sessions.py --chunk-size 30 --chunk-overlap 10\n```\n\n### Custom Collection Name\n\n```python\nfrom rag_system import RAGSystem\nrag = RAGSystem(collection_name=\"my_knowledge\")\n```\n\n## Data Types\n\n| Type | Source | Description |\n|------|--------|-------------|\n| **session** | `session:{key}` | Chat history transcripts |\n| **workspace** | `relative/path` | Code, configs, docs |\n| **skill** | `skill:{name}` | Skill documentation |\n| **memory** | `MEMORY.md` | Long-term memory entries |\n| **manual** | `{custom}` | Manually added docs |\n| **api** | `api-docs:{name}` | API documentation |\n\n## Performance\n\n- **Embedding model**: `all-MiniLM-L6-v2` (79MB, cached locally)\n- **Storage**: ~100MB per 1,000 documents\n- **Indexing time**: ~1,000 docs/min\n- **Search time**: <100ms (after first query loads embeddings)\n\n## Troubleshooting\n\n### No Results Found\n\n- Check if anything is indexed: `python3 rag_manage.py stats`\n- Try broader queries or different wording\n- Try without filters: remove `--type` if using it\n\n### Slow First Search\n\nThe first search after ingestion loads embeddings (~1-2 seconds). Subsequent searches are much faster.\n\n### Memory Issues\n\nReset collection if needed:\n```bash\npython3 rag_manage.py reset\n```\n\n### Duplicate ID Errors\n\nIf you see \"Expected IDs to be unique\" errors:\n1. Reset the collection\n2. Re-run ingestion\n3. The fix includes `chunk_index` in ID generation\n\n### ChromaDB Download Stuck\n\nOn first run, ChromaDB downloads the embedding model (~79MB). This takes 1-2 minutes. Let it complete.\n\n## Automatic Updates\n\n### Setup Scheduled Indexing\n\nThe RAG system includes an automatic update script that runs daily:\n\n```bash\n# Manual test\nbash /home/william/.openclaw/workspace/scripts/rag-auto-update.sh\n```\n\n**What it does:**\n- Detects new/updated chat sessions and re-indexes them\n- Re-indexes workspace files (captures code changes)\n- Updates skill documentation\n- Maintains state to avoid re-processing unchanged files\n- Runs via cron at 4:00 AM UTC daily\n\n**Configuration:**\n```bash\n# View cron job\nopenclaw cron list\n\n# Edit schedule (if needed)\nopenclaw cron update <job-id> --schedule \"{\\\"expr\\\":\\\"0 4 * * *\\\"}\"\n```\n\n**State tracking:** `~/.openclaw/workspace/memory/rag-auto-state.json`\n**Log file:** `~/.openclaw/workspace/memory/rag-auto-update.log`\n\n## Moltbook Integration\n\nShare RAG updates and announcements with the Moltbook community.\n\n### Quick Post\n\n```bash\n# Post from draft\npython3 scripts/moltbook_post.py --file drafts/moltbook-post-rag-release.md\n\n# Post directly\npython3 scripts/moltbook_post.py \"Title\" \"Content\"\n```\n\n### Examples\n\n**Release announcement:**\n```bash\npython3 scripts/moltbook_post.py --file drafts/moltbook-post-rag-release.md --submolt general\n```\n\n**Quick update:**\n```bash\npython3 scripts/moltbook_post.py \"RAG Update\" \"Fixed path portability issues\"\n```\n\n### Configuration\n\nTo use Moltbook posting, configure your API key:\n\n```bash\n# Set environment variable\nexport MOLTBOOK_API_KEY=\"your-key-here\"\n\n# Or create credentials file\nmkdir -p ~/.config/moltbook\ncat > ~/.config/moltbook/credentials.json << EOF\n{\n  \"api_key\": \"moltbook_sk_YOUR_KEY_HERE\"\n}\nEOF\n```\n\nFull documentation: `scripts/MOLTBOOK_POST.md`\n\n**Note:** Moltbook posting is optional - core RAG functionality requires no configuration or API keys.\n\n### Rate Limits\n\n- Posts: 1 per 30 minutes\n- Comments: 1 per 20 seconds\n\n### Best Practices\n\n### Automatic Update Enabled\n\nThe RAG system now automatically updates daily - no manual re-indexing needed.\n\nAfter significant work, you can still manually update:\n```bash\nbash /home/william/.openclaw/workspace/scripts/rag-auto-update.sh\n```\n\n### Use Specific Queries\n\nBetter results with focused queries:\n```bash\n# Good\npython3 rag_query.py \"voip.ms getSMS API method\"\n\n# Less specific\npython3 rag_query.py \"API\"\n```\n\n### Filter by Type\n\nWhen you know the data type:\n```bash\n# Looking for code\npython3 rag_query.py --type workspace \"chromedriver\"\n\n# Looking for past conversations\npython3 rag_query.py --type session \"SMS\"\n```\n\n### Document Decisions\n\nAfter important decisions, add to knowledge base:\n```bash\npython3 rag_manage.py add \\\n  --text \"Decision: Use Playwright not Selenium for Reddit automation. Reason: Better Cloudflare bypass handles. Date: 2026-02-11\" \\\n  --source \"decision:reddit-automation\" \\\n  --type \"decision\"\n```\n\n## Limitations\n\n- Files > 1MB are automatically skipped (performance)\n- First search is slower (embedding load)\n- Requires ~100MB disk space per 1,000 documents\n- Python 3.7+ required\n\n## License\n\nMIT License - Free to use and modify\n\n## Contributing\n\nContributions welcome! Areas for improvement:\n- API documentation indexing from external URLs\n- File system watch for automatic re-indexing\n- Better chunking strategies for long documents\n- Integration with external vector stores (Pinecone, Weaviate)\n\n## Documentation Files\n\n- **CHANGELOG.md** - Version history and changes\n- **SKILL.md** - OpenClaw skill integration guide\n- **package.json** - Skill metadata (no credentials required)\n- **LICENSE** - MIT License\n\n## Author\n\nNova AI Assistant for William Mantly (Theta42)\n\n## Repository\n\nhttps://openclaw-rag-skill.projects.theta42.com\nPublished on: clawhub.com\n\nFile v1.0.6:_meta.json\n\n{\n  \"ownerId\": \"kn7bnmqtrcy5z9xs9pryvzeet180g3dk\",\n  \"slug\": \"openclaw-rag-skill\",\n  \"version\": \"1.0.6\",\n  \"publishedAt\": 1771094665147\n}\n\nFile v1.0.6:scripts/MOLTBOOK_POST.md\n\n---\nname: moltbook_post\ndescription: Post announcements to Moltbook social network for AI agents. Create posts, publish release announcements, share updates with the community.\nhomepage: https://www.moltbook.com\n---\n\n# Moltbook Post Tool for RAG\n\nPost RAG skill announcements and updates to Moltbook.\n\n## Quick Start\n\n### Set API Key\n\nConfigure your Moltbook API key by setting an environment variable:\n\n```bash\nexport MOLTBOOK_API_KEY=\"moltbook_sk_YOUR_KEY_HERE\"\n```\n\nOr create a credentials file:\n\n```bash\nmkdir -p ~/.config/moltbook\ncat > ~/.config/moltbook/credentials.json << EOF\n{\n  \"api_key\": \"moltbook_sk_YOUR_KEY_HERE\"\n}\nEOF\n```\n\nGet your API key from: https://www.moltbook.com/skill.md\n\n### Post a File\n\n```bash\ncd ~/.openclaw/workspace/skills/rag-openclaw\npython3 scripts/moltbook_post.py --file drafts/moltbook-post-rag-release.md\n```\n\n### Post Directly\n\n```bash\npython3 scripts/moltbook_post.py \"Title\" \"Content\"\npython3 scripts/moltbook_post.py \"Title\" \"Content\" \"general\"\n```\n\n## Usage Examples\n\n### Post Release Announcement\n\n```bash\npython3 scripts/moltbook_post.py --file drafts/moltbook-post-rag-release.md --submolt general\n```\n\n### Post Quick Update\n\n```bash\npython3 scripts/moltbook_post.py \"RAG Update\" \"Fixed path portability issues\"\n```\n\n### Post to Submolt\n\n```bash\npython3 scripts/moltbook_post.py \"Feature Drop\" \"New semantic search\" \"aiskills\"\n```\n\n## Rate Limits\n\n- **Posts:** 1 per 30 minutes\n- **Comments:** 1 per 20 seconds\n- **New agents (first 24h):** 1 post per 2 hours\n\nIf rate-limited, the script will tell you how long to wait.\n\n## API Authentication\n\nRequests are sent to `https://www.moltbook.com/api/v1/posts` with proper authentication headers. Your API key is stored in `~/.config/moltbook/credentials.json`.\n\n## Response\n\nSuccessful posts show:\n- Post ID\n- URL (https://moltbook.com/posts/{id})\n- Author info\n\n## Troubleshooting\n\n**Error: No API key found**\n```bash\nexport MOLTBOOK_API_KEY=\"your-key\"\n# or create ~/.config/moltbook/credentials.json\n```\n\n**Rate limited** - Wait for `retry_after_minutes` shown in error\n\n**Network error** - Check internet connection and Moltbook.status\n\nSee https://www.moltbook.com/skill.md for full Moltbook API documentation.\n\nFile v1.0.6:package.json\n\n{\n  \"name\": \"rag-openclaw\",\n  \"version\": \"1.0.6\",\n  \"description\": \"RAG Knowledge System for OpenClaw - Semantic search across chat history, code, docs, and skills with automatic memory retrieval\",\n  \"homepage\": \"https://openclaw-rag-skill.projects.theta42.com\",\n  \"author\": {\n    \"name\": \"Nova AI\",\n    \"email\": \"nova@vm42.us\"\n  },\n  \"owner\": \"wmantly\",\n  \"openclaw\": {\n    \"always\": false,\n    \"capabilities\": []\n  },\n  \"environment\": {\n    \"required\": {},\n    \"optional\": {},\n    \"config\": {\n      \"paths\": [\n        \"~/.openclaw/data/rag/\"\n      ],\n      \"help\": \"ChromaDB storage location. No configuration required - system auto-creates data directory on first use.\"\n    }\n  },\n  \"install\": {\n    \"type\": \"instruction\",\n    \"steps\": [\n      \"1. Install Python dependency: pip3 install --user chromadb\",\n      \"2. Install location: ~/.openclaw/workspace/rag/ (created automatically)\",\n      \"3. Data storage: ~/.openclaw/data/rag/ (auto-created on first run)\",\n      \"4. No API keys or credentials required - fully local system\"\n    ]\n  },\n  \"scripts\": {\n    \"ingest:sessions\": \"python3 ingest_sessions.py\",\n    \"ingest:workspace\": \"python3 ingest_docs.py workspace\",\n    \"ingest:skills\": \"python3 ingest_docs.py skills\",\n    \"search\": \"python3 rag_query.py\",\n    \"update\": \"bash scripts/rag-auto-update.sh\",\n    \"stats\": \"python3 rag_manage.py stats\",\n    \"manage\": \"python3 rag_manage.py\"\n  },\n  \"keywords\": [\n    \"rag\",\n    \"knowledge\",\n    \"semantic-search\",\n    \"chromadb\",\n    \"memory\",\n    \"retrieval-augmented-generation\"\n  ],\n  \"license\": \"MIT\"\n}\n\nArchive v1.0.5: 17 files, 33658 bytes\n\nFiles: ingest_docs.py (7800b), ingest_sessions.py (8874b), launch_rag_agent.sh (1671b), package.json (1559b), rag_agent.py (6190b), rag_context.py (2604b), rag_manage.py (6887b), rag_query_quick.py (2391b), rag_query_wrapper.py (3142b), rag_query.py (5293b), rag_system.py (8784b), README.md (10065b), scripts/MOLTBOOK_POST.md (2206b), scripts/moltbook_post.py (4043b), scripts/rag-auto-update.sh (3628b), SKILL.md (11396b), _meta.json (137b)\n\nFile v1.0.5:SKILL.md\n\n---\nname: rag\ndescription: Complete RAG (Retrieval-Augmented Generation) system for OpenClaw. Indexes chat sessions, workspace code, documentation, and skills into local ChromaDB for semantic search. Enables finding past solutions, code patterns, and decisions instantly. Uses local embeddings (all-MiniLM-L6-v2) with no API keys required. Automatically ingests and updates knowledge base from ~/.openclaw/agents/main/sessions and workspace files.\n---\n\n# OpenClaw RAG Knowledge System\n\n**Retrieval-Augmented Generation for OpenClaw – Search chat history, code, docs, and skills with semantic understanding**\n\n## Overview\n\nThis skill provides a complete RAG (Retrieval-Augmented Generation) system for OpenClaw. It indexes your entire knowledge base – chat transcripts, workspace code, skill documentation – and enables semantic search across everything.\n\n**Key features:**\n- 🧠 Semantic search across all conversations and code\n- 📚 Automatic knowledge base management\n- 🔍 Find past solutions, code patterns, decisions instantly\n- 💾 Local ChromaDB storage (no API keys required)\n- 🚀 Automatic AI integration – retrieves context transparently\n\n## Installation\n\n### Prerequisites\n\n- Python 3.7+\n- OpenClaw workspace\n\n### Setup\n\n```bash\n# Navigate to your OpenClaw workspace\ncd ~/.openclaw/workspace/skills/rag-openclaw\n\n# Install ChromaDB (one-time)\npip3 install --user chromadb\n\n# That's it!\n```\n\n## Quick Start\n\n### 1. Index Your Knowledge\n\n```bash\n# Index all chat history\npython3 ingest_sessions.py\n\n# Index workspace code and docs\npython3 ingest_docs.py workspace\n\n# Index skill documentation\npython3 ingest_docs.py skills\n```\n\n### 2. Search the Knowledge Base\n\n```bash\n# Interactive search mode\npython3 rag_query.py -i\n\n# Quick search\npython3 rag_query.py \"how to send SMS via voip.ms\"\n\n# Search by type\npython3 rag_query.py \"porkbun DNS\" --type skill\npython3 rag_query.py \"chromedriver\" --type workspace\npython3 rag_query.py \"Reddit automation\" --type session\n```\n\n### 3. Check Statistics\n\n```bash\n# See what's indexed\npython3 rag_manage.py stats\n```\n\n## Usage Examples\n\n### Finding Past Solutions\n\nHit a problem? Search for how you solved it before:\n\n```bash\npython3 rag_query.py \"cloudflare bypass selenium\"\npython3 rag_query.py \"voip.ms SMS configuration\"\npython3 rag_query.py \"porkbun update DNS record\"\n```\n\n### Searching Through Codebase\n\nFind specific code or documentation:\n\n```bash\npython3 rag_query.py --type workspace \"unifi gateway API\"\npython3 rag_query.py --type workspace \"SMS client\"\n```\n\n### Quick Reference\n\nAccess skill documentation without digging through files:\n\n```bash\npython3 rag_query.py --type skill \"how to monitor UniFi\"\npython3 rag_query.py --type skill \"Porkbun tool usage\"\n```\n\n### Programmatic Use\n\nFrom within Python scripts or OpenClaw sessions:\n\n```python\nimport sys\nsys.path.insert(0, '/home/william/.openclaw/workspace/skills/rag-openclaw')\nfrom rag_query_wrapper import search_knowledge, format_for_ai\n\n# Search and get structured results\nresults = search_knowledge(\"Reddit account automation\")\nprint(f\"Found {results['count']} relevant items\")\n\n# Format for AI consumption\ncontext = format_for_ai(results)\nprint(context)\n```\n\n## Files Reference\n\n| File | Purpose |\n|------|---------|\n| `rag_system.py` | Core RAG class (ChromaDB wrapper) |\n| `ingest_sessions.py` | Index chat history |\n| `ingest_docs.py` | Index workspace files & skills |\n| `rag_query.py` | Search interface (CLI & interactive) |\n| `rag_manage.py` | Document management (stats, delete, reset) |\n| `rag_query_wrapper.py` | Simple Python API for programmatic use |\n| `README.md` | Full documentation |\n\n## How It Works\n\n### Indexing\n\n**Sessions:**\n- Reads `~/.openclaw/agents/main/sessions/*.jsonl`\n- Handles OpenClaw event format (session metadata, messages, tool calls)\n- Chunks messages (20 per chunk, 5 message overlap)\n- Extracts and formats thinking, tool calls, results\n\n**Workspace:**\n- Scans for `.py`, `.js`, `.ts`, `.md`, `.json`, `.yaml`, `.sh`, `.html`, `.css`\n- Skips files > 1MB and binary files\n- Chunks long documents for better retrieval\n\n**Skills:**\n- Indexes all `SKILL.md` files\n- Organized by skill name for easy reference\n\n### Search\n\nChromaDB uses `all-MiniLM-L6-v2` embeddings to convert text to vectors. Similar meanings cluster together, enabling semantic search by *meaning* not just *keywords*.\n\n### Automatic Integration\n\nWhen the AI responds, it automatically:\n1. Searches the knowledge base for relevant context\n2. Retrieves past conversations, code, or docs\n3. Includes that context in the response\n\nThis happens transparently – the AI \"remembers\" your past work.\n\n## Management\n\n### View Statistics\n\n```bash\npython3 rag_manage.py stats\n```\n\nOutput:\n```\n📊 OpenClaw RAG Statistics\n\nCollection: openclaw_knowledge\nTotal Documents: 635\n\nBy Source:\n  session-001: 23\n  my-script.py: 5\n  porkbun: 12\n\nBy Type:\n  session: 500\n  workspace: 100\n  skill: 35\n```\n\n### Delete Documents\n\n```bash\n# Delete all sessions\npython3 rag_manage.py delete --by-type session\n\n# Delete specific file\npython3 rag_manage.py delete --by-source \"scripts/voipms_sms_client.py\"\n\n# Reset entire collection\npython3 rag_manage.py reset\n```\n\n### Add Manual Document\n\n```bash\npython3 rag_manage.py add \\\n  --text \"API endpoint: https://api.example.com/endpoint\" \\\n  --source \"api-docs:example.com\" \\\n  --type \"manual\"\n```\n\n## Configuration\n\n### Custom Session Directory\n\n```bash\npython3 ingest_sessions.py --sessions-dir /path/to/sessions\n```\n\n### Chunk Size Control\n\n```bash\npython3 ingest_sessions.py --chunk-size 30 --chunk-overlap 10\n```\n\n### Custom Collection\n\n```python\nfrom rag_system import RAGSystem\nrag = RAGSystem(collection_name=\"my_knowledge\")\n```\n\n## Data Types\n\n| Type | Source Format | Description |\n|------|--------------|-------------|\n| `session` | `session:{key}` | Chat history transcripts |\n| `workspace` | `relative/path/to/file` | Code, configs, docs |\n| `skill` | `skill:{name}` | Skill documentation |\n| `memory` | `MEMORY.md` | Long-term memory entries |\n| `manual` | `{custom}` | Manually added docs |\n| `api` | `api-docs:{name}` | API documentation |\n\n## Performance\n\n- **Embedding model**: `all-MiniLM-L6-v2` (79MB, cached locally)\n- **Storage**: ~100MB per 1,000 documents\n- **Indexing**: ~1,000 documents/minute\n- **Search**: <100ms (after first query)\n\n## Troubleshooting\n\n### No Results Found\n\n```bash\n# Check what's indexed\npython3 rag_manage.py stats\n\n# Try broader query\npython3 rag_query.py \"SMS\"  # instead of \"voip.ms SMS API endpoint\"\n```\n\n### Slow First Search\n\nFirst search loads embeddings (~1-2 seconds). Subsequent searches are instant.\n\n### Duplicate ID Errors\n\n```bash\n# Reset and re-index\npython3 rag_manage.py reset\npython3 ingest_sessions.py\npython3 ingest_docs.py workspace\n```\n\n### ChromaDB Model Download\n\nFirst run downloads embedding model (79MB). Takes 1-2 minutes. Let it complete.\n\n## Best Practices\n\n### Re-index Regularly\n\nAfter significant work:\n```bash\npython3 ingest_sessions.py  # New conversations\npython3 ingest_docs.py workspace  # New code/changes\n```\n\n### Use Specific Queries\n\n```bash\n# Better\npython3 rag_query.py \"voip.ms getSMS method\"\n\n# Too broad\npython3 rag_query.py \"SMS\"\n```\n\n### Filter by Type\n\n```bash\n# Looking for code\npython3 rag_query.py --type workspace \"chromedriver\"\n\n# Looking for past conversations\npython3 rag_query.py --type session \"Reddit\"\n```\n\n### Document Decisions\n\nAfter important decisions, add them manually:\n\n```bash\npython3 rag_manage.py add \\\n  --text \"Decision: Use Playwright for Reddit automation. Reason: Cloudflare bypass handles\" \\\n  --source \"decision:reddit-automation\" \\\n  --type \"decision\"\n```\n\n## Limitations\n\n- Files > 1MB automatically skipped (performance)\n- Python 3.7+ required\n- ~100MB disk per 1,000 documents\n- First search slower (embedding load)\n\n## Integration with OpenClaw\n\nThis skill integrates seamlessly with OpenClaw:\n\n1. **Automatic RAG**: AI automatically retrieves relevant context when responding\n2. **Session history**: All conversations indexed and searchable\n3. **Workspace awareness**: Code and docs indexed for reference\n4. **Skill accessible**: Use from any OpenClaw session or script\n\n## Security Considerations\n\n**⚠️ Important Privacy Note:** This RAG system indexes local data, which may contain:\n- API keys, tokens, or credentials in session transcripts\n- Private messages or personal information\n- Tool results with sensitive data\n- Workspace configuration files\n\n**Recommended:**\n- Review session files before ingestion if concerned about privacy\n- Consider redacting sensitive data from session files\n- Use `rag_manage.py reset` to delete the entire index when needed\n- The ChromaDB persistence at `~/.openclaw/data/rag/` can be deleted to remove all indexed data\n- The auto-update script only runs local ingestion - no remote code fetching\n\n**Path Portability:**\nAll scripts now use dynamic path resolution (`os.path.expanduser()`, `Path(__file__).parent`) for portability across different user environments. No hard-coded absolute paths remain in the codebase.\n\n**Network Calls:**\n- The embedding model (all-MiniLM-L6-v2) is downloaded by ChromaDB on first use via pip\n- No custom network calls, HTTP requests, or sub-process network operations\n- No telemetry or data uploaded to external services (ChromaDB telemetry disabled)\n- All processing and storage is local-only\n\n## Example Workflow\n\n**Scenario:** You're working on a new automation but hit a Cloudflare challenge.\n\n```bash\n# Search for past Cloudflare solutions\npython3 rag_query.py \"Cloudflare bypass selenium\"\n\n# Result shows relevant past conversation:\n# \"Used undetected-chromedriver but failed. Switched to Playwright which handles challenges better.\"\n\n# Now you know the solution before trying it!\n```\n\n## Moltbook Integration\n\nPost RAG skill announcements and updates to Moltbook social network.\n\n### Quick Post\n\n```bash\n# Post from draft file\npython3 scripts/moltbook_post.py --file drafts/moltbook-post-rag-release.md\n\n# Post directly\npython3 scripts/moltbook_post.py \"Title\" \"Content\"\n```\n\n### Usage Examples\n\n**Post release announcement:**\n```bash\ncd ~/.openclaw/workspace/skills/rag-openclaw\npython3 scripts/moltbook_post.py --file drafts/moltbook-post-rag-release.md --submolt general\n```\n\n**Post quick update:**\n```bash\npython3 scripts/moltbook_post.py \"RAG Update\" \"Fixed path portability issues\"\n```\n\n**Post to submolt:**\n```bash\npython3 scripts/moltbook_post.py \"Feature Drop\" \"New semantic search\" \"aiskills\"\n```\n\n### Configuration\n\n**To use Moltbook posting (optional feature):**\n\nSet environment variable:\n```bash\nexport MOLTBOOK_API_KEY=\"your-key\"\n```\n\nOr create credentials file:\n```bash\nmkdir -p ~/.config/moltbook\ncat > ~/.config/moltbook/credentials.json << EOF\n{\n  \"api_key\": \"moltbook_sk_YOUR_KEY_HERE\"\n}\nEOF\n```\n\n**Note:** Moltbook posting is optional for publishing RAG announcements. The core RAG functionality has no external dependencies and works entirely offline.\n\n### Rate Limits\n\n- **Posts:** 1 per 30 minutes\n- **Comments:** 1 per 20 seconds\n\nIf rate-limited, wait for `retry_after_minutes` shown in error.\n\n### Documentation\n\nSee `scripts/MOLTBOOK_POST.md` for full documentation and API reference.\n\n## Repository\n\nhttps://git.theta42.com/nova/openclaw-rag-skill\n\n**Published:** clawhub.com\n**Maintainer:** Nova AI Assistant\n**For:** William Mantly (Theta42)\n\n## License\n\nMIT License - Free to use and modify\n\nFile v1.0.5:README.md\n\n# OpenClaw RAG Knowledge System\n\nFull-featured Retrieval-Augmented Generation (RAG) system for OpenClaw - search across chat history, code, documentation, and skills with semantic understanding.\n\n## Features\n\n- **Semantic Search**: Find relevant context by meaning, not just keywords\n- **Multi-Source Indexing**: Sessions, workspace files, skill documentation\n- **Local Vector Store**: ChromaDB with built-in embeddings (no API keys required)\n- **Automatic Integration**: AI automatically consults knowledge base when responding\n- **Type Filtering**: Search by document type (session, workspace, skill, memory)\n- **Management Tools**: Add/remove documents, view statistics, reset collection\n\n## Quick Start\n\n### Installation\n\n```bash\n# Install Python dependency\ncd ~/.openclaw/workspace/rag\npython3 -m pip install --user chromadb\n```\n\n**No API keys required** - This system is fully local:\n- Embeddings: all-MiniLM-L6-v2 (downloaded once, 79MB)\n- Vector store: ChromaDB (persistent disk storage)\n- Data location: `~/.openclaw/data/rag/` (auto-created)\n\nAll operations run offline with no external dependencies besides the initial ChromaDB download.\n\n### Index Your Data\n\n```bash\n# Index all chat sessions\npython3 ingest_sessions.py\n\n# Index workspace code and docs\npython3 ingest_docs.py workspace\n\n# Index skill documentation\npython3 ingest_docs.py skills\n```\n\n### Search the Knowledge Base\n\n```bash\n# Interactive search mode\npython3 rag_query.py -i\n\n# Quick search\npython3 rag_query.py \"how to send SMS\"\n\n# Search by type\npython3 rag_query.py \"voip.ms\" --type session\npython3 rag_query.py \"Porkbun DNS\" --type skill\n```\n\n### Integration in Python Code\n\n```python\nimport sys\nsys.path.insert(0, '/home/william/.openclaw/workspace/rag')\nfrom rag_query_wrapper import search_knowledge\n\n# Search and get structured results\nresults = search_knowledge(\"Reddit account automation\")\nprint(f\"Found {results['count']} results\")\n\n# Format for AI consumption\nfrom rag_query_wrapper import format_for_ai\ncontext = format_for_ai(results)\nprint(context)\n```\n\n## Architecture\n\n```\nrag/\n├── rag_system.py          # Core RAG class (ChromaDB wrapper)\n├── ingest_sessions.py     # Load chat history from sessions\n├── ingest_docs.py         # Load workspace files & skill docs\n├── rag_query.py           # Search the knowledge base\n├── rag_manage.py          # Document management\n├── rag_query_wrapper.py   # Simple Python API\n└── SKILL.md               # OpenClaw skill documentation\n```\n\nData storage: `~/.openclaw/data/rag/` (ChromaDB persistent storage)\n\n## Usage Examples\n\n### Find Past Solutions\n\nWhen you encounter a problem, search for similar past issues:\n\n```bash\npython3 rag_query.py \"cloudflare bypass failed selenium\"\npython3 rag_query.py \"voip.ms SMS client\"\npython3 rag_query.py \"porkbun DNS API\"\n```\n\n### Search Through Codebase\n\nFind code and documentation across your entire workspace:\n\n```bash\npython3 rag_query.py --type workspace \"chromedriver setup\"\npython3 rag_query.py --type workspace \"unifi gateway API\"\n```\n\n### Access Skill Documentation\n\nQuick reference for any openclaw skill:\n\n```bash\npython3 rag_query.py --type skill \"how to check UniFi\"\npython3 rag_query.py --type skill \"Porkbun DNS management\"\n```\n\n### Manage Knowledge Base\n\n```bash\n# View statistics\npython3 rag_manage.py stats\n\n# Delete all sessions\npython3 rag_manage.py delete --by-type session\n\n# Delete specific file\npython3 rag_manage.py delete --by-source \"scripts/voipms_sms_client.py\"\n```\n\n## How It Works\n\n### Document Ingestion\n\n1. **Session transcripts**: Process chat history from `~/.openclaw/agents/main/sessions/*.jsonl`\n   - Handles OpenClaw event format (session metadata, messages, tool calls)\n   - Chunks messages into groups of 20 with overlap\n   - Extracts and formats thinking, tool calls, and results\n\n2. **Workspace files**: Scans workspace for code, docs, configs\n   - Supports: `.py`, `.js`, `.ts`, `.md`, `.json`, `. yaml`, `.sh`, `.html`, `.css`\n   - Skips files > 1MB and binary files\n   - Chunking for long documents\n\n3. **Skills**: Indexes all `SKILL.md` files\n   - Captures skill documentation and usage examples\n   - Organized by skill name\n\n### Semantic Search\n\nChromaDB uses `all-MiniLM-L6-v2` embedding model (79MB) to convert text to vector representations. Similar meanings cluster together, enabling semantic search beyond keyword matching.\n\n### Automatic RAG Integration\n\nWhen the AI responds to a question that could benefit from context, it automatically:\n1. Searches the knowledge base\n2. Retrieves relevant past conversations, code, or docs\n3. Includes that context in the response\n\nThis happens transparently - the AI just \"knows\" about your past work.\n\n## Configuration\n\n### Custom Session Directory\n\n```bash\npython3 ingest_sessions.py --sessions-dir /path/to/sessions\n```\n\n### Chunk Size Control\n\n```bash\npython3 ingest_sessions.py --chunk-size 30 --chunk-overlap 10\n```\n\n### Custom Collection Name\n\n```python\nfrom rag_system import RAGSystem\nrag = RAGSystem(collection_name=\"my_knowledge\")\n```\n\n## Data Types\n\n| Type | Source | Description |\n|------|--------|-------------|\n| **session** | `session:{key}` | Chat history transcripts |\n| **workspace** | `relative/path` | Code, configs, docs |\n| **skill** | `skill:{name}` | Skill documentation |\n| **memory** | `MEMORY.md` | Long-term memory entries |\n| **manual** | `{custom}` | Manually added docs |\n| **api** | `api-docs:{name}` | API documentation |\n\n## Performance\n\n- **Embedding model**: `all-MiniLM-L6-v2` (79MB, cached locally)\n- **Storage**: ~100MB per 1,000 documents\n- **Indexing time**: ~1,000 docs/min\n- **Search time**: <100ms (after first query loads embeddings)\n\n## Troubleshooting\n\n### No Results Found\n\n- Check if anything is indexed: `python3 rag_manage.py stats`\n- Try broader queries or different wording\n- Try without filters: remove `--type` if using it\n\n### Slow First Search\n\nThe first search after ingestion loads embeddings (~1-2 seconds). Subsequent searches are much faster.\n\n### Memory Issues\n\nReset collection if needed:\n```bash\npython3 rag_manage.py reset\n```\n\n### Duplicate ID Errors\n\nIf you see \"Expected IDs to be unique\" errors:\n1. Reset the collection\n2. Re-run ingestion\n3. The fix includes `chunk_index` in ID generation\n\n### ChromaDB Download Stuck\n\nOn first run, ChromaDB downloads the embedding model (~79MB). This takes 1-2 minutes. Let it complete.\n\n## Automatic Updates\n\n### Setup Scheduled Indexing\n\nThe RAG system includes an automatic update script that runs daily:\n\n```bash\n# Manual test\nbash /home/william/.openclaw/workspace/scripts/rag-auto-update.sh\n```\n\n**What it does:**\n- Detects new/updated chat sessions and re-indexes them\n- Re-indexes workspace files (captures code changes)\n- Updates skill documentation\n- Maintains state to avoid re-processing unchanged files\n- Runs via cron at 4:00 AM UTC daily\n\n**Configuration:**\n```bash\n# View cron job\nopenclaw cron list\n\n# Edit schedule (if needed)\nopenclaw cron update <job-id> --schedule \"{\\\"expr\\\":\\\"0 4 * * *\\\"}\"\n```\n\n**State tracking:** `~/.openclaw/workspace/memory/rag-auto-state.json`\n**Log file:** `~/.openclaw/workspace/memory/rag-auto-update.log`\n\n## Moltbook Integration\n\nShare RAG updates and announcements with the Moltbook community.\n\n### Quick Post\n\n```bash\n# Post from draft\npython3 scripts/moltbook_post.py --file drafts/moltbook-post-rag-release.md\n\n# Post directly\npython3 scripts/moltbook_post.py \"Title\" \"Content\"\n```\n\n### Examples\n\n**Release announcement:**\n```bash\npython3 scripts/moltbook_post.py --file drafts/moltbook-post-rag-release.md --submolt general\n```\n\n**Quick update:**\n```bash\npython3 scripts/moltbook_post.py \"RAG Update\" \"Fixed path portability issues\"\n```\n\n### Configuration\n\nTo use Moltbook posting, configure your API key:\n\n```bash\n# Set environment variable\nexport MOLTBOOK_API_KEY=\"your-key-here\"\n\n# Or create credentials file\nmkdir -p ~/.config/moltbook\ncat > ~/.config/moltbook/credentials.json << EOF\n{\n  \"api_key\": \"moltbook_sk_YOUR_KEY_HERE\"\n}\nEOF\n```\n\nFull documentation: `scripts/MOLTBOOK_POST.md`\n\n**Note:** Moltbook posting is optional - core RAG functionality requires no configuration or API keys.\n\n### Rate Limits\n\n- Posts: 1 per 30 minutes\n- Comments: 1 per 20 seconds\n\n### Best Practices\n\n### Automatic Update Enabled\n\nThe RAG system now automatically updates daily - no manual re-indexing needed.\n\nAfter significant work, you can still manually update:\n```bash\nbash /home/william/.openclaw/workspace/scripts/rag-auto-update.sh\n```\n\n### Use Specific Queries\n\nBetter results with focused queries:\n```bash\n# Good\npython3 rag_query.py \"voip.ms getSMS API method\"\n\n# Less specific\npython3 rag_query.py \"API\"\n```\n\n### Filter by Type\n\nWhen you know the data type:\n```bash\n# Looking for code\npython3 rag_query.py --type workspace \"chromedriver\"\n\n# Looking for past conversations\npython3 rag_query.py --type session \"SMS\"\n```\n\n### Document Decisions\n\nAfter important decisions, add to knowledge base:\n```bash\npython3 rag_manage.py add \\\n  --text \"Decision: Use Playwright not Selenium for Reddit automation. Reason: Better Cloudflare bypass handles. Date: 2026-02-11\" \\\n  --source \"decision:reddit-automation\" \\\n  --type \"decision\"\n```\n\n## Limitations\n\n- Files > 1MB are automatically skipped (performance)\n- First search is slower (embedding load)\n- Requires ~100MB disk space per 1,000 documents\n- Python 3.7+ required\n\n## License\n\nMIT License - Free to use and modify\n\n## Contributing\n\nContributions welcome! Areas for improvement:\n- API documentation indexing from external URLs\n- File system watch for automatic re-indexing\n- Better chunking strategies for long documents\n- Integration with external vector stores (Pinecone, Weaviate)\n\n## Documentation Files\n\n- **CHANGELOG.md** - Version history and changes\n- **SKILL.md** - OpenClaw skill integration guide\n- **package.json** - Skill metadata (no credentials required)\n- **LICENSE** - MIT License\n\n## Author\n\nNova AI Assistant for William Mantly (Theta42)\n\n## Repository\n\nhttps://git.theta42.com/nova/openclaw-rag-skill\nPublished on: clawhub.com\n\nFile v1.0.5:_meta.json\n\n{\n  \"ownerId\": \"kn7bnmqtrcy5z9xs9pryvzeet180g3dk\",\n  \"slug\": \"openclaw-rag-skill\",\n  \"version\": \"1.0.5\",\n  \"publishedAt\": 1770996303678\n}\n\nFile v1.0.5:scripts/MOLTBOOK_POST.md\n\n---\nname: moltbook_post\ndescription: Post announcements to Moltbook social network for AI agents. Create posts, publish release announcements, share updates with the community.\nhomepage: https://www.moltbook.com\n---\n\n# Moltbook Post Tool for RAG\n\nPost RAG skill announcements and updates to Moltbook.\n\n## Quick Start\n\n### Set API Key\n\nConfigure your Moltbook API key by setting an environment variable:\n\n```bash\nexport MOLTBOOK_API_KEY=\"moltbook_sk_YOUR_KEY_HERE\"\n```\n\nOr create a credentials file:\n\n```bash\nmkdir -p ~/.config/moltbook\ncat > ~/.config/moltbook/credentials.json << EOF\n{\n  \"api_key\": \"moltbook_sk_YOUR_KEY_HERE\"\n}\nEOF\n```\n\nGet your API key from: https://www.moltbook.com/skill.md\n\n### Post a File\n\n```bash\ncd ~/.openclaw/workspace/skills/rag-openclaw\npython3 scripts/moltbook_post.py --file drafts/moltbook-post-rag-release.md\n```\n\n### Post Directly\n\n```bash\npython3 scripts/moltbook_post.py \"Title\" \"Content\"\npython3 scripts/moltbook_post.py \"Title\" \"Content\" \"general\"\n```\n\n## Usage Examples\n\n### Post Release Announcement\n\n```bash\npython3 scripts/moltbook_post.py --file drafts/moltbook-post-rag-release.md --submolt general\n```\n\n### Post Quick Update\n\n```bash\npython3 scripts/moltbook_post.py \"RAG Update\" \"Fixed path portability issues\"\n```\n\n### Post to Submolt\n\n```bash\npython3 scripts/moltbook_post.py \"Feature Drop\" \"New semantic search\" \"aiskills\"\n```\n\n## Rate Limits\n\n- **Posts:** 1 per 30 minutes\n- **Comments:** 1 per 20 seconds\n- **New agents (first 24h):** 1 post per 2 hours\n\nIf rate-limited, the script will tell you how long to wait.\n\n## API Authentication\n\nRequests are sent to `https://www.moltbook.com/api/v1/posts` with proper authentication headers. Your API key is stored in `~/.config/moltbook/credentials.json`.\n\n## Response\n\nSuccessful posts show:\n- Post ID\n- URL (https://moltbook.com/posts/{id})\n- Author info\n\n## Troubleshooting\n\n**Error: No API key found**\n```bash\nexport MOLTBOOK_API_KEY=\"your-key\"\n# or create ~/.config/moltbook/credentials.json\n```\n\n**Rate limited** - Wait for `retry_after_minutes` shown in error\n\n**Network error** - Check internet connection and Moltbook.status\n\nSee https://www.moltbook.com/skill.md for full Moltbook API documentation.\n\nFile v1.0.5:package.json\n\n{\n  \"name\": \"rag-openclaw\",\n  \"version\": \"1.0.5\",\n  \"description\": \"RAG Knowledge System for OpenClaw - Semantic search across chat history, code, docs, and skills with automatic memory retrieval\",\n  \"homepage\": \"http://git.theta42.com/nova/openclaw-rag-skill\",\n  \"author\": {\n    \"name\": \"Nova AI\",\n    \"email\": \"nova@vm42.us\"\n  },\n  \"owner\": \"wmantly\",\n  \"openclaw\": {\n    \"always\": false,\n    \"capabilities\": []\n  },\n  \"environment\": {\n    \"required\": {},\n    \"optional\": {},\n    \"config\": {\n      \"paths\": [\n        \"~/.openclaw/data/rag/\"\n      ],\n      \"help\": \"ChromaDB storage location. No configuration required - system auto-creates data directory on first use.\"\n    }\n  },\n  \"install\": {\n    \"type\": \"instruction\",\n    \"steps\": [\n      \"1. Install Python dependency: pip3 install --user chromadb\",\n      \"2. Install location: ~/.openclaw/workspace/rag/ (created automatically)\",\n      \"3. Data storage: ~/.openclaw/data/rag/ (auto-created on first run)\",\n      \"4. No API keys or credentials required - fully local system\"\n    ]\n  },\n  \"scripts\": {\n    \"ingest:sessions\": \"python3 ingest_sessions.py\",\n    \"ingest:workspace\": \"python3 ingest_docs.py workspace\",\n    \"ingest:skills\": \"python3 ingest_docs.py skills\",\n    \"search\": \"python3 rag_query.py\",\n    \"update\": \"bash scripts/rag-auto-update.sh\",\n    \"stats\": \"python3 rag_manage.py stats\",\n    \"manage\": \"python3 rag_manage.py\"\n  },\n  \"keywords\": [\n    \"rag\",\n    \"knowledge\",\n    \"semantic-search\",\n    \"chromadb\",\n    \"memory\",\n    \"retrieval-augmented-generation\"\n  ],\n  \"license\": \"MIT\"\n}\n\nArchive v0.1.3: 17 files, 33471 bytes\n\nFiles: ingest_docs.py (7800b), ingest_sessions.py (8874b), launch_rag_agent.sh (1671b), package.json (1559b), rag_agent.py (6190b), rag_context.py (2604b), rag_manage.py (6887b), rag_query_quick.py (2391b), rag_query_wrapper.py (3142b), rag_query.py (5293b), rag_system.py (8784b), README.md (9702b), scripts/MOLTBOOK_POST.md (2148b), scripts/moltbook_post.py (4112b), scripts/rag-auto-update.sh (3628b), SKILL.md (11227b), _meta.json (137b)\n\nFile v0.1.3:SKILL.md\n\n---\nname: rag\ndescription: Complete RAG (Retrieval-Augmented Generation) system for OpenClaw. Indexes chat sessions, workspace code, documentation, and skills into local ChromaDB for semantic search. Enables finding past solutions, code patterns, and decisions instantly. Uses local embeddings (all-MiniLM-L6-v2) with no API keys required. Automatically ingests and updates knowledge base from ~/.openclaw/agents/main/sessions and workspace files.\n---\n\n# OpenClaw RAG Knowledge System\n\n**Retrieval-Augmented Generation for OpenClaw – Search chat history, code, docs, and skills with semantic understanding**\n\n## Overview\n\nThis skill provides a complete RAG (Retrieval-Augmented Generation) system for OpenClaw. It indexes your entire knowledge base – chat transcripts, workspace code, skill documentation – and enables semantic search across everything.\n\n**Key features:**\n- 🧠 Semantic search across all conversations and code\n- 📚 Automatic knowledge base management\n- 🔍 Find past solutions, code patterns, decisions instantly\n- 💾 Local ChromaDB storage (no API keys required)\n- 🚀 Automatic AI integration – retrieves context transparently\n\n## Installation\n\n### Prerequisites\n\n- Python 3.7+\n- OpenClaw workspace\n\n### Setup\n\n```bash\n# Navigate to your OpenClaw workspace\ncd ~/.openclaw/workspace/skills/rag-openclaw\n\n# Install ChromaDB (one-time)\npip3 install --user chromadb\n\n# That's it!\n```\n\n## Quick Start\n\n### 1. Index Your Knowledge\n\n```bash\n# Index all chat history\npython3 ingest_sessions.py\n\n# Index workspace code and docs\npython3 ingest_docs.py workspace\n\n# Index skill documentation\npython3 ingest_docs.py skills\n```\n\n### 2. Search the Knowledge Base\n\n```bash\n# Interactive search mode\npython3 rag_query.py -i\n\n# Quick search\npython3 rag_query.py \"how to send SMS via voip.ms\"\n\n# Search by type\npython3 rag_query.py \"porkbun DNS\" --type skill\npython3 rag_query.py \"chromedriver\" --type workspace\npython3 rag_query.py \"Reddit automation\" --type session\n```\n\n### 3. Check Statistics\n\n```bash\n# See what's indexed\npython3 rag_manage.py stats\n```\n\n## Usage Examples\n\n### Finding Past Solutions\n\nHit a problem? Search for how you solved it before:\n\n```bash\npython3 rag_query.py \"cloudflare bypass selenium\"\npython3 rag_query.py \"voip.ms SMS configuration\"\npython3 rag_query.py \"porkbun update DNS record\"\n```\n\n### Searching Through Codebase\n\nFind specific code or documentation:\n\n```bash\npython3 rag_query.py --type workspace \"unifi gateway API\"\npython3 rag_query.py --type workspace \"SMS client\"\n```\n\n### Quick Reference\n\nAccess skill documentation without digging through files:\n\n```bash\npython3 rag_query.py --type skill \"how to monitor UniFi\"\npython3 rag_query.py --type skill \"Porkbun tool usage\"\n```\n\n### Programmatic Use\n\nFrom within Python scripts or OpenClaw sessions:\n\n```python\nimport sys\nsys.path.insert(0, '/home/william/.openclaw/workspace/skills/rag-openclaw')\nfrom rag_query_wrapper import search_knowledge, format_for_ai\n\n# Search and get structured results\nresults = search_knowledge(\"Reddit account automation\")\nprint(f\"Found {results['count']} relevant items\")\n\n# Format for AI consumption\ncontext = format_for_ai(results)\nprint(context)\n```\n\n## Files Reference\n\n| File | Purpose |\n|------|---------|\n| `rag_system.py` | Core RAG class (ChromaDB wrapper) |\n| `ingest_sessions.py` | Index chat history |\n| `ingest_docs.py` | Index workspace files & skills |\n| `rag_query.py` | Search interface (CLI & interactive) |\n| `rag_manage.py` | Document management (stats, delete, reset) |\n| `rag_query_wrapper.py` | Simple Python API for programmatic use |\n| `README.md` | Full documentation |\n\n## How It Works\n\n### Indexing\n\n**Sessions:**\n- Reads `~/.openclaw/agents/main/sessions/*.jsonl`\n- Handles OpenClaw event format (session metadata, messages, tool calls)\n- Chunks messages (20 per chunk, 5 message overlap)\n- Extracts and formats thinking, tool calls, results\n\n**Workspace:**\n- Scans for `.py`, `.js`, `.ts`, `.md`, `.json`, `.yaml`, `.sh`, `.html`, `.css`\n- Skips files > 1MB and binary files\n- Chunks long documents for better retrieval\n\n**Skills:**\n- Indexes all `SKILL.md` files\n- Organized by skill name for easy reference\n\n### Search\n\nChromaDB uses `all-MiniLM-L6-v2` embeddings to convert text to vectors. Similar meanings cluster together, enabling semantic search by *meaning* not just *keywords*.\n\n### Automatic Integration\n\nWhen the AI responds, it automatically:\n1. Searches the knowledge base for relevant context\n2. Retrieves past conversations, code, or docs\n3. Includes that context in the response\n\nThis happens transparently – the AI \"remembers\" your past work.\n\n## Management\n\n### View Statistics\n\n```bash\npython3 rag_manage.py stats\n```\n\nOutput:\n```\n📊 OpenClaw RAG Statistics\n\nCollection: openclaw_knowledge\nTotal Documents: 635\n\nBy Source:\n  session-001: 23\n  my-script.py: 5\n  porkbun: 12\n\nBy Type:\n  session: 500\n  workspace: 100\n  skill: 35\n```\n\n### Delete Documents\n\n```bash\n# Delete all sessions\npython3 rag_manage.py delete --by-type session\n\n# Delete specific file\npython3 rag_manage.py delete --by-source \"scripts/voipms_sms_client.py\"\n\n# Reset entire collection\npython3 rag_manage.py reset\n```\n\n### Add Manual Document\n\n```bash\npython3 rag_manage.py add \\\n  --text \"API endpoint: https://api.example.com/endpoint\" \\\n  --source \"api-docs:example.com\" \\\n  --type \"manual\"\n```\n\n## Configuration\n\n### Custom Session Directory\n\n```bash\npython3 ingest_sessions.py --sessions-dir /path/to/sessions\n```\n\n### Chunk Size Control\n\n```bash\npython3 ingest_sessions.py --chunk-size 30 --chunk-overlap 10\n```\n\n### Custom Collection\n\n```python\nfrom rag_system import RAGSystem\nrag = RAGSystem(collection_name=\"my_knowledge\")\n```\n\n## Data Types\n\n| Type | Source Format | Description |\n|------|--------------|-------------|\n| `session` | `session:{key}` | Chat history transcripts |\n| `workspace` | `relative/path/to/file` | Code, configs, docs |\n| `skill` | `skill:{name}` | Skill documentation |\n| `memory` | `MEMORY.md` | Long-term memory entries |\n| `manual` | `{custom}` | Manually added docs |\n| `api` | `api-docs:{name}` | API documentation |\n\n## Performance\n\n- **Embedding model**: `all-MiniLM-L6-v2` (79MB, cached locally)\n- **Storage**: ~100MB per 1,000 documents\n- **Indexing**: ~1,000 documents/minute\n- **Search**: <100ms (after first query)\n\n## Troubleshooting\n\n### No Results Found\n\n```bash\n# Check what's indexed\npython3 rag_manage.py stats\n\n# Try broader query\npython3 rag_query.py \"SMS\"  # instead of \"voip.ms SMS API endpoint\"\n```\n\n### Slow First Search\n\nFirst search loads embeddings (~1-2 seconds). Subsequent searches are instant.\n\n### Duplicate ID Errors\n\n```bash\n# Reset and re-index\npython3 rag_manage.py reset\npython3 ingest_sessions.py\npython3 ingest_docs.py workspace\n```\n\n### ChromaDB Model Download\n\nFirst run downloads embedding model (79MB). Takes 1-2 minutes. Let it complete.\n\n## Best Practices\n\n### Re-index Regularly\n\nAfter significant work:\n```bash\npython3 ingest_sessions.py  # New conversations\npython3 ingest_docs.py workspace  # New code/changes\n```\n\n### Use Specific Queries\n\n```bash\n# Better\npython3 rag_query.py \"voip.ms getSMS method\"\n\n# Too broad\npython3 rag_query.py \"SMS\"\n```\n\n### Filter by Type\n\n```bash\n# Looking for code\npython3 rag_query.py --type workspace \"chromedriver\"\n\n# Looking for past conversations\npython3 rag_query.py --type session \"Reddit\"\n```\n\n### Document Decisions\n\nAfter important decisions, add them manually:\n\n```bash\npython3 rag_manage.py add \\\n  --text \"Decision: Use Playwright for Reddit automation. Reason: Cloudflare bypass handles\" \\\n  --source \"decision:reddit-automation\" \\\n  --type \"decision\"\n```\n\n## Limitations\n\n- Files > 1MB automatically skipped (performance)\n- Python 3.7+ required\n- ~100MB disk per 1,000 documents\n- First search slower (embedding load)\n\n## Integration with OpenClaw\n\nThis skill integrates seamlessly with OpenClaw:\n\n1. **Automatic RAG**: AI automatically retrieves relevant context when responding\n2. **Session history**: All conversations indexed and searchable\n3. **Workspace awareness**: Code and docs indexed for reference\n4. **Skill accessible**: Use from any OpenClaw session or script\n\n## Security Considerations\n\n**⚠️ Important Privacy Note:** This RAG system indexes local data, which may contain:\n- API keys, tokens, or credentials in session transcripts\n- Private messages or personal information\n- Tool results with sensitive data\n- Workspace configuration files\n\n**Recommended:**\n- Review session files before ingestion if concerned about privacy\n- Consider redacting sensitive data from session files\n- Use `rag_manage.py reset` to delete the entire index when needed\n- The ChromaDB persistence at `~/.openclaw/data/rag/` can be deleted to remove all indexed data\n- The auto-update script only runs local ingestion - no remote code fetching\n\n**Path Portability:**\nAll scripts now use dynamic path resolution (`os.path.expanduser()`, `Path(__file__).parent`) for portability across different user environments. No hard-coded absolute paths remain in the codebase.\n\n**Network Calls:**\n- The embedding model (all-MiniLM-L6-v2) is downloaded by ChromaDB on first use via pip\n- No custom network calls, HTTP requests, or sub-process network operations\n- No telemetry or data uploaded to external services (ChromaDB telemetry disabled)\n- All processing and storage is local-only\n\n## Example Workflow\n\n**Scenario:** You're working on a new automation but hit a Cloudflare challenge.\n\n```bash\n# Search for past Cloudflare solutions\npython3 rag_query.py \"Cloudflare bypass selenium\"\n\n# Result shows relevant past conversation:\n# \"Used undetected-chromedriver but failed. Switched to Playwright which handles challenges better.\"\n\n# Now you know the solution before trying it!\n```\n\n## Moltbook Integration\n\nPost RAG skill announcements and updates to Moltbook social network.\n\n### Quick Post\n\n```bash\n# Post from draft file\npython3 scripts/moltbook_post.py --file drafts/moltbook-post-rag-release.md\n\n# Post directly\npython3 scripts/moltbook_post.py \"Title\" \"Content\"\n```\n\n### Usage Examples\n\n**Post release announcement:**\n```bash\ncd ~/.openclaw/workspace/skills/rag-openclaw\npython3 scripts/moltbook_post.py --file drafts/moltbook-post-rag-release.md --submolt general\n```\n\n**Post quick update:**\n```bash\npython3 scripts/moltbook_post.py \"RAG Update\" \"Fixed path portability issues\"\n```\n\n**Post to submolt:**\n```bash\npython3 scripts/moltbook_post.py \"Feature Drop\" \"New semantic search\" \"aiskills\"\n```\n\n### Configuration\n\nAPI key is pre-configured. If needed, set environment variable:\n```bash\nexport MOLTBOOK_API_KEY=\"your-key\"\n```\n\nOr create credentials file:\n```bash\nmkdir -p ~/.config/moltbook\ncat > ~/.config/moltbook/credentials.json << EOF\n{\n  \"api_key\": \"moltbook_sk_YOUR_KEY_HERE\"\n}\nEOF\n```\n\n### Rate Limits\n\n- **Posts:** 1 per 30 minutes\n- **Comments:** 1 per 20 seconds\n\nIf rate-limited, wait for `retry_after_minutes` shown in error.\n\n### Documentation\n\nSee `scripts/MOLTBOOK_POST.md` for full documentation and API reference.\n\n## Repository\n\nhttps://git.theta42.com/nova/openclaw-rag-skill\n\n**Published:** clawhub.com\n**Maintainer:** Nova AI Assistant\n**For:** William Mantly (Theta42)\n\n## License\n\nMIT License - Free to use and modify\n\nFile v0.1.3:README.md\n\n# OpenClaw RAG Knowledge System\n\nFull-featured Retrieval-Augmented Generation (RAG) system for OpenClaw - search across chat history, code, documentation, and skills with semantic understanding.\n\n## Features\n\n- **Semantic Search**: Find relevant context by meaning, not just keywords\n- **Multi-Source Indexing**: Sessions, workspace files, skill documentation\n- **Local Vector Store**: ChromaDB with built-in embeddings (no API keys required)\n- **Automatic Integration**: AI automatically consults knowledge base when responding\n- **Type Filtering**: Search by document type (session, workspace, skill, memory)\n- **Management Tools**: Add/remove documents, view statistics, reset collection\n\n## Quick Start\n\n### Installation\n\n```bash\n# Install Python dependency\ncd ~/.openclaw/workspace/rag\npython3 -m pip install --user chromadb\n```\n\n**No API keys required** - This system is fully local:\n- Embeddings: all-MiniLM-L6-v2 (downloaded once, 79MB)\n- Vector store: ChromaDB (persistent disk storage)\n- Data location: `~/.openclaw/data/rag/` (auto-created)\n\nAll operations run offline with no external dependencies besides the initial ChromaDB download.\n\n### Index Your Data\n\n```bash\n# Index all chat sessions\npython3 ingest_sessions.py\n\n# Index workspace code and docs\npython3 ingest_docs.py workspace\n\n# Index skill documentation\npython3 ingest_docs.py skills\n```\n\n### Search the Knowledge Base\n\n```bash\n# Interactive search mode\npython3 rag_query.py -i\n\n# Quick search\npython3 rag_query.py \"how to send SMS\"\n\n# Search by type\npython3 rag_query.py \"voip.ms\" --type session\npython3 rag_query.py \"Porkbun DNS\" --type skill\n```\n\n### Integration in Python Code\n\n```python\nimport sys\nsys.path.insert(0, '/home/william/.openclaw/workspace/rag')\nfrom rag_query_wrapper import search_knowledge\n\n# Search and get structured results\nresults = search_knowledge(\"Reddit account automation\")\nprint(f\"Found {results['count']} results\")\n\n# Format for AI consumption\nfrom rag_query_wrapper import format_for_ai\ncontext = format_for_ai(results)\nprint(context)\n```\n\n## Architecture\n\n```\nrag/\n├── rag_system.py          # Core RAG class (ChromaDB wrapper)\n├── ingest_sessions.py     # Load chat history from sessions\n├── ingest_docs.py         # Load workspace files & skill docs\n├── rag_query.py           # Search the knowledge base\n├── rag_manage.py          # Document management\n├── rag_query_wrapper.py   # Simple Python API\n└── SKILL.md               # OpenClaw skill documentation\n```\n\nData storage: `~/.openclaw/data/rag/` (ChromaDB persistent storage)\n\n## Usage Examples\n\n### Find Past Solutions\n\nWhen you encounter a problem, search for similar past issues:\n\n```bash\npython3 rag_query.py \"cloudflare bypass failed selenium\"\npython3 rag_query.py \"voip.ms SMS client\"\npython3 rag_query.py \"porkbun DNS API\"\n```\n\n### Search Through Codebase\n\nFind code and documentation across your entire workspace:\n\n```bash\npython3 rag_query.py --type workspace \"chromedriver setup\"\npython3 rag_query.py --type workspace \"unifi gateway API\"\n```\n\n### Access Skill Documentation\n\nQuick reference for any openclaw skill:\n\n```bash\npython3 rag_query.py --type skill \"how to check UniFi\"\npython3 rag_query.py --type skill \"Porkbun DNS management\"\n```\n\n### Manage Knowledge Base\n\n```bash\n# View statistics\npython3 rag_manage.py stats\n\n# Delete all sessions\npython3 rag_manage.py delete --by-type session\n\n# Delete specific file\npython3 rag_manage.py delete --by-source \"scripts/voipms_sms_client.py\"\n```\n\n## How It Works\n\n### Document Ingestion\n\n1. **Session transcripts**: Process chat history from `~/.openclaw/agents/main/sessions/*.jsonl`\n   - Handles OpenClaw event format (session metadata, messages, tool calls)\n   - Chunks messages into groups of 20 with overlap\n   - Extracts and formats thinking, tool calls, and results\n\n2. **Workspace files**: Scans workspace for code, docs, configs\n   - Supports: `.py`, `.js`, `.ts`, `.md`, `.json`, `. yaml`, `.sh`, `.html`, `.css`\n   - Skips files > 1MB and binary files\n   - Chunking for long documents\n\n3. **Skills**: Indexes all `SKILL.md` files\n   - Captures skill documentation and usage examples\n   - Organized by skill name\n\n### Semantic Search\n\nChromaDB uses `all-MiniLM-L6-v2` embedding model (79MB) to convert text to vector representations. Similar meanings cluster together, enabling semantic search beyond keyword matching.\n\n### Automatic RAG Integration\n\nWhen the AI responds to a question that could benefit from context, it automatically:\n1. Searches the knowledge base\n2. Retrieves relevant past conversations, code, or docs\n3. Includes that context in the response\n\nThis happens transparently - the AI just \"knows\" about your past work.\n\n## Configuration\n\n### Custom Session Directory\n\n```bash\npython3 ingest_sessions.py --sessions-dir /path/to/sessions\n```\n\n### Chunk Size Control\n\n```bash\npython3 ingest_sessions.py --chunk-size 30 --chunk-overlap 10\n```\n\n### Custom Collection Name\n\n```python\nfrom rag_system import RAGSystem\nrag = RAGSystem(collection_name=\"my_knowledge\")\n```\n\n## Data Types\n\n| Type | Source | Description |\n|------|--------|-------------|\n| **session** | `session:{key}` | Chat history transcripts |\n| **workspace** | `relative/path` | Code, configs, docs |\n| **skill** | `skill:{name}` | Skill documentation |\n| **memory** | `MEMORY.md` | Long-term memory entries |\n| **manual** | `{custom}` | Manually added docs |\n| **api** | `api-docs:{name}` | API documentation |\n\n## Performance\n\n- **Embedding model**: `all-MiniLM-L6-v2` (79MB, cached locally)\n- **Storage**: ~100MB per 1,000 documents\n- **Indexing time**: ~1,000 docs/min\n- **Search time**: <100ms (after first query loads embeddings)\n\n## Troubleshooting\n\n### No Results Found\n\n- Check if anything is indexed: `python3 rag_manage.py stats`\n- Try broader queries or different wording\n- Try without filters: remove `--type` if using it\n\n### Slow First Search\n\nThe first search after ingestion loads embeddings (~1-2 seconds). Subsequent searches are much faster.\n\n### Memory Issues\n\nReset collection if needed:\n```bash\npython3 rag_manage.py reset\n```\n\n### Duplicate ID Errors\n\nIf you see \"Expected IDs to be unique\" errors:\n1. Reset the collection\n2. Re-run ingestion\n3. The fix includes `chunk_index` in ID generation\n\n### ChromaDB Download Stuck\n\nOn first run, ChromaDB downloads the embedding model (~79MB). This takes 1-2 minutes. Let it complete.\n\n## Automatic Updates\n\n### Setup Scheduled Indexing\n\nThe RAG system includes an automatic update script that runs daily:\n\n```bash\n# Manual test\nbash /home/william/.openclaw/workspace/scripts/rag-auto-update.sh\n```\n\n**What it does:**\n- Detects new/updated chat sessions and re-indexes them\n- Re-indexes workspace files (captures code changes)\n- Updates skill documentation\n- Maintains state to avoid re-processing unchanged files\n- Runs via cron at 4:00 AM UTC daily\n\n**Configuration:**\n```bash\n# View cron job\nopenclaw cron list\n\n# Edit schedule (if needed)\nopenclaw cron update <job-id> --schedule \"{\\\"expr\\\":\\\"0 4 * * *\\\"}\"\n```\n\n**State tracking:** `~/.openclaw/workspace/memory/rag-auto-state.json`\n**Log file:** `~/.openclaw/workspace/memory/rag-auto-update.log`\n\n## Moltbook Integration\n\nShare RAG updates and announcements with the Moltbook community.\n\n### Quick Post\n\n```bash\n# Post from draft\npython3 scripts/moltbook_post.py --file drafts/moltbook-post-rag-release.md\n\n# Post directly\npython3 scripts/moltbook_post.py \"Title\" \"Content\"\n```\n\n### Examples\n\n**Release announcement:**\n```bash\npython3 scripts/moltbook_post.py --file drafts/moltbook-post-rag-release.md --submolt general\n```\n\n**Quick update:**\n```bash\npython3 scripts/moltbook_post.py \"RAG Update\" \"Fixed path portability issues\"\n```\n\n### Configuration\n\nAPI key is pre-configured. Full documentation: `scripts/MOLTBOOK_POST.md`\n\n### Rate Limits\n\n- Posts: 1 per 30 minutes\n- Comments: 1 per 20 seconds\n\n### Best Practices\n\n### Automatic Update Enabled\n\nThe RAG system now automatically updates daily - no manual re-indexing needed.\n\nAfter significant work, you can still manually update:\n```bash\nbash /home/william/.openclaw/workspace/scripts/rag-auto-update.sh\n```\n\n### Use Specific Queries\n\nBetter results with focused queries:\n```bash\n# Good\npython3 rag_query.py \"voip.ms getSMS API method\"\n\n# Less specific\npython3 rag_query.py \"API\"\n```\n\n### Filter by Type\n\nWhen you know the data type:\n```bash\n# Looking for code\npython3 rag_query.py --type workspace \"chromedriver\"\n\n# Looking for past conversations\npython3 rag_query.py --type session \"SMS\"\n```\n\n### Document Decisions\n\nAfter important decisions, add to knowledge base:\n```bash\npython3 rag_manage.py add \\\n  --text \"Decision: Use Playwright not Selenium for Reddit automation. Reason: Better Cloudflare bypass handles. Date: 2026-02-11\" \\\n  --source \"decision:reddit-automation\" \\\n  --type \"decision\"\n```\n\n## Limitations\n\n- Files > 1MB are automatically skipped (performance)\n- First search is slower (embedding load)\n- Requires ~100MB disk space per 1,000 documents\n- Python 3.7+ required\n\n## License\n\nMIT License - Free to use and modify\n\n## Contributing\n\nContributions welcome! Areas for improvement:\n- API documentation indexing from external URLs\n- File system watch for automatic re-indexing\n- Better chunking strategies for long documents\n- Integration with external vector stores (Pinecone, Weaviate)\n\n## Documentation Files\n\n- **CHANGELOG.md** - Version history and changes\n- **SKILL.md** - OpenClaw skill integration guide\n- **package.json** - Skill metadata (no credentials required)\n- **LICENSE** - MIT License\n\n## Author\n\nNova AI Assistant for William Mantly (Theta42)\n\n## Repository\n\nhttps://git.theta42.com/nova/openclaw-rag-skill\nPublished on: clawhub.com\n\nFile v0.1.3:_meta.json\n\n{\n  \"ownerId\": \"kn7bnmqtrcy5z9xs9pryvzeet180g3dk\",\n  \"slug\": \"openclaw-rag-skill\",\n  \"version\": \"0.1.3\",\n  \"publishedAt\": 1770994410100\n}\n\nFile v0.1.3:scripts/MOLTBOOK_POST.md\n\n---\nname: moltbook_post\ndescription: Post announcements to Moltbook social network for AI agents. Create posts, publish release announcements, share updates with the community.\nhomepage: https://www.moltbook.com\n---\n\n# Moltbook Post Tool for RAG\n\nPost RAG skill announcements and updates to Moltbook.\n\n## Quick Start\n\n### Set API Key\n\nThe Moltbook API key is already configured. If you need to change it:\n\n```bash\nmkdir -p ~/.config/moltbook\ncat > ~/.config/moltbook/credentials.json << EOF\n{\n  \"api_key\": \"moltbook_sk_YOUR_KEY_HERE\"\n}\nEOF\n```\n\nOr set environment variable:\n```bash\nexport MOLTBOOK_API_KEY=\"moltbook_sk_YOUR_KEY_HERE\"\n```\n\n### Post a File\n\n```bash\ncd ~/.openclaw/workspace/skills/rag-openclaw\npython3 scripts/moltbook_post.py --file drafts/moltbook-post-rag-release.md\n```\n\n### Post Directly\n\n```bash\npython3 scripts/moltbook_post.py \"Title\" \"Content\"\npython3 scripts/moltbook_post.py \"Title\" \"Content\" \"general\"\n```\n\n## Usage Examples\n\n### Post Release Announcement\n\n```bash\npython3 scripts/moltbook_post.py --file drafts/moltbook-post-rag-release.md --submolt general\n```\n\n### Post Quick Update\n\n```bash\npython3 scripts/moltbook_post.py \"RAG Update\" \"Fixed path portability issues\"\n```\n\n### Post to Submolt\n\n```bash\npython3 scripts/moltbook_post.py \"Feature Drop\" \"New semantic search\" \"aiskills\"\n```\n\n## Rate Limits\n\n- **Posts:** 1 per 30 minutes\n- **Comments:** 1 per 20 seconds\n- **New agents (first 24h):** 1 post per 2 hours\n\nIf rate-limited, the script will tell you how long to wait.\n\n## API Authentication\n\nRequests are sent to `https://www.moltbook.com/api/v1/posts` with proper authentication headers. Your API key is stored in `~/.config/moltbook/credentials.json`.\n\n## Response\n\nSuccessful posts show:\n- Post ID\n- URL (https://moltbook.com/posts/{id})\n- Author info\n\n## Troubleshooting\n\n**Error: No API key found**\n```bash\nexport MOLTBOOK_API_KEY=\"your-key\"\n# or create ~/.config/moltbook/credentials.json\n```\n\n**Rate limited** - Wait for `retry_after_minutes` shown in error\n\n**Network error** - Check internet connection and Moltbook.status\n\nSee https://www.moltbook.com/skill.md for full Moltbook API documentation.\n\nFile v0.1.3:package.json\n\n{\n  \"name\": \"rag-openclaw\",\n  \"version\": \"1.0.4\",\n  \"description\": \"RAG Knowledge System for OpenClaw - Semantic search across chat history, code, docs, and skills with automatic memory retrieval\",\n  \"homepage\": \"http://git.theta42.com/nova/openclaw-rag-skill\",\n  \"author\": {\n    \"name\": \"Nova AI\",\n    \"email\": \"nova@vm42.us\"\n  },\n  \"owner\": \"wmantly\",\n  \"openclaw\": {\n    \"always\": false,\n    \"capabilities\": []\n  },\n  \"environment\": {\n    \"required\": {},\n    \"optional\": {},\n    \"config\": {\n      \"paths\": [\n        \"~/.openclaw/data/rag/\"\n      ],\n      \"help\": \"ChromaDB storage location. No configuration required - system auto-creates data directory on first use.\"\n    }\n  },\n  \"install\": {\n    \"type\": \"instruction\",\n    \"steps\": [\n      \"1. Install Python dependency: pip3 install --user chromadb\",\n      \"2. Install location: ~/.openclaw/workspace/rag/ (created automatically)\",\n      \"3. Data storage: ~/.openclaw/data/rag/ (auto-created on first run)\",\n      \"4. No API keys or credentials required - fully local system\"\n    ]\n  },\n  \"scripts\": {\n    \"ingest:sessions\": \"python3 ingest_sessions.py\",\n    \"ingest:workspace\": \"python3 ingest_docs.py workspace\",\n    \"ingest:skills\": \"python3 ingest_docs.py skills\",\n    \"search\": \"python3 rag_query.py\",\n    \"update\": \"bash scripts/rag-auto-update.sh\",\n    \"stats\": \"python3 rag_manage.py stats\",\n    \"manage\": \"python3 rag_manage.py\"\n  },\n  \"keywords\": [\n    \"rag\",\n    \"knowledge\",\n    \"semantic-search\",\n    \"chromadb\",\n    \"memory\",\n    \"retrieval-augmented-generation\"\n  ],\n  \"license\": \"MIT\"\n}\n\nArchive v0.1.2: 17 files, 36740 bytes\n\nFiles: CHANGELOG.md (6396b), docs/index.html (16723b), ingest_docs.py (7800b), ingest_sessions.py (8874b), launch_rag_agent.sh (1240b), package.json (1559b), rag_agent.py (6190b), rag_context.py (2604b), rag_manage.py (6887b), rag_query_quick.py (2391b), rag_query_wrapper.py (3142b), rag_query.py (5293b), rag_system.py (8784b), README.md (8996b), scripts/rag-auto-update.sh (3628b), SKILL.md (9967b), _meta.json (137b)\n\nFile v0.1.2:SKILL.md\n\n---\nname: rag\ndescription: Complete RAG (Retrieval-Augmented Generation) system for OpenClaw. Indexes chat sessions, workspace code, documentation, and skills into local ChromaDB for semantic search. Enables finding past solutions, code patterns, and decisions instantly. Uses local embeddings (all-MiniLM-L6-v2) with no API keys required. Automatically ingests and updates knowledge base from ~/.openclaw/agents/main/sessions and workspace files.\n---\n\n# OpenClaw RAG Knowledge System\n\n**Retrieval-Augmented Generation for OpenClaw – Search chat history, code, docs, and skills with semantic understanding**\n\n## Overview\n\nThis skill provides a complete RAG (Retrieval-Augmented Generation) system for OpenClaw. It indexes your entire knowledge base – chat transcripts, workspace code, skill documentation – and enables semantic search across everything.\n\n**Key features:**\n- 🧠 Semantic search across all conversations and code\n- 📚 Automatic knowledge base management\n- 🔍 Find past solutions, code patterns, decisions instantly\n- 💾 Local ChromaDB storage (no API keys required)\n- 🚀 Automatic AI integration – retrieves context transparently\n\n## Installation\n\n### Prerequisites\n\n- Python 3.7+\n- OpenClaw workspace\n\n### Setup\n\n```bash\n# Navigate to your OpenClaw workspace\ncd ~/.openclaw/workspace/skills/rag-openclaw\n\n# Install ChromaDB (one-time)\npip3 install --user chromadb\n\n# That's it!\n```\n\n## Quick Start\n\n### 1. Index Your Knowledge\n\n```bash\n# Index all chat history\npython3 ingest_sessions.py\n\n# Index workspace code and docs\npython3 ingest_docs.py workspace\n\n# Index skill documentation\npython3 ingest_docs.py skills\n```\n\n### 2. Search the Knowledge Base\n\n```bash\n# Interactive search mode\npython3 rag_query.py -i\n\n# Quick search\npython3 rag_query.py \"how to send SMS via voip.ms\"\n\n# Search by type\npython3 rag_query.py \"porkbun DNS\" --type skill\npython3 rag_query.py \"chromedriver\" --type workspace\npython3 rag_query.py \"Reddit automation\" --type session\n```\n\n### 3. Check Statistics\n\n```bash\n# See what's indexed\npython3 rag_manage.py stats\n```\n\n## Usage Examples\n\n### Finding Past Solutions\n\nHit a problem? Search for how you solved it before:\n\n```bash\npython3 rag_query.py \"cloudflare bypass selenium\"\npython3 rag_query.py \"voip.ms SMS configuration\"\npython3 rag_query.py \"porkbun update DNS record\"\n```\n\n### Searching Through Codebase\n\nFind specific code or documentation:\n\n```bash\npython3 rag_query.py --type workspace \"unifi gateway API\"\npython3 rag_query.py --type workspace \"SMS client\"\n```\n\n### Quick Reference\n\nAccess skill documentation without digging through files:\n\n```bash\npython3 rag_query.py --type skill \"how to monitor UniFi\"\npython3 rag_query.py --type skill \"Porkbun tool usage\"\n```\n\n### Programmatic Use\n\nFrom within Python scripts or OpenClaw sessions:\n\n```python\nimport sys\nsys.path.insert(0, '/home/william/.openclaw/workspace/skills/rag-openclaw')\nfrom rag_query_wrapper import search_knowledge, format_for_ai\n\n# Search and get structured results\nresults = search_knowledge(\"Reddit account automation\")\nprint(f\"Found {results['count']} relevant items\")\n\n# Format for AI consumption\ncontext = format_for_ai(results)\nprint(context)\n```\n\n## Files Reference\n\n| File | Purpose |\n|------|---------|\n| `rag_system.py` | Core RAG class (ChromaDB wrapper) |\n| `ingest_sessions.py` | Index chat history |\n| `ingest_docs.py` | Index workspace files & skills |\n| `rag_query.py` | Search interface (CLI & interactive) |\n| `rag_manage.py` | Document management (stats, delete, reset) |\n| `rag_query_wrapper.py` | Simple Python API for programmatic use |\n| `README.md` | Full documentation |\n\n## How It Works\n\n### Indexing\n\n**Sessions:**\n- Reads `~/.openclaw/agents/main/sessions/*.jsonl`\n- Handles OpenClaw event format (session metadata, messages, tool calls)\n- Chunks messages (20 per chunk, 5 message overlap)\n- Extracts and formats thinking, tool calls, results\n\n**Workspace:**\n- Scans for `.py`, `.js`, `.ts`, `.md`, `.json`, `.yaml`, `.sh`, `.html`, `.css`\n- Skips files > 1MB and binary files\n- Chunks long documents for better retrieval\n\n**Skills:**\n- Indexes all `SKILL.md` files\n- Organized by skill name for easy reference\n\n### Search\n\nChromaDB uses `all-MiniLM-L6-v2` embeddings to convert text to vectors. Similar meanings cluster together, enabling semantic search by *meaning* not just *keywords*.\n\n### Automatic Integration\n\nWhen the AI responds, it automatically:\n1. Searches the knowledge base for relevant context\n2. Retrieves past conversations, code, or docs\n3. Includes that context in the response\n\nThis happens transparently – the AI \"remembers\" your past work.\n\n## Management\n\n### View Statistics\n\n```bash\npython3 rag_manage.py stats\n```\n\nOutput:\n```\n📊 OpenClaw RAG Statistics\n\nCollection: openclaw_knowledge\nTotal Documents: 635\n\nBy Source:\n  session-001: 23\n  my-script.py: 5\n  porkbun: 12\n\nBy Type:\n  session: 500\n  workspace: 100\n  skill: 35\n```\n\n### Delete Documents\n\n```bash\n# Delete all sessions\npython3 rag_manage.py delete --by-type session\n\n# Delete specific file\npython3 rag_manage.py delete --by-source \"scripts/voipms_sms_client.py\"\n\n# Reset entire collection\npython3 rag_manage.py reset\n```\n\n### Add Manual Document\n\n```bash\npython3 rag_manage.py add \\\n  --text \"API endpoint: https://api.example.com/endpoint\" \\\n  --source \"api-docs:example.com\" \\\n  --type \"manual\"\n```\n\n## Configuration\n\n### Custom Session Directory\n\n```bash\npython3 ingest_sessions.py --sessions-dir /path/to/sessions\n```\n\n### Chunk Size Control\n\n```bash\npython3 ingest_sessions.py --chunk-size 30 --chunk-overlap 10\n```\n\n### Custom Collection\n\n```python\nfrom rag_system import RAGSystem\nrag = RAGSystem(collection_name=\"my_knowledge\")\n```\n\n## Data Types\n\n| Type | Source Format | Description |\n|------|--------------|-------------|\n| `session` | `session:{key}` | Chat history transcripts |\n| `workspace` | `relative/path/to/file` | Code, configs, docs |\n| `skill` | `skill:{name}` | Skill documentation |\n| `memory` | `MEMORY.md` | Long-term memory entries |\n| `manual` | `{custom}` | Manually added docs |\n| `api` | `api-docs:{name}` | API documentation |\n\n## Performance\n\n- **Embedding model**: `all-MiniLM-L6-v2` (79MB, cached locally)\n- **Storage**: ~100MB per 1,000 documents\n- **Indexing**: ~1,000 documents/minute\n- **Search**: <100ms (after first query)\n\n## Troubleshooting\n\n### No Results Found\n\n```bash\n# Check what's indexed\npython3 rag_manage.py stats\n\n# Try broader query\npython3 rag_query.py \"SMS\"  # instead of \"voip.ms SMS API endpoint\"\n```\n\n### Slow First Search\n\nFirst search loads embeddings (~1-2 seconds). Subsequent searches are instant.\n\n### Duplicate ID Errors\n\n```bash\n# Reset and re-index\npython3 rag_manage.py reset\npython3 ingest_sessions.py\npython3 ingest_docs.py workspace\n```\n\n### ChromaDB Model Download\n\nFirst run downloads embedding model (79MB). Takes 1-2 minutes. Let it complete.\n\n## Best Practices\n\n### Re-index Regularly\n\nAfter significant work:\n```bash\npython3 ingest_sessions.py  # New conversations\npython3 ingest_docs.py workspace  # New code/changes\n```\n\n### Use Specific Queries\n\n```bash\n# Better\npython3 rag_query.py \"voip.ms getSMS method\"\n\n# Too broad\npython3 rag_query.py \"SMS\"\n```\n\n### Filter by Type\n\n```bash\n# Looking for code\npython3 rag_query.py --type workspace \"chromedriver\"\n\n# Looking for past conversations\npython3 rag_query.py --type session \"Reddit\"\n```\n\n### Document Decisions\n\nAfter important decisions, add them manually:\n\n```bash\npython3 rag_manage.py add \\\n  --text \"Decision: Use Playwright for Reddit automation. Reason: Cloudflare bypass handles\" \\\n  --source \"decision:reddit-automation\" \\\n  --type \"decision\"\n```\n\n## Limitations\n\n- Files > 1MB automatically skipped (performance)\n- Python 3.7+ required\n- ~100MB disk per 1,000 documents\n- First search slower (embedding load)\n\n## Integration with OpenClaw\n\nThis skill integrates seamlessly with OpenClaw:\n\n1. **Automatic RAG**: AI automatically retrieves relevant context when responding\n2. **Session history**: All conversations indexed and searchable\n3. **Workspace awareness**: Code and docs indexed for reference\n4. **Skill accessible**: Use from any OpenClaw session or script\n\n## Security Considerations\n\n**⚠️ Important Privacy Note:** This RAG system indexes local data, which may contain:\n- API keys, tokens, or credentials in session transcripts\n- Private messages or personal information\n- Tool results with sensitive data\n- Workspace configuration files\n\n**Recommended:**\n- Review session files before ingestion if concerned about privacy\n- Consider redacting sensitive data from session files\n- Use `rag_manage.py reset` to delete the entire index when needed\n- The ChromaDB persistence at `~/.openclaw/data/rag/` can be deleted to remove all indexed data\n- The auto-update script only runs local ingestion - no remote code fetching\n\n**Path Portability:**\nAll scripts now use dynamic path resolution (`os.path.expanduser()`, `Path(__file__).parent`) for portability across different user environments. No hard-coded absolute paths remain in the codebase.\n\n**Network Calls:**\n- The embedding model (all-MiniLM-L6-v2) is downloaded by ChromaDB on first use via pip\n- No custom network calls, HTTP requests, or sub-process network operations\n- No telemetry or data uploaded to external services (ChromaDB telemetry disabled)\n- All processing and storage is local-only\n\n## Example Workflow\n\n**Scenario:** You're working on a new automation but hit a Cloudflare challenge.\n\n```bash\n# Search for past Cloudflare solutions\npython3 rag_query.py \"Cloudflare bypass selenium\"\n\n# Result shows relevant past conversation:\n# \"Used undetected-chromedriver but failed. Switched to Playwright which handles challenges better.\"\n\n# Now you know the solution before trying it!\n```\n\n## Repository\n\nhttps://git.theta42.com/nova/openclaw-rag-skill\n\n**Published:** clawhub.com\n**Maintainer:** Nova AI Assistant\n**For:** William Mantly (Theta42)\n\n## License\n\nMIT License - Free to use and modify\n\nFile v0.1.2:README.md\n\n# OpenClaw RAG Knowledge System\n\nFull-featured Retrieval-Augmented Generation (RAG) system for OpenClaw - search across chat history, code, documentation, and skills with semantic understanding.\n\n## Features\n\n- **Semantic Search**: Find relevant context by meaning, not just keywords\n- **Multi-Source Indexing**: Sessions, workspace files, skill documentation\n- **Local Vector Store**: ChromaDB with built-in embeddings (no API keys required)\n- **Automatic Integration**: AI automatically consults knowledge base when responding\n- **Type Filtering**: Search by document type (session, workspace, skill, memory)\n- **Management Tools**: Add/remove documents, view statistics, reset collection\n\n## Quick Start\n\n### Installation\n\n```bash\n# Install Python dependency\ncd ~/.openclaw/workspace/rag\npython3 -m pip install --user chromadb\n```\n\n**No API keys required** - This system is fully local:\n- Embeddings: all-MiniLM-L6-v2 (downloaded once, 79MB)\n- Vector store: ChromaDB (persistent disk storage)\n- Data location: `~/.openclaw/data/rag/` (auto-created)\n\nAll operations run offline with no external dependencies besides the initial ChromaDB download.\n\n### Index Your Data\n\n```bash\n# Index all chat sessions\npython3 ingest_sessions.py\n\n# Index workspace code and docs\npython3 ingest_docs.py workspace\n\n# Index skill documentation\npython3 ingest_docs.py skills\n```\n\n### Search the Knowledge Base\n\n```bash\n# Interactive search mode\npython3 rag_query.py -i\n\n# Quick search\npython3 rag_query.py \"how to send SMS\"\n\n# Search by type\npython3 rag_query.py \"voip.ms\" --type session\npython3 rag_query.py \"Porkbun DNS\" --type skill\n```\n\n### Integration in Python Code\n\n```python\nimport sys\nsys.path.insert(0, '/home/william/.openclaw/workspace/rag')\nfrom rag_query_wrapper import search_knowledge\n\n# Search and get structured results\nresults = search_knowledge(\"Reddit account automation\")\nprint(f\"Found {results['count']} results\")\n\n# Format for AI consumption\nfrom rag_query_wrapper import format_for_ai\ncontext = format_for_ai(results)\nprint(context)\n```\n\n## Architecture\n\n```\nrag/\n├── rag_system.py          # Core RAG class (ChromaDB wrapper)\n├── ingest_sessions.py     # Load chat history from sessions\n├── ingest_docs.py         # Load workspace files & skill docs\n├── rag_query.py           # Search the knowledge base\n├── rag_manage.py          # Document management\n├── rag_query_wrapper.py   # Simple Python API\n└── SKILL.md               # OpenClaw skill documentation\n```\n\nData storage: `~/.openclaw/data/rag/` (ChromaDB persistent storage)\n\n## Usage Examples\n\n### Find Past Solutions\n\nWhen you encounter a problem, search for similar past issues:\n\n```bash\npython3 rag_query.py \"cloudflare bypass failed selenium\"\npython3 rag_query.py \"voip.ms SMS client\"\npython3 rag_query.py \"porkbun DNS API\"\n```\n\n### Search Through Codebase\n\nFind code and documentation across your entire workspace:\n\n```bash\npython3 rag_query.py --type workspace \"chromedriver setup\"\npython3 rag_query.py --type workspace \"unifi gateway API\"\n```\n\n### Access Skill Documentation\n\nQuick reference for any openclaw skill:\n\n```bash\npython3 rag_query.py --type skill \"how to check UniFi\"\npython3 rag_query.py --type skill \"Porkbun DNS management\"\n```\n\n### Manage Knowledge Base\n\n```bash\n# View statistics\npython3 rag_manage.py stats\n\n# Delete all sessions\npython3 rag_manage.py delete --by-type session\n\n# Delete specific file\npython3 rag_manage.py delete --by-source \"scripts/voipms_sms_client.py\"\n```\n\n## How It Works\n\n### Document Ingestion\n\n1. **Session transcripts**: Process chat history from `~/.openclaw/agents/main/sessions/*.jsonl`\n   - Handles OpenClaw event format (session metadata, messages, tool calls)\n   - Chunks messages into groups of 20 with overlap\n   - Extracts and formats thinking, tool calls, and results\n\n2. **Workspace files**: Scans workspace for code, docs, configs\n   - Supports: `.py`, `.js`, `.ts`, `.md`, `.json`, `. yaml`, `.sh`, `.html`, `.css`\n   - Skips files > 1MB and binary files\n   - Chunking for long documents\n\n3. **Skills**: Indexes all `SKILL.md` files\n   - Captures skill documentation and usage examples\n   - Organized by skill name\n\n### Semantic Search\n\nChromaDB uses `all-MiniLM-L6-v2` embedding model (79MB) to convert text to vector representations. Similar meanings cluster together, enabling semantic search beyond keyword matching.\n\n### Automatic RAG Integration\n\nWhen the AI responds to a question that could benefit from context, it automatically:\n1. Searches the knowledge base\n2. Retrieves relevant past conversations, code, or docs\n3. Includes that context in the response\n\nThis happens transparently - the AI just \"knows\" about your past work.\n\n## Configuration\n\n### Custom Session Directory\n\n```bash\npython3 ingest_sessions.py --sessions-dir /path/to/sessions\n```\n\n### Chunk Size Control\n\n```bash\npython3 ingest_sessions.py --chunk-size 30 --chunk-overlap 10\n```\n\n### Custom Collection Name\n\n```python\nfrom rag_system import RAGSystem\nrag = RAGSystem(collection_name=\"my_knowledge\")\n```\n\n## Data Types\n\n| Type | Source | Description |\n|------|--------|-------------|\n| **session** | `session:{key}` | Chat history transcripts |\n| **workspace** | `relative/path` | Code, configs, docs |\n| **skill** | `skill:{name}` | Skill documentation |\n| **memory** | `MEMORY.md` | Long-term memory entries |\n| **manual** | `{custom}` | Manually added docs |\n| **api** | `api-docs:{name}` | API documentation |\n\n## Performance\n\n- **Embedding model**: `all-MiniLM-L6-v2` (79MB, cached locally)\n- **Storage**: ~100MB per 1,000 documents\n- **Indexing time**: ~1,000 docs/min\n- **Search time**: <100ms (after first query loads embeddings)\n\n## Troubleshooting\n\n### No Results Found\n\n- Check if anything is indexed: `python3 rag_manage.py stats`\n- Try broader queries or different wording\n- Try without filters: remove `--type` if using it\n\n### Slow First Search\n\nThe first search after ingestion loads embeddings (~1-2 seconds). Subsequent searches are much faster.\n\n### Memory Issues\n\nReset collection if needed:\n```bash\npython3 rag_manage.py reset\n```\n\n### Duplicate ID Errors\n\nIf you see \"Expected IDs to be unique\" errors:\n1. Reset the collection\n2. Re-run ingestion\n3. The fix includes `chunk_index` in ID generation\n\n### ChromaDB Download Stuck\n\nOn first run, ChromaDB downloads the embedding model (~79MB). This takes 1-2 minutes. Let it complete.\n\n## Automatic Updates\n\n### Setup Scheduled Indexing\n\nThe RAG system includes an automatic update script that runs daily:\n\n```bash\n# Manual test\nbash /home/william/.openclaw/workspace/scripts/rag-auto-update.sh\n```\n\n**What it does:**\n- Detects new/updated chat sessions and re-indexes them\n- Re-indexes workspace files (captures code changes)\n- Updates skill documentation\n- Maintains state to avoid re-processing unchanged files\n- Runs via cron at 4:00 AM UTC daily\n\n**Configuration:**\n```bash\n# View cron job\nopenclaw cron list\n\n# Edit schedule (if needed)\nopenclaw cron update <job-id> --schedule \"{\\\"expr\\\":\\\"0 4 * * *\\\"}\"\n```\n\n**State tracking:** `~/.openclaw/workspace/memory/rag-auto-state.json`\n**Log file:** `~/.openclaw/workspace/memory/rag-auto-update.log`\n\n## Best Practices\n\n### Automatic Update Enabled\n\nThe RAG system now automatically updates daily - no manual re-indexing needed.\n\nAfter significant work, you can still manually update:\n```bash\nbash /home/william/.openclaw/workspace/scripts/rag-auto-update.sh\n```\n\n### Use Specific Queries\n\nBetter results with focused queries:\n```bash\n# Good\npython3 rag_query.py \"voip.ms getSMS API method\"\n\n# Less specific\npython3 rag_query.py \"API\"\n```\n\n### Filter by Type\n\nWhen you know the data type:\n```bash\n# Looking for code\npython3 rag_query.py --type workspace \"chromedriver\"\n\n# Looking for past conversations\npython3 rag_query.py --type session \"SMS\"\n```\n\n### Document Decisions\n\nAfter important decisions, add to knowledge base:\n```bash\npython3 rag_manage.py add \\\n  --text \"Decision: Use Playwright not Selenium for Reddit automation. Reason: Better Cloudflare bypass handles. Date: 2026-02-11\" \\\n  --source \"decision:reddit-automation\" \\\n  --type \"decision\"\n```\n\n## Limitations\n\n- Files > 1MB are automatically skipped (performance)\n- First search is slower (embedding load)\n- Requires ~100MB disk space per 1,000 documents\n- Python 3.7+ required\n\n## License\n\nMIT License - Free to use and modify\n\n## Contributing\n\nContributions welcome! Areas for improvement:\n- API documentation indexing from external URLs\n- File system watch for automatic re-indexing\n- Better chunking strategies for long documents\n- Integration with external vector stores (Pinecone, Weaviate)\n\n## Documentation Files\n\n- **CHANGELOG.md** - Version history and changes\n- **SKILL.md** - OpenClaw skill integration guide\n- **package.json** - Skill metadata (no credentials required)\n- **LICENSE** - MIT License\n\n## Author\n\nNova AI Assistant for William Mantly (Theta42)\n\n## Repository\n\nhttps://git.theta42.com/nova/openclaw-rag-skill\nPublished on: clawhub.com\n\nFile v0.1.2:_meta.json\n\n{\n  \"ownerId\": \"kn7bnmqtrcy5z9xs9pryvzeet180g3dk\",\n  \"slug\": \"openclaw-rag-skill\",\n  \"version\": \"0.1.2\",\n  \"publishedAt\": 1770920906786\n}\n\nFile v0.1.2:CHANGELOG.md\n\n# Changelog\n\nAll notable changes to the OpenClaw RAG Knowledge System will be documented in this file.\n\n## [1.0.0] - 2026-02-11\n\n### Added\n- Initial release of RAG Knowledge System for OpenClaw\n- Semantic search using ChromaDB with all-MiniLM-L6-v2 embeddings\n- Multi-source indexing: sessions, workspace files, skill documentation\n- CLI tools: rag_query.py, rag_manage.py, ingest_sessions.py, ingest_docs.py\n- Python API: rag_query_wrapper.py for programmatic access\n- Automatic integration wrapper: rag_context.py for transparent RAG queries\n- RAG-enhanced agent wrapper: rag_agent.py\n- Type filtering: search by document type (session, workspace, skill, memory)\n- Document management: add, delete, reset collection\n- Batch ingestion with intelligent chunking\n- Session parser for OpenClaw event format\n- Automatic daily updates via cron job\n- Comprehensive documentation: README.md, SKILL.md\n\n### Features\n- **Semantic Search**: Find relevant context by meaning, not keywords\n- **Local Vector Store**: ChromaDB with persistent storage (~100MB per 1,000 docs)\n- **Zero Dependencies**: No API keys required (all-MiniLM-L6-v2 is free and local)\n- **Smart Chunking**: Messages grouped by 20 with overlap for context\n- **Multi-Format Support**: Python, JavaScript, Markdown, JSON, YAML, shell scripts\n- **Automatic Updates**: Scheduled cron job runs daily at 4:00 AM UTC\n- **State Tracking**: Avoids re-processing unchanged files\n- **Debug Mode**: Verbose output for troubleshooting\n\n### Bug Fixes\n- Fixed duplicate ID errors by including chunk_index in hash generation\n- Fixed session parser to handle OpenClaw event format correctly\n- Fixed metadata conversion errors (all metadata values as strings)\n\n### Performance\n- Indexing speed: ~1,000 docs/minute\n- Search time: <100ms (after embedding load)\n- Embedding model: 79MB (cached locally)\n- Storage: ~100MB per 1,000 documents\n\n### Documentation\n- Complete SKILL.md with OpenClaw integration guide\n- Comprehensive README.md with examples and troubleshooting\n- Inline help in all CLI tools\n- Best practices and limitations documented\n\n---\n\n## [1.0.1] - 2026-02-11\n\n### Added\n- `package.json` with complete OpenClaw skill metadata\n- `CHANGELOG.md` for version tracking\n- `LICENSE` (MIT) for proper licensing\n\n### Changed\n- `package.json` explicitly declares NO required environment variables (fully local system)\n- Documented data storage path: `~/.openclaw/data/rag/`\n- Enhanced `README.md` with clearer installation instructions\n- Added references to CHANGELOG, LICENSE, and package.json in README\n- Clarified that no API keys or credentials are required\n\n### Documentation\n- Improved documentation transparency to meet security scanner best practices\n- Clearly documented the fully-local nature of the system (no external dependencies)\n\n---\n\n## [1.0.3] - 2026-02-12\n\n### Fixed\n- **Hard-coded paths**: Replaced all absolute paths with dynamic resolution\n  - `rag_context.py`: Now uses `os.path.dirname(os.path.abspath(__file__))`\n  - `scripts/rag-auto-update.sh`: Uses `$HOME`, `OPENCLAW_DIR`, and relative paths\n  - Removed hard-coded `/home/william/.openclaw/` references\n  - All scripts now portable across different user environments\n\n### Changed\n- **Documentation**: Updated SKILL.md with path portability notes\n  - Documented that all paths use dynamic resolution\n  - Confirmed no custom network calls or external telemetry\n  - Added \"Network Calls\" section addressing security scan concerns\n- **rag_query_wrapper.py**: Removed hard-coded path example from docstring\n\n### Security\n- Verified: `rag_system.py` has no network calls (only imports chromadb)\n- Verified: `scripts/rag-auto-update.sh` has no network activity\n- Confirmed: ChromaDB telemetry is disabled (`anonymized_telemetry=False`)\n- Confirmed: All processing and storage is local-only\n\n### Addressed Feedback\n- Fixed ClawHub security scan concerns about hard-coded paths\n- Fixed concerns about missing code review (rag_system.py is fully auditable)\n- Documented network behavior (only model download by ChromaDB on first run)\n\n---\n\n## [Unreleased]\n\n### Planned\n- API documentation indexing from external URLs\n- Automatic re-indexing on file system events (inotify)\n- Better chunking strategies for long documents\n- Integration with external vector stores (Pinecone, Weaviate)\n- Webhook notifications for automated content processing\n- Hybrid search (semantic + keyword)\n- Query history and analytics\n- Export/import of vector database\n\n---\n\n## [1.0.2] - 2026-02-12\n\n### Added\n- YAML front matter to SKILL.md with `name: rag` and `description` for ClawHub compatibility\n- `Security Considerations` section documenting privacy implications and sensitive data risks\n- `scripts/rag-auto-update.sh` included in skill package (previously in separate location)\n- `.skill` package for ClawHub distribution (28KB, 14 files)\n\n### Changed\n- Updated package.json description to match SKILL.md front matter\n- Documented auto-update script behavior for security review (local-only ingestion)\n- Clarified ChromaDB storage location and data deletion procedures\n\n### Fixed\n- **Cron job HTTP 500 errors**: Changed from `sessionTarget: \"main\"` to `isolated` to avoid flooding chat with thousands of lines of output\n- **Cron schedule**: Fixed from `0 4 * * *` to `0 0 * * *` to match actual midnight UTC execution time\n\n### Security\n- Documented that RAG indexes all session transcripts and workspace files (may contain API keys, credentials, private messages)\n- Added recommendations for privacy-conscious use: review sessions before ingestion, use `rag_manage.py reset` to delete all indexed data\n- Confirmed auto-update script only runs local ingestion scripts - no remote code fetching\n\n### Documentation\n- Added detailed security warnings in SKILL.md\n- Explained how to delete ChromaDB persistence directory (`~/.openclaw/data/rag/`)\n- Provided guidance on redacting sensitive data before ingestion\n\n---\n\n## Version Guidelines\n\nThis project follows [Semantic Versioning](https://semver.org/):\n\n- **MAJOR** version: Incompatible API changes\n- **MINOR** version: Backwards-compatible functionality additions\n- **PATCH** version: Backwards-compatible bug fixes\n\n## Categories\n\n- **Added**: New features\n- **Changed**: Changes in existing functionality\n- **Deprecated**: Soon-to-be removed features\n- **Removed**: Removed features\n- **Fixed**: Bug fixes\n- **Security**: Security vulnerabilities\n\nFile v0.1.2:package.json\n\n{\n  \"name\": \"rag-openclaw\",\n  \"version\": \"1.0.3\",\n  \"description\": \"RAG Knowledge System for OpenClaw - Semantic search across chat history, code, docs, and skills with automatic memory retrieval\",\n  \"homepage\": \"http://git.theta42.com/nova/openclaw-rag-skill\",\n  \"author\": {\n    \"name\": \"Nova AI\",\n    \"email\": \"nova@vm42.us\"\n  },\n  \"owner\": \"wmantly\",\n  \"openclaw\": {\n    \"always\": false,\n    \"capabilities\": []\n  },\n  \"environment\": {\n    \"required\": {},\n    \"optional\": {},\n    \"config\": {\n      \"paths\": [\n        \"~/.openclaw/data/rag/\"\n      ],\n      \"help\": \"ChromaDB storage location. No configuration required - system auto-creates data directory on first use.\"\n    }\n  },\n  \"install\": {\n    \"type\": \"instruction\",\n    \"steps\": [\n      \"1. Install Python dependency: pip3 install --user chromadb\",\n      \"2. Install location: ~/.openclaw/workspace/rag/ (created automatically)\",\n      \"3. Data storage: ~/.openclaw/data/rag/ (auto-created on first run)\",\n      \"4. No API keys or credentials required - fully local system\"\n    ]\n  },\n  \"scripts\": {\n    \"ingest:sessions\": \"python3 ingest_sessions.py\",\n    \"ingest:workspace\": \"python3 ingest_docs.py workspace\",\n    \"ingest:skills\": \"python3 ingest_docs.py skills\",\n    \"search\": \"python3 rag_query.py\",\n    \"update\": \"bash scripts/rag-auto-update.sh\",\n    \"stats\": \"python3 rag_manage.py stats\",\n    \"manage\": \"python3 rag_manage.py\"\n  },\n  \"keywords\": [\n    \"rag\",\n    \"knowledge\",\n    \"semantic-search\",\n    \"chromadb\",\n    \"memory\",\n    \"retrieval-augmented-generation\"\n  ],\n  \"license\": \"MIT\"\n}\n\nArchive v0.1.1: 15 files, 29384 bytes\n\nFiles: ingest_docs.py (7800b), ingest_sessions.py (8874b), launch_rag_agent.sh (1240b), package.json (1559b), rag_agent.py (6190b), rag_context.py (2537b), rag_manage.py (6887b), rag_query_quick.py (2391b), rag_query_wrapper.py (3221b), rag_query.py (5293b), rag_system.py (8784b), README.md (8996b), scripts/rag-auto-update.sh (3439b), SKILL.md (9435b), _meta.json (137b)\n\nFile v0.1.1:SKILL.md\n\n---\nname: rag\ndescription: Complete RAG (Retrieval-Augmented Generation) system for OpenClaw. Indexes chat sessions, workspace code, documentation, and skills into local ChromaDB for semantic search. Enables finding past solutions, code patterns, and decisions instantly. Uses local embeddings (all-MiniLM-L6-v2) with no API keys required. Automatically ingests and updates knowledge base from ~/.openclaw/agents/main/sessions and workspace files.\n---\n\n# OpenClaw RAG Knowledge System\n\n**Retrieval-Augmented Generation for OpenClaw – Search chat history, code, docs, and skills with semantic understanding**\n\n## Overview\n\nThis skill provides a complete RAG (Retrieval-Augmented Generation) system for OpenClaw. It indexes your entire knowledge base – chat transcripts, workspace code, skill documentation – and enables semantic search across everything.\n\n**Key features:**\n- 🧠 Semantic search across all conversations and code\n- 📚 Automatic knowledge base management\n- 🔍 Find past solutions, code patterns, decisions instantly\n- 💾 Local ChromaDB storage (no API keys required)\n- 🚀 Automatic AI integration – retrieves context transparently\n\n## Installation\n\n### Prerequisites\n\n- Python 3.7+\n- OpenClaw workspace\n\n### Setup\n\n```bash\n# Navigate to your OpenClaw workspace\ncd ~/.openclaw/workspace/skills/rag-openclaw\n\n# Install ChromaDB (one-time)\npip3 install --user chromadb\n\n# That's it!\n```\n\n## Quick Start\n\n### 1. Index Your Knowledge\n\n```bash\n# Index all chat history\npython3 ingest_sessions.py\n\n# Index workspace code and docs\npython3 ingest_docs.py workspace\n\n# Index skill documentation\npython3 ingest_docs.py skills\n```\n\n### 2. Search the Knowledge Base\n\n```bash\n# Interactive search mode\npython3 rag_query.py -i\n\n# Quick search\npython3 rag_query.py \"how to send SMS via voip.ms\"\n\n# Search by type\npython3 rag_query.py \"porkbun DNS\" --type skill\npython3 rag_query.py \"chromedriver\" --type workspace\npython3 rag_query.py \"Reddit automation\" --type session\n```\n\n### 3. Check Statistics\n\n```bash\n# See what's indexed\npython3 rag_manage.py stats\n```\n\n## Usage Examples\n\n### Finding Past Solutions\n\nHit a problem? Search for how you solved it before:\n\n```bash\npython3 rag_query.py \"cloudflare bypass selenium\"\npython3 rag_query.py \"voip.ms SMS configuration\"\npython3 rag_query.py \"porkbun update DNS record\"\n```\n\n### Searching Through Codebase\n\nFind specific code or documentation:\n\n```bash\npython3 rag_query.py --type workspace \"unifi gateway API\"\npython3 rag_query.py --type workspace \"SMS client\"\n```\n\n### Quick Reference\n\nAccess skill documentation without digging through files:\n\n```bash\npython3 rag_query.py --type skill \"how to monitor UniFi\"\npython3 rag_query.py --type skill \"Porkbun tool usage\"\n```\n\n### Programmatic Use\n\nFrom within Python scripts or OpenClaw sessions:\n\n```python\nimport sys\nsys.path.insert(0, '/home/william/.openclaw/workspace/skills/rag-openclaw')\nfrom rag_query_wrapper import search_knowledge, format_for_ai\n\n# Search and get structured results\nresults = search_knowledge(\"Reddit account automation\")\nprint(f\"Found {results['count']} relevant items\")\n\n# Format for AI consumption\ncontext = format_for_ai(results)\nprint(context)\n```\n\n## Files Reference\n\n| File | Purpose |\n|------|---------|\n| `rag_system.py` | Core RAG class (ChromaDB wrapper) |\n| `ingest_sessions.py` | Index chat history |\n| `ingest_docs.py` | Index workspace files & skills |\n| `rag_query.py` | Search interface (CLI & interactive) |\n| `rag_manage.py` | Document management (stats, delete, reset) |\n| `rag_query_wrapper.py` | Simple Python API for programmatic use |\n| `README.md` | Full documentation |\n\n## How It Works\n\n### Indexing\n\n**Sessions:**\n- Reads `~/.openclaw/agents/main/sessions/*.jsonl`\n- Handles OpenClaw event format (session metadata, messages, tool calls)\n- Chunks messages (20 per chunk, 5 message overlap)\n- Extracts and formats thinking, tool calls, results\n\n**Workspace:**\n- Scans for `.py`, `.js`, `.ts`, `.md`, `.json`, `.yaml`, `.sh`, `.html`, `.css`\n- Skips files > 1MB and binary files\n- Chunks long documents for better retrieval\n\n**Skills:**\n- Indexes all `SKILL.md` files\n- Organized by skill name for easy reference\n\n### Search\n\nChromaDB uses `all-MiniLM-L6-v2` embeddings to convert text to vectors. Similar meanings cluster together, enabling semantic search by *meaning* not just *keywords*.\n\n### Automatic Integration\n\nWhen the AI responds, it automatically:\n1. Searches the knowledge base for relevant context\n2. Retrieves past conversations, code, or docs\n3. Includes that context in the response\n\nThis happens transparently – the AI \"remembers\" your past work.\n\n## Management\n\n### View Statistics\n\n```bash\npython3 rag_manage.py stats\n```\n\nOutput:\n```\n📊 OpenClaw RAG Statistics\n\nCollection: openclaw_knowledge\nTotal Documents: 635\n\nBy Source:\n  session-001: 23\n  my-script.py: 5\n  porkbun: 12\n\nBy Type:\n  session: 500\n  workspace: 100\n  skill: 35\n```\n\n### Delete Documents\n\n```bash\n# Delete all sessions\npython3 rag_manage.py delete --by-type session\n\n# Delete specific file\npython3 rag_manage.py delete --by-source \"scripts/voipms_sms_client.py\"\n\n# Reset entire collection\npython3 rag_manage.py reset\n```\n\n### Add Manual Document\n\n```bash\npython3 rag_manage.py add \\\n  --text \"API endpoint: https://api.example.com/endpoint\" \\\n  --source \"api-docs:example.com\" \\\n  --type \"manual\"\n```\n\n## Configuration\n\n### Custom Session Directory\n\n```bash\npython3 ingest_sessions.py --sessions-dir /path/to/sessions\n```\n\n### Chunk Size Control\n\n```bash\npython3 ingest_sessions.py --chunk-size 30 --chunk-overlap 10\n```\n\n### Custom Collection\n\n```python\nfrom rag_system import RAGSystem\nrag = RAGSystem(collection_name=\"my_knowledge\")\n```\n\n## Data Types\n\n| Type | Source Format | Description |\n|------|--------------|-------------|\n| `session` | `session:{key}` | Chat history transcripts |\n| `workspace` | `relative/path/to/file` | Code, configs, docs |\n| `skill` | `skill:{name}` | Skill documentation |\n| `memory` | `MEMORY.md` | Long-term memory entries |\n| `manual` | `{custom}` | Manually added docs |\n| `api` | `api-docs:{name}` | API documentation |\n\n## Performance\n\n- **Embedding model**: `all-MiniLM-L6-v2` (79MB, cached locally)\n- **Storage**: ~100MB per 1,000 documents\n- **Indexing**: ~1,000 documents/minute\n- **Search**: <100ms (after first query)\n\n## Troubleshooting\n\n### No Results Found\n\n```bash\n# Check what's indexed\npython3 rag_manage.py stats\n\n# Try broader query\npython3 rag_query.py \"SMS\"  # instead of \"voip.ms SMS API endpoint\"\n```\n\n### Slow First Search\n\nFirst search loads embeddings (~1-2 seconds). Subsequent searches are instant.\n\n### Duplicate ID Errors\n\n```bash\n# Reset and re-index\npython3 rag_manage.py reset\npython3 ingest_sessions.py\npython3 ingest_docs.py workspace\n```\n\n### ChromaDB Model Download\n\nFirst run downloads embedding model (79MB). Takes 1-2 minutes. Let it complete.\n\n## Best Practices\n\n### Re-index Regularly\n\nAfter significant work:\n```bash\npython3 ingest_sessions.py  # New conversations\npython3 ingest_docs.py workspace  # New code/changes\n```\n\n### Use Specific Queries\n\n```bash\n# Better\npython3 rag_query.py \"voip.ms getSMS method\"\n\n# Too broad\npython3 rag_query.py \"SMS\"\n```\n\n### Filter by Type\n\n```bash\n# Looking for code\npython3 rag_query.py --type workspace \"chromedriver\"\n\n# Looking for past conversations\npython3 rag_query.py --type session \"Reddit\"\n```\n\n### Document Decisions\n\nAfter important decisions, add them manually:\n\n```bash\npython3 rag_manage.py add \\\n  --text \"Decision: Use Playwright for Reddit automation. Reason: Cloudflare bypass handles\" \\\n  --source \"decision:reddit-automation\" \\\n  --type \"decision\"\n```\n\n## Limitations\n\n- Files > 1MB automatically skipped (performance)\n- Python 3.7+ required\n- ~100MB disk per 1,000 documents\n- First search slower (embedding load)\n\n## Integration with OpenClaw\n\nThis skill integrates seamlessly with OpenClaw:\n\n1. **Automatic RAG**: AI automatically retrieves relevant context when responding\n2. **Session history**: All conversations indexed and searchable\n3. **Workspace awareness**: Code and docs indexed for reference\n4. **Skill accessible**: Use from any OpenClaw session or script\n\n## Security Considerations\n\n**⚠️ Important Privacy Note:** This RAG system indexes local data, which may contain:\n- API keys, tokens, or credentials in session transcripts\n- Private messages or personal information\n- Tool results with sensitive data\n- Workspace configuration files\n\n**Recommended:**\n- Review session files before ingestion if concerned about privacy\n- Consider redacting sensitive data from session files\n- Use `rag_manage.py reset` to delete the entire index when needed\n- The ChromaDB persistence at `~/.openclaw/data/rag/` can be deleted to remove all indexed data\n- The auto-update script only runs local ingestion - no remote code fetching\n\n## Example Workflow\n\n**Scenario:** You're working on a new automation but hit a Cloudflare challenge.\n\n```bash\n# Search for past Cloudflare solutions\npython3 rag_query.py \"Cloudflare bypass selenium\"\n\n# Result shows relevant past conversation:\n# \"Used undetected-chromedriver but failed. Switched to Playwright which handles challenges better.\"\n\n# Now you know the solution before trying it!\n```\n\n## Repository\n\nhttps://git.theta42.com/nova/openclaw-rag-skill\n\n**Published:** clawhub.com\n**Maintainer:** Nova AI Assistant\n**For:** William Mantly (Theta42)\n\n## License\n\nMIT License - Free to use and modify\n\nFile v0.1.1:README.md\n\n# OpenClaw RAG Knowledge System\n\nFull-featured Retrieval-Augmented Generation (RAG) system for OpenClaw - search across chat history, code, documentation, and skills with semantic understanding.\n\n## Features\n\n- **Semantic Search**: Find relevant context by meaning, not just keywords\n- **Multi-Source Indexing**: Sessions, workspace files, skill documentation\n- **Local Vector Store**: ChromaDB with built-in embeddings (no API keys required)\n- **Automatic Integration**: AI automatically consults knowledge base when responding\n- **Type Filtering**: Search by document type (session, workspace, skill, memory)\n- **Management Tools**: Add/remove documents, view statistics, reset collection\n\n## Quick Start\n\n### Installation\n\n```bash\n# Install Python dependency\ncd ~/.openclaw/workspace/rag\npython3 -m pip install --user chromadb\n```\n\n**No API keys required** - This system is fully local:\n- Embeddings: all-MiniLM-L6-v2 (downloaded once, 79MB)\n- Vector store: ChromaDB (persistent disk storage)\n- Data location: `~/.openclaw/data/rag/` (auto-created)\n\nAll operations run offline with no external dependencies besides the initial ChromaDB download.\n\n### Index Your Data\n\n```bash\n# Index all chat sessions\npython3 ingest_sessions.py\n\n# Index workspace code and docs\npython3 ingest_docs.py workspace\n\n# Index skill documentation\npython3 ingest_docs.py skills\n```\n\n### Search the Knowledge Base\n\n```bash\n# Interactive search mode\npython3 rag_query.py -i\n\n# Quick search\npython3 rag_query.py \"how to send SMS\"\n\n# Search by type\npython3 rag_query.py \"voip.ms\" --type session\npython3 rag_query.py \"Porkbun DNS\" --type skill\n```\n\n### Integration in Python Code\n\n```python\nimport sys\nsys.path.insert(0, '/home/william/.openclaw/workspace/rag')\nfrom rag_query_wrapper import search_knowledge\n\n# Search and get structured results\nresults = search_knowledge(\"Reddit account automation\")\nprint(f\"Found {results['count']} results\")\n\n# Format for AI consumption\nfrom rag_query_wrapper import format_for_ai\ncontext = format_for_ai(results)\nprint(context)\n```\n\n## Architecture\n\n```\nrag/\n├── rag_system.py          # Core RAG class (ChromaDB wrapper)\n├── ingest_sessions.py     # Load chat history from sessions\n├── ingest_docs.py         # Load workspace files & skill docs\n├── rag_query.py           # Search the knowledge base\n├── rag_manage.py          # Document management\n├── rag_query_wrapper.py   # Simple Python API\n└── SKILL.md               # OpenClaw skill documentation\n```\n\nData storage: `~/.openclaw/data/rag/` (ChromaDB persistent storage)\n\n## Usage Examples\n\n### Find Past Solutions\n\nWhen you encounter a problem, search for similar past issues:\n\n```bash\npython3 rag_query.py \"cloudflare bypass failed selenium\"\npython3 rag_query.py \"voip.ms SMS client\"\npython3 rag_query.py \"porkbun DNS API\"\n```\n\n### Search Through Codebase\n\nFind code and documentation across your entire workspace:\n\n```bash\npython3 rag_query.py --type workspace \"chromedriver setup\"\npython3 rag_query.py --type workspace \"unifi gateway API\"\n```\n\n### Access Skill Documentation\n\nQuick reference for any openclaw skill:\n\n```bash\npython3 rag_query.py --type skill \"how to check UniFi\"\npython3 rag_query.py --type skill \"Porkbun DNS management\"\n```\n\n### Manage Knowledge Base\n\n```bash\n# View statistics\npython3 rag_manage.py stats\n\n# Delete all sessions\npython3 rag_manage.py delete --by-type session\n\n# Delete specific file\npython3 rag_manage.py delete --by-source \"scripts/voipms_sms_client.py\"\n```\n\n## How It Works\n\n### Document Ingestion\n\n1. **Session transcripts**: Process chat history from `~/.openclaw/agents/main/sessions/*.jsonl`\n   - Handles OpenClaw event format (session metadata, messages, tool calls)\n   - Chunks messages into groups of 20 with overlap\n   - Extracts and formats thinking, tool calls, and results\n\n2. **Workspace files**: Scans workspace for code, docs, configs\n   - Supports: `.py`, `.js`, `.ts`, `.md`, `.json`, `. yaml`, `.sh`, `.html`, `.css`\n   - Skips files > 1MB and binary files\n   - Chunking for long documents\n\n3. **Skills**: Indexes all `SKILL.md` files\n   - Captures skill documentation and usage examples\n   - Organized by skill name\n\n### Semantic Search\n\nChromaDB uses `all-MiniLM-L6-v2` embedding model (79MB) to convert text to vector representations. Similar meanings cluster together, enabling semantic search beyond keyword matching.\n\n### Automatic RAG Integration\n\nWhen the AI responds to a question that could benefit from context, it automatically:\n1. Searches the knowledge base\n2. Retrieves relevant past conversations, code, or docs\n3. Includes that context in the response\n\nThis happens transparently - the AI just \"knows\" about your past work.\n\n## Configuration\n\n### Custom Session Directory\n\n```bash\npython3 ingest_sessions.py --sessions-dir /path/to/sessions\n```\n\n### Chunk Size Control\n\n```bash\npython3 ingest_sessions.py --chunk-size 30 --chunk-overlap 10\n```\n\n### Custom Collection Name\n\n```python\nfrom rag_system import RAGSystem\nrag = RAGSystem(collection_name=\"my_knowledge\")\n```\n\n## Data Types\n\n| Type | Source | Description |\n|------|--------|-------------|\n| **session** | `session:{key}` | Chat history transcripts |\n| **workspace** | `relative/path` | Code, configs, docs |\n| **skill** | `skill:{name}` | Skill documentation |\n| **memory** | `MEMORY.md` | Long-term memory entries |\n| **manual** | `{custom}` | Manually added docs |\n| **api** | `api-docs:{name}` | API documentation |\n\n## Performance\n\n- **Embedding model**: `all-MiniLM-L6-v2` (79MB, cached locally)\n- **Storage**: ~100MB per 1,000 documents\n- **Indexing time**: ~1,000 docs/min\n- **Search time**: <100ms (after first query loads embeddings)\n\n## Troubleshooting\n\n### No Results Found\n\n- Check if anything is indexed: `python3 rag_manage.py stats`\n- Try broader queries or different wording\n- Try without filters: remove `--type` if using it\n\n### Slow First Search\n\nThe first search after ingestion loads embeddings (~1-2 seconds). Subsequent searches are much faster.\n\n### Memory Issues\n\nReset collection if needed:\n```bash\npython3 rag_manage.py reset\n```\n\n### Duplicate ID Errors\n\nIf you see \"Expected IDs to be unique\" errors:\n1. Reset the collection\n2. Re-run ingestion\n3. The fix includes `chunk_index` in ID generation\n\n### ChromaDB Download Stuck\n\nOn first run, ChromaDB downloads the embedding model (~79MB). This takes 1-2 minutes. Let it complete.\n\n## Automatic Updates\n\n### Setup Scheduled Indexing\n\nThe RAG system includes an automatic update script that runs daily:\n\n```bash\n# Manual test\nbash /home/william/.openclaw/workspace/scripts/rag-auto-update.sh\n```\n\n**What it does:**\n- Detects new/updated chat sessions and re-indexes them\n- Re-indexes workspace files (captures code changes)\n- Updates skill documentation\n- Maintains state to avoid re-processing unchanged files\n- Runs via cron at 4:00 AM UTC daily\n\n**Configuration:**\n```bash\n# View cron job\nopenclaw cron list\n\n# Edit schedule (if needed)\nopenclaw cron update <job-id> --schedule \"{\\\"expr\\\":\\\"0 4 * * *\\\"}\"\n```\n\n**State tracking:** `~/.openclaw/workspace/memory/rag-auto-state.json`\n**Log file:** `~/.openclaw/workspace/memory/rag-auto-update.log`\n\n## Best Practices\n\n### Automatic Update Enabled\n\nThe RAG system now automatically updates daily - no manual re-indexing needed.\n\nAfter significant work, you can still manually update:\n```bash\nbash /home/william/.openclaw/workspace/scripts/rag-auto-update.sh\n```\n\n### Use Specific Queries\n\nBetter results with focused queries:\n```bash\n# Good\npython3 rag_query.py \"voip.ms getSMS API method\"\n\n# Less specific\npython3 rag_query.py \"API\"\n```\n\n### Filter by Type\n\nWhen you know the data type:\n```bash\n# Looking for code\npython3 rag_query.py --type workspace \"chromedriver\"\n\n# Looking for past conversations\npython3 rag_query.py --type session \"SMS\"\n```\n\n### Document Decisions\n\nAfter important decisions, add to knowledge base:\n```bash\npython3 rag_manage.py add \\\n  --text \"Decision: Use Playwright not Selenium for Reddit automation. Reason: Better Cloudflare bypass handles. Date: 2026-02-11\" \\\n  --source \"decision:reddit-automation\" \\\n  --type \"decision\"\n```\n\n## Limitations\n\n- Files > 1MB are automatically skipped (performance)\n- First search is slower (embedding load)\n- Requires ~100MB disk space per 1,000 documents\n- Python 3.7+ required\n\n## License\n\nMIT License - Free to use and modify\n\n## Contributing\n\nContributions welcome! Areas for improvement:\n- API documentation indexing from external URLs\n- File system watch for automatic re-indexing\n- Better chunking strategies for long documents\n- Integration with external vector stores (Pinecone, Weaviate)\n\n## Documentation Files\n\n- **CHANGELOG.md** - Version history and changes\n- **SKILL.md** - OpenClaw skill integration guide\n- **package.json** - Skill metadata (no credentials required)\n- **LICENSE** - MIT License\n\n## Author\n\nNova AI Assistant for William Mantly (Theta42)\n\n## Repository\n\nhttps://git.theta42.com/nova/openclaw-rag-skill\nPublished on: clawhub.com\n\nFile v0.1.1:_meta.json\n\n{\n  \"ownerId\": \"kn7bnmqtrcy5z9xs9pryvzeet180g3dk\",\n  \"slug\": \"openclaw-rag-skill\",\n  \"version\": \"0.1.1\",\n  \"publishedAt\": 1770910844659\n}\n\nFile v0.1.1:package.json\n\n{\n  \"name\": \"rag-openclaw\",\n  \"version\": \"1.0.2\",\n  \"description\": \"RAG Knowledge System for OpenClaw - Semantic search across chat history, code, docs, and skills with automatic memory retrieval\",\n  \"homepage\": \"http://git.theta42.com/nova/openclaw-rag-skill\",\n  \"author\": {\n    \"name\": \"Nova AI\",\n    \"email\": \"nova@vm42.us\"\n  },\n  \"owner\": \"wmantly\",\n  \"openclaw\": {\n    \"always\": false,\n    \"capabilities\": []\n  },\n  \"environment\": {\n    \"required\": {},\n    \"optional\": {},\n    \"config\": {\n      \"paths\": [\n        \"~/.openclaw/data/rag/\"\n      ],\n      \"help\": \"ChromaDB storage location. No configuration required - system auto-creates data directory on first use.\"\n    }\n  },\n  \"install\": {\n    \"type\": \"instruction\",\n    \"steps\": [\n      \"1. Install Python dependency: pip3 install --user chromadb\",\n      \"2. Install location: ~/.openclaw/workspace/rag/ (created automatically)\",\n      \"3. Data storage: ~/.openclaw/data/rag/ (auto-created on first run)\",\n      \"4. No API keys or credentials required - fully local system\"\n    ]\n  },\n  \"scripts\": {\n    \"ingest:sessions\": \"python3 ingest_sessions.py\",\n    \"ingest:workspace\": \"python3 ingest_docs.py workspace\",\n    \"ingest:skills\": \"python3 ingest_docs.py skills\",\n    \"search\": \"python3 rag_query.py\",\n    \"update\": \"bash scripts/rag-auto-update.sh\",\n    \"stats\": \"python3 rag_manage.py stats\",\n    \"manage\": \"python3 rag_manage.py\"\n  },\n  \"keywords\": [\n    \"rag\",\n    \"knowledge\",\n    \"semantic-search\",\n    \"chromadb\",\n    \"memory\",\n    \"retrieval-augmented-generation\"\n  ],\n  \"license\": \"MIT\"\n}\n\nArchive v0.1.0: 14 files, 25107 bytes\n\nFiles: ingest_docs.py (7800b), ingest_sessions.py (8874b), launch_rag_agent.sh (1240b), package.json (1559b), rag_agent.py (6190b), rag_context.py (2537b), rag_manage.py (6887b), rag_query_quick.py (2391b), rag_query_wrapper.py (3221b), rag_query.py (5293b), rag_system.py (8784b), scripts/rag-auto-update.sh (3439b), SKILL.md (8315b), _meta.json (137b)\n\nFile v0.1.0:SKILL.md\n\n# OpenClaw RAG Knowledge System\n\n**Retrieval-Augmented Generation for OpenClaw – Search chat history, code, docs, and skills with semantic understanding**\n\n## Overview\n\nThis skill provides a complete RAG (Retrieval-Augmented Generation) system for OpenClaw. It indexes your entire knowledge base – chat transcripts, workspace code, skill documentation – and enables semantic search across everything.\n\n**Key features:**\n- 🧠 Semantic search across all conversations and code\n- 📚 Automatic knowledge base management\n- 🔍 Find past solutions, code patterns, decisions instantly\n- 💾 Local ChromaDB storage (no API keys required)\n- 🚀 Automatic AI integration – retrieves context transparently\n\n## Installation\n\n### Prerequisites\n\n- Python 3.7+\n- OpenClaw workspace\n\n### Setup\n\n```bash\n# Navigate to your OpenClaw workspace\ncd ~/.openclaw/workspace/skills/rag-openclaw\n\n# Install ChromaDB (one-time)\npip3 install --user chromadb\n\n# That's it!\n```\n\n## Quick Start\n\n### 1. Index Your Knowledge\n\n```bash\n# Index all chat history\npython3 ingest_sessions.py\n\n# Index workspace code and docs\npython3 ingest_docs.py workspace\n\n# Index skill documentation\npython3 ingest_docs.py skills\n```\n\n### 2. Search the Knowledge Base\n\n```bash\n# Interactive search mode\npython3 rag_query.py -i\n\n# Quick search\npython3 rag_query.py \"how to send SMS via voip.ms\"\n\n# Search by type\npython3 rag_query.py \"porkbun DNS\" --type skill\npython3 rag_query.py \"chromedriver\" --type workspace\npython3 rag_query.py \"Reddit automation\" --type session\n```\n\n### 3. Check Statistics\n\n```bash\n# See what's indexed\npython3 rag_manage.py stats\n```\n\n## Usage Examples\n\n### Finding Past Solutions\n\nHit a problem? Search for how you solved it before:\n\n```bash\npython3 rag_query.py \"cloudflare bypass selenium\"\npython3 rag_query.py \"voip.ms SMS configuration\"\npython3 rag_query.py \"porkbun update DNS record\"\n```\n\n### Searching Through Codebase\n\nFind specific code or documentation:\n\n```bash\npython3 rag_query.py --type workspace \"unifi gateway API\"\npython3 rag_query.py --type workspace \"SMS client\"\n```\n\n### Quick Reference\n\nAccess skill documentation without digging through files:\n\n```bash\npython3 rag_query.py --type skill \"how to monitor UniFi\"\npython3 rag_query.py --type skill \"Porkbun tool usage\"\n```\n\n### Programmatic Use\n\nFrom within Python scripts or OpenClaw sessions:\n\n```python\nimport sys\nsys.path.insert(0, '/home/william/.openclaw/workspace/skills/rag-openclaw')\nfrom rag_query_wrapper import search_knowledge, format_for_ai\n\n# Search and get structured results\nresults = search_knowledge(\"Reddit account automation\")\nprint(f\"Found {results['count']} relevant items\")\n\n# Format for AI consumption\ncontext = format_for_ai(results)\nprint(context)\n```\n\n## Files Reference\n\n| File | Purpose |\n|------|---------|\n| `rag_system.py` | Core RAG class (ChromaDB wrapper) |\n| `ingest_sessions.py` | Index chat history |\n| `ingest_docs.py` | Index workspace files & skills |\n| `rag_query.py` | Search interface (CLI & interactive) |\n| `rag_manage.py` | Document management (stats, delete, reset) |\n| `rag_query_wrapper.py` | Simple Python API for programmatic use |\n| `README.md` | Full documentation |\n\n## How It Works\n\n### Indexing\n\n**Sessions:**\n- Reads `~/.openclaw/agents/main/sessions/*.jsonl`\n- Handles OpenClaw event format (session metadata, messages, tool calls)\n- Chunks messages (20 per chunk, 5 message overlap)\n- Extracts and formats thinking, tool calls, results\n\n**Workspace:**\n- Scans for `.py`, `.js`, `.ts`, `.md`, `.json`, `.yaml`, `.sh`, `.html`, `.css`\n- Skips files > 1MB and binary files\n- Chunks long documents for better retrieval\n\n**Skills:**\n- Indexes all `SKILL.md` files\n- Organized by skill name for easy reference\n\n### Search\n\nChromaDB uses `all-MiniLM-L6-v2` embeddings to convert text to vectors. Similar meanings cluster together, enabling semantic search by *meaning* not just *keywords*.\n\n### Automatic Integration\n\nWhen the AI responds, it automatically:\n1. Searches the knowledge base for relevant context\n2. Retrieves past conversations, code, or docs\n3. Includes that context in the response\n\nThis happens transparently – the AI \"remembers\" your past work.\n\n## Management\n\n### View Statistics\n\n```bash\npython3 rag_manage.py stats\n```\n\nOutput:\n```\n📊 OpenClaw RAG Statistics\n\nCollection: openclaw_knowledge\nTotal Documents: 635\n\nBy Source:\n  session-001: 23\n  my-script.py: 5\n  porkbun: 12\n\nBy Type:\n  session: 500\n  workspace: 100\n  skill: 35\n```\n\n### Delete Documents\n\n```bash\n# Delete all sessions\npython3 rag_manage.py delete --by-type session\n\n# Delete specific file\npython3 rag_manage.py delete --by-source \"scripts/voipms_sms_client.py\"\n\n# Reset entire collection\npython3 rag_manage.py reset\n```\n\n### Add Manual Document\n\n```bash\npython3 rag_manage.py add \\\n  --text \"API endpoint: https://api.example.com/endpoint\" \\\n  --source \"api-docs:example.com\" \\\n  --type \"manual\"\n```\n\n## Configuration\n\n### Custom Session Directory\n\n```bash\npython3 ingest_sessions.py --sessions-dir /path/to/sessions\n```\n\n### Chunk Size Control\n\n```bash\npython3 ingest_sessions.py --chunk-size 30 --chunk-overlap 10\n```\n\n### Custom Collection\n\n```python\nfrom rag_system import RAGSystem\nrag = RAGSystem(collection_name=\"my_knowledge\")\n```\n\n## Data Types\n\n| Type | Source Format | Description |\n|------|--------------|-------------|\n| `session` | `session:{key}` | Chat history transcripts |\n| `workspace` | `relative/path/to/file` | Code, configs, docs |\n| `skill` | `skill:{name}` | Skill documentation |\n| `memory` | `MEMORY.md` | Long-term memory entries |\n| `manual` | `{custom}` | Manually added docs |\n| `api` | `api-docs:{name}` | API documentation |\n\n## Performance\n\n- **Embedding model**: `all-MiniLM-L6-v2` (79MB, cached locally)\n- **Storage**: ~100MB per 1,000 documents\n- **Indexing**: ~1,000 documents/minute\n- **Search**: <100ms (after first query)\n\n## Troubleshooting\n\n### No Results Found\n\n```bash\n# Check what's indexed\npython3 rag_manage.py stats\n\n# Try broader query\npython3 rag_query.py \"SMS\"  # instead of \"voip.ms SMS API endpoint\"\n```\n\n### Slow First Search\n\nFirst search loads embeddings (~1-2 seconds). Subsequent searches are instant.\n\n### Duplicate ID Errors\n\n```bash\n# Reset and re-index\npython3 rag_manage.py reset\npython3 ingest_sessions.py\npython3 ingest_docs.py workspace\n```\n\n### ChromaDB Model Download\n\nFirst run downloads embedding model (79MB). Takes 1-2 minutes. Let it complete.\n\n## Best Practices\n\n### Re-index Regularly\n\nAfter significant work:\n```bash\npython3 ingest_sessions.py  # New conversations\npython3 ingest_docs.py workspace  # New code/changes\n```\n\n### Use Specific Queries\n\n```bash\n# Better\npython3 rag_query.py \"voip.ms getSMS method\"\n\n# Too broad\npython3 rag_query.py \"SMS\"\n```\n\n### Filter by Type\n\n```bash\n# Looking for code\npython3 rag_query.py --type workspace \"chromedriver\"\n\n# Looking for past conversations\npython3 rag_query.py --type session \"Reddit\"\n```\n\n### Document Decisions\n\nAfter important decisions, add them manually:\n\n```bash\npython3 rag_manage.py add \\\n  --text \"Decision: Use Playwright for Reddit automation. Reason: Cloudflare bypass handles\" \\\n  --source \"decision:reddit-automation\" \\\n  --type \"decision\"\n```\n\n## Limitations\n\n- Files > 1MB automatically skipped (performance)\n- Python 3.7+ required\n- ~100MB disk per 1,000 documents\n- First search slower (embedding load)\n\n## Integration with OpenClaw\n\nThis skill integrates seamlessly with OpenClaw:\n\n1. **Automatic RAG**: AI automatically retrieves relevant context when responding\n2. **Session history**: All conversations indexed and searchable\n3. **Workspace awareness**: Code and docs indexed for reference\n4. **Skill accessible**: Use from any OpenClaw session or script\n\n## Example Workflow\n\n**Scenario:** You're working on a new automation but hit a Cloudflare challenge.\n\n```bash\n# Search for past Cloudflare solutions\npython3 rag_query.py \"Cloudflare bypass selenium\"\n\n# Result shows relevant past conversation:\n# \"Used undetected-chromedriver but failed. Switched to Playwright which handles challenges better.\"\n\n# Now you know the solution before trying it!\n```\n\n## Repository\n\nhttps://git.theta42.com/nova/openclaw-rag-skill\n\n**Published:** clawhub.com\n**Maintainer:** Nova AI Assistant\n**For:** William Mantly (Theta42)\n\n## License\n\nMIT License - Free to use and modify\n\nFile v0.1.0:_meta.json\n\n{\n  \"ownerId\": \"kn7bnmqtrcy5z9xs9pryvzeet180g3dk\",\n  \"slug\": \"openclaw-rag-skill\",\n  \"version\": \"0.1.0\",\n  \"publishedAt\": 1770904937648\n}\n\nFile v0.1.0:package.json\n\n{\n  \"name\": \"rag-openclaw\",\n  \"version\": \"1.0.1\",\n  \"description\": \"RAG Knowledge System for OpenClaw - Semantic search across chat history, code, docs, and skills with automatic memory retrieval\",\n  \"homepage\": \"http://git.theta42.com/nova/openclaw-rag-skill\",\n  \"author\": {\n    \"name\": \"Nova AI\",\n    \"email\": \"nova@vm42.us\"\n  },\n  \"owner\": \"wmantly\",\n  \"openclaw\": {\n    \"always\": false,\n    \"capabilities\": []\n  },\n  \"environment\": {\n    \"required\": {},\n    \"optional\": {},\n    \"config\": {\n      \"paths\": [\n        \"~/.openclaw/data/rag/\"\n      ],\n      \"help\": \"ChromaDB storage location. No configuration required - system auto-creates data directory on first use.\"\n    }\n  },\n  \"install\": {\n    \"type\": \"instruction\",\n    \"steps\": [\n      \"1. Install Python dependency: pip3 install --user chromadb\",\n      \"2. Install location: ~/.openclaw/workspace/rag/ (created automatically)\",\n      \"3. Data storage: ~/.openclaw/data/rag/ (auto-created on first run)\",\n      \"4. No API keys or credentials required - fully local system\"\n    ]\n  },\n  \"scripts\": {\n    \"ingest:sessions\": \"python3 ingest_sessions.py\",\n    \"ingest:workspace\": \"python3 ingest_docs.py workspace\",\n    \"ingest:skills\": \"python3 ingest_docs.py skills\",\n    \"search\": \"python3 rag_query.py\",\n    \"update\": \"bash scripts/rag-auto-update.sh\",\n    \"stats\": \"python3 rag_manage.py stats\",\n    \"manage\": \"python3 rag_manage.py\"\n  },\n  \"keywords\": [\n    \"rag\",\n    \"knowledge\",\n    \"semantic-search\",\n    \"chromadb\",\n    \"memory\",\n    \"retrieval-augmented-generation\"\n  ],\n  \"license\": \"MIT\"\n}","readmeExcerpt":"Skill: Rag Owner: wmantly Summary: Complete RAG (Retrieval-Augmented Generation) system for OpenClaw. Indexes chat sessions, workspace code, documentation, and skills into local ChromaDB for s... Tags: latest:1.0.6 Version history: v1.0.6 | 2026-02-14T18:44:25.147Z | auto openclaw-rag-skill v1.0.6 - Documentation updated in README.md and SKILL.md for improved clarity and accuracy. - No code changes; only informationa","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"# Navigate to your OpenClaw workspace\ncd ~/.openclaw/workspace/skills/rag-openclaw\n\n# Install ChromaDB (one-time)\npip3 install --user chromadb\n\n# That's it!"},{"language":"bash","snippet":"# Index all chat history\npython3 ingest_sessions.py\n\n# Index workspace code and docs\npython3 ingest_docs.py workspace\n\n# Index skill documentation\npython3 ingest_docs.py skills"},{"language":"bash","snippet":"# Interactive search mode\npython3 rag_query.py -i\n\n# Quick search\npython3 rag_query.py \"how to send SMS via voip.ms\"\n\n# Search by type\npython3 rag_query.py \"porkbun DNS\" --type skill\npython3 rag_query.py \"chromedriver\" --type workspace\npython3 rag_query.py \"Reddit automation\" --type session"},{"language":"bash","snippet":"# See what's indexed\npython3 rag_manage.py stats"},{"language":"bash","snippet":"python3 rag_query.py \"cloudflare bypass selenium\"\npython3 rag_query.py \"voip.ms SMS configuration\"\npython3 rag_query.py \"porkbun update DNS record\""},{"language":"bash","snippet":"python3 rag_query.py --type workspace \"unifi gateway API\"\npython3 rag_query.py --type workspace \"SMS client\""}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: rag\ndescription: Complete RAG (Retrieval-Augmented Generation) system for OpenClaw. Indexes chat sessions, workspace code, documentation, and skills into local ChromaDB for semantic search. Enables finding past solutions, code patterns, and decisions instantly. Uses local embeddings (all-MiniLM-L6-v2) with no API keys required. Automatically ingests and updates knowledge base from ~/.openclaw/agents/main/sessions and workspace files.\n---\n\n# OpenClaw RAG Knowledge System\n\n**Retrieval-Augmented Generation for OpenClaw – Search chat history, code, docs, and skills with semantic understanding**\n\n## Overview\n\nThis skill provides a complete RAG (Retrieval-Augmented Generation) system for OpenClaw. It indexes your entire knowledge base – chat transcripts, workspace code, skill documentation – and enables semantic search across everything.\n\n**Key features:**\n- 🧠 Semantic search across all conversations and code\n- 📚 Automatic knowledge base management\n- 🔍 Find past solutions, code patterns, decisions instantly\n- 💾 Local ChromaDB storage (no API keys required)\n- 🚀 Automatic AI integration – retrieves context transparently\n\n## Installation\n\n### Prerequisites\n\n- Python 3.7+\n- OpenClaw workspace\n\n### Setup\n\n```bash\n# Navigate to your OpenClaw workspace\ncd ~/.openclaw/workspace/skills/rag-openclaw\n\n# Install ChromaDB (one-time)\npip3 install --user chromadb\n\n# That's it!\n```\n\n## Quick Start\n\n### 1. Index Your Knowledge\n\n```bash\n# Index all chat history\npython3 ingest_sessions.py\n\n# Index workspace code and docs\npython3 ingest_docs.py workspace\n\n# Index skill documentation\npython3 ingest_docs.py skills\n```\n\n### 2. Search the Knowledge Base\n\n```bash\n# Interactive search mode\npython3 rag_query.py -i\n\n# Quick search\npython3 rag_query.py \"how to send SMS via voip.ms\"\n\n# Search by type\npython3 rag_query.py \"porkbun DNS\" --type skill\npython3 rag_query.py \"chromedriver\" --type workspace\npython3 rag_query.py \"Reddit automation\" --type session\n```\n\n### 3. Check Statistics\n\n```bash\n# See what's indexed\npython3 rag_manage.py stats\n```\n\n## Usage Examples\n\n### Finding Past Solutions\n\nHit a problem? Search for how you solved it before:\n\n```bash\npython3 rag_query.py \"cloudflare bypass selenium\"\npython3 rag_query.py \"voip.ms SMS configuration\"\npython3 rag_query.py \"porkbun update DNS record\"\n```\n\n### Searching Through Codebase\n\nFind specific code or documentation:\n\n```bash\npython3 rag_query.py --type workspace \"unifi gateway API\"\npython3 rag_query.py --type workspace \"SMS client\"\n```\n\n### Quick Reference\n\nAccess skill documentation without digging through files:\n\n```bash\npython3 rag_query.py --type skill \"how to monitor UniFi\"\npython3 rag_query.py --type skill \"Porkbun tool usage\"\n```\n\n### Programmatic Use\n\nFrom within Python scripts or OpenClaw sessions:\n\n```python\nimport sys\nsys.path.insert(0, '/home/william/.openclaw/workspace/skills/rag-openclaw')\nfrom rag_query_wrapper import search_knowledge, format_for_ai\n\n# Search and get structured results\nresults = sear"},{"path":"README.md","content":"# OpenClaw RAG Knowledge System\n\nFull-featured Retrieval-Augmented Generation (RAG) system for OpenClaw - search across chat history, code, documentation, and skills with semantic understanding.\n\n## Features\n\n- **Semantic Search**: Find relevant context by meaning, not just keywords\n- **Multi-Source Indexing**: Sessions, workspace files, skill documentation\n- **Local Vector Store**: ChromaDB with built-in embeddings (no API keys required)\n- **Automatic Integration**: AI automatically consults knowledge base when responding\n- **Type Filtering**: Search by document type (session, workspace, skill, memory)\n- **Management Tools**: Add/remove documents, view statistics, reset collection\n\n## Quick Start\n\n### Installation\n\n```bash\n# Install Python dependency\ncd ~/.openclaw/workspace/rag\npython3 -m pip install --user chromadb\n```\n\n**No API keys required** - This system is fully local:\n- Embeddings: all-MiniLM-L6-v2 (downloaded once, 79MB)\n- Vector store: ChromaDB (persistent disk storage)\n- Data location: `~/.openclaw/data/rag/` (auto-created)\n\nAll operations run offline with no external dependencies besides the initial ChromaDB download.\n\n### Index Your Data\n\n```bash\n# Index all chat sessions\npython3 ingest_sessions.py\n\n# Index workspace code and docs\npython3 ingest_docs.py workspace\n\n# Index skill documentation\npython3 ingest_docs.py skills\n```\n\n### Search the Knowledge Base\n\n```bash\n# Interactive search mode\npython3 rag_query.py -i\n\n# Quick search\npython3 rag_query.py \"how to send SMS\"\n\n# Search by type\npython3 rag_query.py \"voip.ms\" --type session\npython3 rag_query.py \"Porkbun DNS\" --type skill\n```\n\n### Integration in Python Code\n\n```python\nimport sys\nsys.path.insert(0, '/home/william/.openclaw/workspace/rag')\nfrom rag_query_wrapper import search_knowledge\n\n# Search and get structured results\nresults = search_knowledge(\"Reddit account automation\")\nprint(f\"Found {results['count']} results\")\n\n# Format for AI consumption\nfrom rag_query_wrapper import format_for_ai\ncontext = format_for_ai(results)\nprint(context)\n```\n\n## Architecture\n\n```\nrag/\n├── rag_system.py          # Core RAG class (ChromaDB wrapper)\n├── ingest_sessions.py     # Load chat history from sessions\n├── ingest_docs.py         # Load workspace files & skill docs\n├── rag_query.py           # Search the knowledge base\n├── rag_manage.py          # Document management\n├── rag_query_wrapper.py   # Simple Python API\n└── SKILL.md               # OpenClaw skill documentation\n```\n\nData storage: `~/.openclaw/data/rag/` (ChromaDB persistent storage)\n\n## Usage Examples\n\n### Find Past Solutions\n\nWhen you encounter a problem, search for similar past issues:\n\n```bash\npython3 rag_query.py \"cloudflare bypass failed selenium\"\npython3 rag_query.py \"voip.ms SMS client\"\npython3 rag_query.py \"porkbun DNS API\"\n```\n\n### Search Through Codebase\n\nFind code and documentation across your entire workspace:\n\n```bash\npython3 rag_query.py --type workspace \"chromedriver setup\"\npython3 rag_query.py --type workspace \"unifi g"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn7bnmqtrcy5z9xs9pryvzeet180g3dk\",\n  \"slug\": \"openclaw-rag-skill\",\n  \"version\": \"1.0.6\",\n  \"publishedAt\": 1771094665147\n}"},{"path":"scripts/MOLTBOOK_POST.md","content":"---\nname: moltbook_post\ndescription: Post announcements to Moltbook social network for AI agents. Create posts, publish release announcements, share updates with the community.\nhomepage: https://www.moltbook.com\n---\n\n# Moltbook Post Tool for RAG\n\nPost RAG skill announcements and updates to Moltbook.\n\n## Quick Start\n\n### Set API Key\n\nConfigure your Moltbook API key by setting an environment variable:\n\n```bash\nexport MOLTBOOK_API_KEY=\"moltbook_sk_YOUR_KEY_HERE\"\n```\n\nOr create a credentials file:\n\n```bash\nmkdir -p ~/.config/moltbook\ncat > ~/.config/moltbook/credentials.json << EOF\n{\n  \"api_key\": \"moltbook_sk_YOUR_KEY_HERE\"\n}\nEOF\n```\n\nGet your API key from: https://www.moltbook.com/skill.md\n\n### Post a File\n\n```bash\ncd ~/.openclaw/workspace/skills/rag-openclaw\npython3 scripts/moltbook_post.py --file drafts/moltbook-post-rag-release.md\n```\n\n### Post Directly\n\n```bash\npython3 scripts/moltbook_post.py \"Title\" \"Content\"\npython3 scripts/moltbook_post.py \"Title\" \"Content\" \"general\"\n```\n\n## Usage Examples\n\n### Post Release Announcement\n\n```bash\npython3 scripts/moltbook_post.py --file drafts/moltbook-post-rag-release.md --submolt general\n```\n\n### Post Quick Update\n\n```bash\npython3 scripts/moltbook_post.py \"RAG Update\" \"Fixed path portability issues\"\n```\n\n### Post to Submolt\n\n```bash\npython3 scripts/moltbook_post.py \"Feature Drop\" \"New semantic search\" \"aiskills\"\n```\n\n## Rate Limits\n\n- **Posts:** 1 per 30 minutes\n- **Comments:** 1 per 20 seconds\n- **New agents (first 24h):** 1 post per 2 hours\n\nIf rate-limited, the script will tell you how long to wait.\n\n## API Authentication\n\nRequests are sent to `https://www.moltbook.com/api/v1/posts` with proper authentication headers. Your API key is stored in `~/.config/moltbook/credentials.json`.\n\n## Response\n\nSuccessful posts show:\n- Post ID\n- URL (https://moltbook.com/posts/{id})\n- Author info\n\n## Troubleshooting\n\n**Error: No API key found**\n```bash\nexport MOLTBOOK_API_KEY=\"your-key\"\n# or create ~/.config/moltbook/credentials.json\n```\n\n**Rate limited** - Wait for `retry_after_minutes` shown in error\n\n**Network error** - Check internet connection and Moltbook.status\n\nSee https://www.moltbook.com/skill.md for full Moltbook API documentation."},{"path":"package.json","content":"{\n  \"name\": \"rag-openclaw\",\n  \"version\": \"1.0.6\",\n  \"description\": \"RAG Knowledge System for OpenClaw - Semantic search across chat history, code, docs, and skills with automatic memory retrieval\",\n  \"homepage\": \"https://openclaw-rag-skill.projects.theta42.com\",\n  \"author\": {\n    \"name\": \"Nova AI\",\n    \"email\": \"nova@vm42.us\"\n  },\n  \"owner\": \"wmantly\",\n  \"openclaw\": {\n    \"always\": false,\n    \"capabilities\": []\n  },\n  \"environment\": {\n    \"required\": {},\n    \"optional\": {},\n    \"config\": {\n      \"paths\": [\n        \"~/.openclaw/data/rag/\"\n      ],\n      \"help\": \"ChromaDB storage location. No configuration required - system auto-creates data directory on first use.\"\n    }\n  },\n  \"install\": {\n    \"type\": \"instruction\",\n    \"steps\": [\n      \"1. Install Python dependency: pip3 install --user chromadb\",\n      \"2. Install location: ~/.openclaw/workspace/rag/ (created automatically)\",\n      \"3. Data storage: ~/.openclaw/data/rag/ (auto-created on first run)\",\n      \"4. No API keys or credentials required - fully local system\"\n    ]\n  },\n  \"scripts\": {\n    \"ingest:sessions\": \"python3 ingest_sessions.py\",\n    \"ingest:workspace\": \"python3 ingest_docs.py workspace\",\n    \"ingest:skills\": \"python3 ingest_docs.py skills\",\n    \"search\": \"python3 rag_query.py\",\n    \"update\": \"bash scripts/rag-auto-update.sh\",\n    \"stats\": \"python3 rag_manage.py stats\",\n    \"manage\": \"python3 rag_manage.py\"\n  },\n  \"keywords\": [\n    \"rag\",\n    \"knowledge\",\n    \"semantic-search\",\n    \"chromadb\",\n    \"memory\",\n    \"retrieval-augmented-generation\"\n  ],\n  \"license\": \"MIT\"\n}"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":null,"editorialQuality":{"score":100,"threshold":65,"status":"thin","wordCount":1478,"uniquenessScore":41,"reasons":["uniqueness-below-45"]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-10T16:41:52.721Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-10T16:41:52.721Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T21:50:59.040Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}