{"id":"6b8edbf0-d498-47f2-ba17-b29033f75ae8","entityType":"agent","slug":"spillwavesolutions-developing-llamaindex-systems","name":"developing-llamaindex-systems","canonicalUrl":"https://www.xpersona.co/agent/spillwavesolutions-developing-llamaindex-systems","canonicalPath":"/agent/spillwavesolutions-developing-llamaindex-systems","generatedAt":"2026-10-09T11:00:07.860Z","source":"GITHUB_OPENCLEW","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-04-15T05:21:22.124Z","emptyReason":null},"description":"Production-grade agentic system development with LlamaIndex in Python. Covers semantic ingestion (SemanticSplitterNodeParser, CodeSplitter, IngestionPipeline), retrieval strategies (BM25Retriever, hybrid search, alpha weighting), PropertyGraphIndex with graph stores (Neo4j), context RAG (RouterQueryEngine, SubQuestionQueryEngine, LLMRerank), agentic orchestration (ReAct, Workflows, FunctionTool), and observability (Arize Phoenix). Use when asked to \"build a LlamaIndex agent\", \"set up semantic chunking\", \"index source code\", \"implement hybrid search\", \"create a knowledge graph with LlamaIndex\", \"implement query routing\", \"debug RAG pipeline\", \"add Phoenix observability\", or \"create an event-driven workflow\". Triggers on \"PropertyGraphIndex\", \"SemanticSplitterNodeParser\", \"CodeSplitter\", \"BM25Retriever\", \"hybrid search\", \"ReAct agent\", \"Workflow pattern\", \"LLMRerank\", \"Text-to-Cypher\". --- name: developing-llamaindex-systems description: >- Production-grade agentic system development with LlamaIndex in Python. Covers semantic ingestion (SemanticSplitterNodeParser, CodeSplitter, IngestionPipeline), retrieval strategies (BM25Retriever, hybrid search, alpha weighting), PropertyGraphIndex with graph stores (Neo4j), context RAG (RouterQueryEngine, SubQuestionQueryEngine, LLMRerank), agentic orchestratio","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 3 GitHub stars reported by the source. Last updated 4/15/2026.","installCommand":"git clone https://github.com/SpillwaveSolutions/developing-llamaindex-systems.git","sourceUrl":"https://github.com/SpillwaveSolutions/developing-llamaindex-systems","homepage":null,"primaryLinks":[{"label":"View Source","url":"https://github.com/SpillwaveSolutions/developing-llamaindex-systems","kind":"source"}],"safetyScore":80,"overallRank":26.4,"popularityScore":15,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Production-grade agentic system development with LlamaIndex in Python. Covers semantic ingestion (SemanticSplitterNodeParser, CodeSplitter, IngestionPipeline), "},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-04-15T05:21:22.124Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"GITHUB OPENCLEW","verified":false,"confidence":"medium","updatedAt":"2026-04-15T05:21:22.124Z","emptyReason":null},"stars":3,"forks":1,"downloads":null,"packageName":null,"latestVersion":null,"tractionLabel":"3 GitHub stars"},"release":{"evidence":{"source":"agent-index","verified":false,"confidence":"medium","updatedAt":"2026-04-15T04:13:12.165Z","emptyReason":null},"lastUpdatedAt":"2026-04-15T05:21:22.124Z","lastCrawledAt":"2026-04-15T04:13:12.165Z","lastIndexedAt":null,"nextCrawlAt":"2026-04-16T04:13:12.165Z","lastVerifiedAt":null,"highlights":[]},"execution":{"evidence":{"source":"GITHUB OPENCLEW","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"git clone https://github.com/SpillwaveSolutions/developing-llamaindex-systems.git","setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/spillwavesolutions-developing-llamaindex-systems/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/spillwavesolutions-developing-llamaindex-systems/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/spillwavesolutions-developing-llamaindex-systems/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/spillwavesolutions-developing-llamaindex-systems/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/spillwavesolutions-developing-llamaindex-systems/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/spillwavesolutions-developing-llamaindex-systems/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"GITHUB_OPENCLEW","generatedAt":"2026-10-09T11:00:07.860Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/spillwavesolutions-developing-llamaindex-systems/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/spillwavesolutions-developing-llamaindex-systems/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/spillwavesolutions-developing-llamaindex-systems/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/spillwavesolutions-developing-llamaindex-systems/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"GITHUB OPENCLEW","verified":false,"confidence":"high","updatedAt":"2026-04-15T05:21:22.124Z","emptyReason":null},"readme":"---\nname: developing-llamaindex-systems\ndescription: >-\n  Production-grade agentic system development with LlamaIndex in Python.\n  Covers semantic ingestion (SemanticSplitterNodeParser, CodeSplitter, IngestionPipeline),\n  retrieval strategies (BM25Retriever, hybrid search, alpha weighting),\n  PropertyGraphIndex with graph stores (Neo4j), context RAG (RouterQueryEngine,\n  SubQuestionQueryEngine, LLMRerank), agentic orchestration (ReAct, Workflows,\n  FunctionTool), and observability (Arize Phoenix). Use when asked to\n  \"build a LlamaIndex agent\", \"set up semantic chunking\", \"index source code\",\n  \"implement hybrid search\", \"create a knowledge graph with LlamaIndex\",\n  \"implement query routing\", \"debug RAG pipeline\", \"add Phoenix observability\",\n  or \"create an event-driven workflow\". Triggers on \"PropertyGraphIndex\",\n  \"SemanticSplitterNodeParser\", \"CodeSplitter\", \"BM25Retriever\", \"hybrid search\",\n  \"ReAct agent\", \"Workflow pattern\", \"LLMRerank\", \"Text-to-Cypher\".\nallowed-tools:\n  - Read\n  - Write\n  - Bash\n  - WebFetch\n  - Grep\n  - Glob\nmetadata:\n  version: 1.2.0\n  last-updated: 2025-12-28\n  category: frameworks\n  python-version: \">=3.9\"\n---\n\n# LlamaIndex Agentic Systems\n\nBuild production-grade agentic RAG systems with semantic ingestion, knowledge graphs, dynamic routing, and observability.\n\n## Quick Start\n\nBuild a working agent in 6 steps:\n\n### Step 1: Install Dependencies\n\n```bash\npip install llama-index-core>=0.10.0 llama-index-llms-openai llama-index-embeddings-openai arize-phoenix\n```\n\nSee [scripts/requirements.txt](scripts/requirements.txt) for full pinned dependencies.\n\n### Step 2: Ingest with Semantic Chunking\n\n```python\nfrom llama_index.core import SimpleDirectoryReader\nfrom llama_index.core.node_parser import SemanticSplitterNodeParser\nfrom llama_index.embeddings.openai import OpenAIEmbedding\n\nembed_model = OpenAIEmbedding(model_name=\"text-embedding-3-small\")\nsplitter = SemanticSplitterNodeParser(\n    buffer_size=1,\n    breakpoint_percentile_threshold=95,\n    embed_model=embed_model\n)\n\ndocs = SimpleDirectoryReader(input_files=[\"data.pdf\"]).load_data()\nnodes = splitter.get_nodes_from_documents(docs)\n```\n\n### Step 3: Build Index\n\n```python\nfrom llama_index.core import VectorStoreIndex\n\nindex = VectorStoreIndex(nodes, embed_model=embed_model)\nindex.storage_context.persist(persist_dir=\"./storage\")\n```\n\n### Step 4: Verify Index\n\n```python\n# Confirm index built correctly\nprint(f\"Indexed {len(index.docstore.docs)} document chunks\")\n\n# Preview a sample node\nsample = list(index.docstore.docs.values())[0]\nprint(f\"Sample chunk: {sample.text[:200]}...\")\n```\n\n### Step 5: Create Query Engine\n\n```python\nquery_engine = index.as_query_engine(similarity_top_k=5)\nresponse = query_engine.query(\"What are the key concepts?\")\nprint(response)\n```\n\n### Step 6: Enable Observability\n\n```python\nimport phoenix as px\nimport llama_index.core\n\npx.launch_app()\nllama_index.core.set_global_handler(\"arize_phoenix\")\n# All subsequent queries are now traced\n```\n\nFor production script, run: `python scripts/ingest_semantic.py`\n\n---\n\n## Architecture Overview\n\nSix pillars for agentic systems:\n\n| Pillar | Purpose | Reference |\n|--------|---------|-----------|\n| **Ingestion** | Semantic chunking, code splitting, metadata | [references/ingestion.md](references/ingestion.md) |\n| **Retrieval** | BM25 keyword search, hybrid fusion | [references/retrieval-strategies.md](references/retrieval-strategies.md) |\n| **Property Graphs** | Knowledge graphs + vector hybrid | [references/property-graphs.md](references/property-graphs.md) |\n| **Context RAG** | Query routing, decomposition, reranking | [references/context-rag.md](references/context-rag.md) |\n| **Orchestration** | ReAct agents, event-driven Workflows | [references/orchestration.md](references/orchestration.md) |\n| **Observability** | Tracing, debugging, evaluation | [references/observability.md](references/observability.md) |\n\n---\n\n## Decision Trees\n\n### Which Node Parser?\n\n```\nIs the content source code?\n├─ Yes → CodeSplitter\n│        language=\"python\" (or typescript, javascript, java, go)\n│        chunk_lines=40, chunk_lines_overlap=15\n│        → See: references/ingestion.md#codesplitter\n│\n└─ No, it's documents:\n    ├─ Need semantic coherence (legal, technical docs)?\n    │   └─ Yes → SemanticSplitterNodeParser\n    │            buffer_size=1 (sensitive), 3 (stable)\n    │            breakpoint_percentile_threshold=95 (fewer), 70 (more)\n    │            → See: references/ingestion.md#semanticsplitternodeparser\n    │\n    ├─ Prioritize speed → SentenceSplitter\n    │        chunk_size=1024, chunk_overlap=20\n    │        → See: references/ingestion.md#sentencesplitter\n    │\n    └─ Need fine-grained retrieval → SentenceWindowNodeParser\n             window_size=3 (surrounding sentences in metadata)\n             → See: references/ingestion.md#sentencewindownodeparser\n```\n\n**Trade-off:** Semantic chunking requires embedding calls during ingestion (cost + latency).\n\n### Which Retrieval Mode?\n\n```\nQuery contains exact terms (function names, error codes, IDs)?\n├─ Yes, exact match critical → BM25\n│        retriever = BM25Retriever.from_defaults(nodes=nodes)\n│        → See: references/retrieval-strategies.md#bm25retriever\n│\n├─ Conceptual/semantic query → Vector\n│        retriever = index.as_retriever(similarity_top_k=5)\n│        → See: references/context-rag.md\n│\n└─ Mixed or unknown query type → Hybrid (recommended default)\n         alpha=0.5 (equal weight), 0.3 (favor BM25), 0.7 (favor vector)\n         → See: references/retrieval-strategies.md#hybrid-search\n```\n\n**Trade-off:** Hybrid adds BM25 index overhead but provides most robust retrieval.\n\n### Which Graph Extractor?\n\n```\nNeed document navigation only (prev/next/parent)?\n├─ Yes → ImplicitPathExtractor (no LLM, zero cost)\n│        → See: references/property-graphs.md#implicitpathextractor\n│\n└─ No, need semantic relationships:\n    ├─ Fixed ontology required (regulated domain)?\n    │   └─ Yes → SchemaLLMPathExtractor\n    │            Pass schema: {\"PERSON\": [\"WORKS_AT\"], \"COMPANY\": [\"LOCATED_IN\"]}\n    │            → See: references/property-graphs.md#schemallmpathextractor\n    │\n    └─ No, discovery/exploration:\n        └─ SimpleLLMPathExtractor\n           max_paths_per_chunk=10 (control noise)\n           → See: references/property-graphs.md#simplellmpathextractor\n```\n\n### Which Graph Retriever?\n\n```\nNeed SQL-like aggregations (COUNT, SUM)?\n├─ Yes, trusted environment → TextToCypherRetriever\n│        Risk: LLM syntax errors, injection\n│        → See: references/property-graphs.md#texttocypherretriever\n│\n├─ Yes, need safety → CypherTemplateRetriever\n│        Pre-define: MATCH (p:Person {name: $name}) RETURN p\n│        LLM only extracts parameters\n│        → See: references/property-graphs.md#cyphertemplateretriever\n│\n└─ No, robustness priority → VectorContextRetriever\n         Vector search → graph traversal (path_depth=2)\n         Most reliable, no code generation\n         → See: references/property-graphs.md#vectorcontextretriever\n```\n\n### Which Agent Pattern?\n\n```\nSimple tool loop sufficient?\n├─ Yes → ReAct Agent (FunctionCallingAgent)\n│        Tools via FunctionTool or ToolSpec\n│        → See: references/orchestration.md#react-agent-pattern\n│\n└─ No, need:\n    ├─ Branching/cycles → Workflow\n    │   → See: references/orchestration.md#branching\n    ├─ Human-in-the-loop → Workflow (suspend/resume)\n    │   → See: references/orchestration.md#human-in-the-loop\n    ├─ Multi-agent handoff → Workflow + Concierge pattern\n    │   → See: references/orchestration.md#concierge-multi-agent\n    └─ Parallel execution → Workflow with multiple event emissions\n        → See: references/orchestration.md#workflows\n```\n\n---\n\n## Common Patterns\n\n### Pattern 1: Metadata-Enriched Ingestion\n\n```python\nfrom llama_index.core.extractors import TitleExtractor, SummaryExtractor, KeywordExtractor\nfrom llama_index.core.ingestion import IngestionPipeline\n\npipeline = IngestionPipeline(\n    transformations=[\n        splitter,\n        TitleExtractor(),\n        SummaryExtractor(),\n        KeywordExtractor(keywords=5),\n        embed_model,\n    ]\n)\nnodes = pipeline.run(documents=docs)\n```\n\n### Pattern 2: PropertyGraphIndex with Hybrid Retrieval\n\n```python\nfrom llama_index.core import PropertyGraphIndex\nfrom llama_index.core.indices.property_graph import SimpleLLMPathExtractor\n\nindex = PropertyGraphIndex.from_documents(\n    docs,\n    embed_model=embed_model,\n    kg_extractors=[SimpleLLMPathExtractor(max_paths_per_chunk=10)],\n)\n\n# Hybrid: vector search + graph traversal\nretriever = index.as_retriever(include_text=True)\n```\n\n### Pattern 3: Router with Multiple Engines\n\n```python\nfrom llama_index.core.query_engine import RouterQueryEngine\nfrom llama_index.core.selectors import LLMSingleSelector\nfrom llama_index.core.tools import QueryEngineTool\n\ntools = [\n    QueryEngineTool.from_defaults(\n        query_engine=summary_engine,\n        description=\"High-level summaries and overviews\"\n    ),\n    QueryEngineTool.from_defaults(\n        query_engine=detail_engine,\n        description=\"Specific facts, numbers, and details\"\n    ),\n]\n\nrouter = RouterQueryEngine(\n    selector=LLMSingleSelector.from_defaults(),\n    query_engine_tools=tools,\n)\n```\n\n### Pattern 4: Event-Driven Workflow\n\n```python\nfrom llama_index.core.workflow import Workflow, step, StartEvent, StopEvent, Event\n\nclass QueryEvent(Event):\n    query: str\n\nclass MyAgent(Workflow):\n    @step\n    async def classify(self, ev: StartEvent) -> QueryEvent:\n        return QueryEvent(query=ev.get(\"query\"))\n\n    @step\n    async def respond(self, ev: QueryEvent) -> StopEvent:\n        result = self.query_engine.query(ev.query)\n        return StopEvent(result=str(result))\n\n# Run\nagent = MyAgent(timeout=60)\nresult = await agent.run(query=\"What is X?\")\n```\n\n### Pattern 5: Reranking Pipeline\n\n```python\nfrom llama_index.core.postprocessor import SimilarityPostprocessor, LLMRerank\n\nquery_engine = index.as_query_engine(\n    similarity_top_k=10,  # Retrieve more\n    node_postprocessors=[\n        SimilarityPostprocessor(similarity_cutoff=0.7),\n        LLMRerank(top_n=3),  # Rerank to top 3\n    ]\n)\n```\n\n---\n\n## Script Reference\n\n| Script | Purpose | Usage |\n|--------|---------|-------|\n| `scripts/ingest_semantic.py` | Build index with semantic chunking + graph | `python scripts/ingest_semantic.py --doc path/to/file.pdf` |\n| `scripts/agent_workflow.py` | Event-driven agent template | `python scripts/agent_workflow.py` |\n| `scripts/requirements.txt` | Pinned dependencies | `pip install -r scripts/requirements.txt` |\n\nAdapt scripts by modifying configuration variables at the top of each file.\n\n---\n\n## Reference Index\n\nLoad references based on task:\n\n| Task | Load Reference |\n|------|----------------|\n| Configure chunking strategy | [references/ingestion.md](references/ingestion.md) |\n| Add metadata extractors | [references/ingestion.md](references/ingestion.md) |\n| Build knowledge graph | [references/property-graphs.md](references/property-graphs.md) |\n| Choose graph store (Neo4j, etc.) | [references/property-graphs.md](references/property-graphs.md) |\n| Implement query routing | [references/context-rag.md](references/context-rag.md) |\n| Decompose complex queries | [references/context-rag.md](references/context-rag.md) |\n| Add reranking | [references/context-rag.md](references/context-rag.md) |\n| Build ReAct agent | [references/orchestration.md](references/orchestration.md) |\n| Create Workflow | [references/orchestration.md](references/orchestration.md) |\n| Multi-agent system | [references/orchestration.md](references/orchestration.md) |\n| Setup Phoenix tracing | [references/observability.md](references/observability.md) |\n| Debug retrieval failures | [references/observability.md](references/observability.md) |\n| Evaluate agent quality | [references/observability.md](references/observability.md) |\n\n---\n\n## Troubleshooting\n\n### Agent says \"I don't know\" with relevant data\n\n**Diagnose:**\n```bash\n# Open Phoenix UI at http://localhost:6006\n# Navigate to Traces → Select query → Retrieval span → Retrieved Nodes\n```\n\n**Fix:**\n```python\n# 1. Increase retrieval candidates\nquery_engine = index.as_query_engine(similarity_top_k=10)  # was 5\n\n# 2. Add reranking to improve precision\nfrom llama_index.core.postprocessor import LLMRerank\nquery_engine = index.as_query_engine(\n    similarity_top_k=10,\n    node_postprocessors=[LLMRerank(top_n=3)]\n)\n```\n\n**Verify:** Re-run query, check Phoenix shows improved relevance scores (>0.7).\n\n### Semantic chunking too slow\n\n**Diagnose:**\n```python\n# Time the ingestion\nimport time\nstart = time.time()\nnodes = splitter.get_nodes_from_documents(docs)\nprint(f\"Chunking took {time.time() - start:.1f}s for {len(docs)} docs\")\n```\n\n**Fix:**\n```python\n# Option 1: Use local embeddings (no API calls)\nfrom llama_index.embeddings.huggingface import HuggingFaceEmbedding\nembed_model = HuggingFaceEmbedding(model_name=\"BAAI/bge-small-en-v1.5\")\n\n# Option 2: Hybrid strategy for large corpora\nbulk_nodes = SentenceSplitter().get_nodes_from_documents(bulk_docs)\ncritical_nodes = SemanticSplitterNodeParser(...).get_nodes_from_documents(critical_docs)\n```\n\n**Verify:** Re-run with `show_progress=True`, confirm <1s per document.\n\n### Graph extraction producing noise\n\n**Diagnose:**\n```python\n# Check extracted triples\nfor node in index.property_graph_store.get_triplets():\n    print(node)  # Look for irrelevant or duplicate relationships\n```\n\n**Fix:**\n```python\n# Option 1: Reduce paths per chunk\nSimpleLLMPathExtractor(max_paths_per_chunk=5)  # was 10\n\n# Option 2: Use strict schema\nSchemaLLMPathExtractor(\n    possible_entities=[\"PERSON\", \"COMPANY\"],\n    possible_relations=[\"WORKS_AT\", \"FOUNDED\"],\n    strict=True\n)\n```\n\n**Verify:** Re-index, confirm triplet count reduced and relationships are relevant.\n\n### Workflow step not triggering\n\n**Diagnose:**\n```python\n# Enable verbose mode\nagent = MyWorkflow(timeout=60, verbose=True)\nresult = await agent.run(query=\"test\")\n# Check console for: [Step Name] Received event: EventType\n```\n\n**Fix:**\n```python\n# Verify type hints match exactly\nclass MyEvent(Event):\n    query: str\n\n@step\nasync def my_step(self, ev: MyEvent) -> StopEvent:  # Type hint must be MyEvent\n    ...\n```\n\n**Verify:** Verbose output shows `[my_step] Received event: MyEvent`.\n\n### Phoenix not showing traces\n\n**Diagnose:**\n```python\nimport phoenix as px\nsession = px.launch_app()\nprint(f\"Phoenix URL: {session.url}\")  # Should print http://localhost:6006\n```\n\n**Fix:**\n```python\n# MUST call BEFORE any LlamaIndex imports/operations\nimport phoenix as px\npx.launch_app()\n\nimport llama_index.core\nllama_index.core.set_global_handler(\"arize_phoenix\")\n\n# Now import and use LlamaIndex\nfrom llama_index.core import VectorStoreIndex\n```\n\n**Verify:** Make a query, refresh Phoenix UI, trace appears within 5 seconds.\n\n---\n\n## When Not to Use This Skill\n\nThis skill is **specific to LlamaIndex in Python**. Do not use for:\n\n- **LangChain projects** — Different framework, different APIs\n- **Pure vector search without agents** — Simpler solutions exist\n- **Non-Python environments** — All examples are Python 3.9+\n- **Local-only / offline setups** — Scripts default to OpenAI APIs; modification required for local models\n- **Simple Q&A bots** — Overkill if you don't need graphs, routing, or workflows\n\n**If unsure:** Check if your use case involves semantic chunking, knowledge graphs, query routing, or multi-step agents. If yes, this skill applies.\n\n---\n\n## Glossary\n\n| Term | Definition |\n|------|------------|\n| **Node** | Chunk of text with metadata, the atomic unit of retrieval |\n| **PropertyGraphIndex** | Index combining vector embeddings with labeled property graph |\n| **Extractor** | Component that generates graph triples from text |\n| **Retriever** | Component that fetches relevant nodes/context |\n| **Postprocessor** | Filters or reranks nodes after retrieval |\n| **Workflow** | Event-driven state machine for agent orchestration |\n| **Span** | Duration-tracked operation in observability |\n","readmeExcerpt":"--- name: developing-llamaindex-systems description: >- Production-grade agentic system development with LlamaIndex in Python. Covers semantic ingestion (SemanticSplitterNodeParser, CodeSplitter, IngestionPipeline), retrieval strategies (BM25Retriever, hybrid search, alpha weighting), PropertyGraphIndex with graph stores (Neo4j), context RAG (RouterQueryEngine, SubQuestionQueryEngine, LLMRerank), agentic orchestratio","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"pip install llama-index-core>=0.10.0 llama-index-llms-openai llama-index-embeddings-openai arize-phoenix"},{"language":"python","snippet":"from llama_index.core import SimpleDirectoryReader\nfrom llama_index.core.node_parser import SemanticSplitterNodeParser\nfrom llama_index.embeddings.openai import OpenAIEmbedding\n\nembed_model = OpenAIEmbedding(model_name=\"text-embedding-3-small\")\nsplitter = SemanticSplitterNodeParser(\n    buffer_size=1,\n    breakpoint_percentile_threshold=95,\n    embed_model=embed_model\n)\n\ndocs = SimpleDirectoryReader(input_files=[\"data.pdf\"]).load_data()\nnodes = splitter.get_nodes_from_documents(docs)"},{"language":"python","snippet":"from llama_index.core import VectorStoreIndex\n\nindex = VectorStoreIndex(nodes, embed_model=embed_model)\nindex.storage_context.persist(persist_dir=\"./storage\")"},{"language":"python","snippet":"# Confirm index built correctly\nprint(f\"Indexed {len(index.docstore.docs)} document chunks\")\n\n# Preview a sample node\nsample = list(index.docstore.docs.values())[0]\nprint(f\"Sample chunk: {sample.text[:200]}...\")"},{"language":"python","snippet":"query_engine = index.as_query_engine(similarity_top_k=5)\nresponse = query_engine.query(\"What are the key concepts?\")\nprint(response)"},{"language":"python","snippet":"import phoenix as px\nimport llama_index.core\n\npx.launch_app()\nllama_index.core.set_global_handler(\"arize_phoenix\")\n# All subsequent queries are now traced"}],"parameters":{},"dependencies":[],"permissions":[],"extractedFiles":[],"languages":["typescript"],"docsSourceLabel":"GITHUB OPENCLEW","editorialOverview":"Production-grade agentic system development with LlamaIndex in Python. Covers semantic ingestion (SemanticSplitterNodeParser, CodeSplitter, IngestionPipeline), retrieval strategies (BM25Retriever, hybrid search, alpha weighting), PropertyGraphIndex with graph stores (Neo4j), context RAG (RouterQueryEngine, SubQuestionQueryEngine, LLMRerank), agentic orchestration (ReAct, Workflows, FunctionTool), and observability (Arize Phoenix). Use when asked to \"build a LlamaIndex agent\", \"set up semantic chunking\", \"index source code\", \"implement hybrid search\", \"create a knowledge graph with LlamaIndex\", \"implement query routing\", \"debug RAG pipeline\", \"add Phoenix observability\", or \"create an event-driven workflow\". Triggers on \"PropertyGraphIndex\", \"SemanticSplitterNodeParser\", \"CodeSplitter\", \"BM25Retriever\", \"hybrid search\", \"ReAct agent\", \"Workflow pattern\", \"LLMRerank\", \"Text-to-Cypher\". --- name: developing-llamaindex-systems description: >- Production-grade agentic system development with LlamaIndex in Python. Covers semantic ingestion (SemanticSplitterNodeParser, CodeSplitter, IngestionPipeline), retrieval strategies (BM25Retriever, hybrid search, alpha weighting), PropertyGraphIndex with graph stores (Neo4j), context RAG (RouterQueryEngine, SubQuestionQueryEngine, LLMRerank), agentic orchestratio","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":448,"uniquenessScore":62,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-04-15T05:21:22.124Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-04-15T05:21:22.124Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-09T11:00:07.860Z","emptyReason":null},"items":[{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-10T18:48:31.762Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/github_openclew","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}