{"id":"b5b87e45-279b-43ee-9312-854eec55ad87","entityType":"agent","slug":"clawhub-awsome-o-grafana-lens","name":"Grafana Lens","canonicalUrl":"https://www.xpersona.co/agent/clawhub-awsome-o-grafana-lens","canonicalPath":"/agent/clawhub-awsome-o-grafana-lens","generatedAt":"2026-10-10T08:16:49.611Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T05:57:39.886Z","emptyReason":null},"description":"Grafana tools for data visualization, monitoring, alerting, security, SRE investigation, and data collection pipeline management via Alloy. Use grafana_query...","descriptionLabel":"Source description","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.6K downloads reported by the source. Last updated 10/10/2026.","installCommand":"clawhub skill install s17bkftvrp76wq7nwsmqa236nh83ps7x:grafana-lens","sourceUrl":"https://clawhub.ai/awsome-o/grafana-lens","homepage":"https://clawhub.ai/awsome-o/skills/grafana-lens","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/awsome-o/grafana-lens","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/awsome-o/skills/grafana-lens","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":54,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Grafana Lens technical dossier on Xpersona with agent coverage, OPENCLEW support, and live trust metadata."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-10T05:57:39.886Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T05:57:39.886Z","emptyReason":null},"stars":null,"forks":null,"downloads":1636,"packageName":null,"latestVersion":"0.5.0","tractionLabel":"1.6K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T05:57:39.886Z","emptyReason":null},"lastUpdatedAt":"2026-10-10T05:57:39.886Z","lastCrawledAt":"2026-10-10T05:57:39.886Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-11T05:57:39.886Z","lastVerifiedAt":null,"highlights":[{"version":"0.5.0","createdAt":"2026-04-05T16:38:57.121Z","changelog":"Major update: Adds Alloy pipeline management, new recipes, and greatly expands data collection capabilities. - Introduced Alloy pipeline management with the new `alloy_pipeline` tool and support for recipes, creation, status, diagnosis, and deletion actions. - Added dozens of Alloy pipeline recipes and helpers for setting up metrics, logs, traces, exporters, and agent integrations. - Expanded documentation and quickstart references for Alloy, pipeline composition, and common data-collection use cases. - Extended the SKILL to include scenarios for managing data collection pipelines, collecting logs and metrics from multiple sources, and handling credentials securely. - Updated and refined limits, troubleshooting, and best-practices guidance in user instructions.","fileCount":145,"zipByteSize":516184},{"version":"0.4.0","createdAt":"2026-03-27T06:25:16.306Z","changelog":"grafana-lens v0.4.0 - Make it compliant to latest openclaw plugin protocol - Support multiple LGTM endpoints now","fileCount":97,"zipByteSize":438251},{"version":"0.3.0","createdAt":"2026-03-15T17:54:04.762Z","changelog":"- Adds SRE investigation support with the new grafana_investigate tool. - Description now covers SRE/root cause/triage scenarios, including \"investigate\", \"debug\", \"why is X broken\", anomaly detection. - Follows a statistics-first approach for log investigations (rate/count before raw logs). - Significant file cleanup: removed 33 files and test files, added core investigation logic and reference content. - Keeps all prior dashboard, alert, and visualization capabilities.","fileCount":57,"zipByteSize":280497},{"version":"0.2.0","createdAt":"2026-03-07T01:10:32.104Z","changelog":"This is the first version.","fileCount":87,"zipByteSize":383849}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s17bkftvrp76wq7nwsmqa236nh83ps7x:grafana-lens","setupComplexity":"low","setupSteps":["Install using `clawhub skill install s17bkftvrp76wq7nwsmqa236nh83ps7x:grafana-lens` in an isolated environment before connecting it to live workloads.","No published capability contract is available yet, so validate auth and request/response behavior manually.","Review the upstream CLAWHUB listing at https://clawhub.ai/awsome-o/grafana-lens before using production credentials."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-awsome-o-grafana-lens/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-awsome-o-grafana-lens/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-awsome-o-grafana-lens/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-awsome-o-grafana-lens/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-awsome-o-grafana-lens/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-awsome-o-grafana-lens/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-10T08:16:49.606Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-awsome-o-grafana-lens/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-awsome-o-grafana-lens/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-awsome-o-grafana-lens/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-awsome-o-grafana-lens/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T05:57:39.886Z","emptyReason":null},"readme":"Skill: Grafana Lens\n\nOwner: awsome-o\n\nSummary: Grafana tools for data visualization, monitoring, alerting, security, SRE investigation, and data collection pipeline management via Alloy. Use grafana_query...\n\nTags: latest:0.5.0\n\nVersion history:\n\nv0.5.0 | 2026-04-05T16:38:57.121Z | user\n\nMajor update: Adds Alloy pipeline management, new recipes, and greatly expands data collection capabilities.\n\n- Introduced Alloy pipeline management with the new `alloy_pipeline` tool and support for recipes, creation, status, diagnosis, and deletion actions.\n- Added dozens of Alloy pipeline recipes and helpers for setting up metrics, logs, traces, exporters, and agent integrations.\n- Expanded documentation and quickstart references for Alloy, pipeline composition, and common data-collection use cases.\n- Extended the SKILL to include scenarios for managing data collection pipelines, collecting logs and metrics from multiple sources, and handling credentials securely.\n- Updated and refined limits, troubleshooting, and best-practices guidance in user instructions.\n\nv0.4.0 | 2026-03-27T06:25:16.306Z | user\n\ngrafana-lens v0.4.0\n\n- Make it compliant to latest openclaw plugin protocol\n- Support multiple LGTM endpoints now\n\nv0.3.0 | 2026-03-15T17:54:04.762Z | user\n\n- Adds SRE investigation support with the new grafana_investigate tool.\n- Description now covers SRE/root cause/triage scenarios, including \"investigate\", \"debug\", \"why is X broken\", anomaly detection.\n- Follows a statistics-first approach for log investigations (rate/count before raw logs).\n- Significant file cleanup: removed 33 files and test files, added core investigation logic and reference content.\n- Keeps all prior dashboard, alert, and visualization capabilities.\n\nv0.2.0 | 2026-03-07T01:10:32.104Z | user\n\nThis is the first version.\n\nArchive index:\n\nArchive v0.5.0: 145 files, 516184 bytes\n\nFiles: index.ts (13936b), llms.txt (7258b), openclaw.plugin.json (9719b), package.json (2826b), README.md (47521b), references/agent-metrics.md (32885b), references/alloy-components.md (12876b), references/alloy-pipelines.md (18513b), references/dashboard-composition.md (14256b), references/external-data.md (4673b), references/sre-investigation.md (18777b), skill-card.md (3213b), SKILL.md (96509b), src/alloy/alloy-client.test.ts (6660b), src/alloy/alloy-client.ts (7299b), src/alloy/config-builder.test.ts (6217b), src/alloy/config-builder.ts (5747b), src/alloy/pipeline-helpers.ts (5725b), src/alloy/pipeline-store.test.ts (9404b), src/alloy/pipeline-store.ts (9512b), src/alloy/recipes/catalog.ts (4228b), src/alloy/recipes/infra/continuous-profiling.ts (5171b), src/alloy/recipes/infra/docker-metrics.ts (2468b), src/alloy/recipes/infra/elasticsearch-exporter.ts (2780b), src/alloy/recipes/infra/kafka-exporter.ts (2684b), src/alloy/recipes/logs/_process-builder.test.ts (11220b), src/alloy/recipes/logs/_process-builder.ts (12527b), src/alloy/recipes/logs/docker-logs.ts (3971b), src/alloy/recipes/logs/faro-frontend.ts (2478b), src/alloy/recipes/logs/file-logs.ts (2743b), src/alloy/recipes/logs/gelf-logs.ts (3518b), src/alloy/recipes/logs/journal-logs.ts (2160b), src/alloy/recipes/logs/kafka-logs.ts (3353b), src/alloy/recipes/logs/kubernetes-logs.ts (2180b), src/alloy/recipes/logs/loki-push-api.ts (2504b), src/alloy/recipes/logs/secret-filter-logs.ts (2544b), src/alloy/recipes/logs/syslog.ts (2530b), src/alloy/recipes/metrics/blackbox-exporter.ts (3873b), src/alloy/recipes/metrics/kubernetes-pods.ts (4174b), src/alloy/recipes/metrics/kubernetes-services.ts (2862b), src/alloy/recipes/metrics/memcached-exporter.ts (2723b), src/alloy/recipes/metrics/mongodb-exporter.ts (2723b), src/alloy/recipes/metrics/mysql-exporter.ts (2707b), src/alloy/recipes/metrics/node-exporter.ts (3241b), src/alloy/recipes/metrics/postgres-exporter.ts (3442b), src/alloy/recipes/metrics/redis-exporter.ts (3062b), src/alloy/recipes/metrics/scrape-endpoint.ts (5183b), src/alloy/recipes/metrics/self-monitoring.ts (2876b), src/alloy/recipes/recipes.test.ts (40598b), src/alloy/recipes/traces/_shared.ts (786b), src/alloy/recipes/traces/application-traces.ts (12527b), src/alloy/recipes/traces/otlp-receiver.ts (2527b), src/alloy/recipes/traces/service-graph.ts (4929b), src/alloy/recipes/traces/span-metrics.ts (4748b), src/alloy/recipes/types.ts (8654b), src/alloy/types.ts (5081b), src/config.test.ts (13764b), src/config.ts (12479b), src/grafana-client-registry.test.ts (4420b), src/grafana-client-registry.ts (2312b), src/grafana-client.test.ts (29539b), src/grafana-client.ts (39052b), src/metric-definitions.test.ts (9869b), src/metric-definitions.ts (15049b), src/sdk-compat.ts (5458b), src/services/alert-webhook.test.ts (8274b), src/services/alert-webhook.ts (8212b), src/services/alloy-service.ts (6580b), src/services/custom-metrics-store.test.ts (29065b), src/services/custom-metrics-store.ts (21458b), src/services/lifecycle-telemetry.test.ts (181342b), src/services/lifecycle-telemetry.ts (103146b), src/services/metrics-collector.test.ts (66879b), src/services/metrics-collector.ts (59945b), src/services/model-pricing.test.ts (2423b), src/services/model-pricing.ts (4406b), src/services/otel-logs.test.ts (3393b), src/services/otel-logs.ts (1971b), src/services/otel-metrics.test.ts (3264b), src/services/otel-metrics.ts (1890b)\n\nFile v0.5.0:SKILL.md\n\n---\nname: grafana-lens\ndescription: \"Grafana tools for data visualization, monitoring, alerting, security, SRE investigation, and data collection pipeline management via Alloy. Use grafana_query, grafana_query_logs, grafana_query_traces, grafana_create_dashboard, grafana_update_dashboard, grafana_create_alert, grafana_share_dashboard, grafana_annotate, grafana_explore_datasources, grafana_list_metrics, grafana_search, grafana_get_dashboard, grafana_check_alerts, grafana_push_metrics, grafana_explain_metric, grafana_security_check, grafana_investigate, and alloy_pipeline. Trigger when asked about metrics, dashboards, monitoring, alerts, costs, token usage, data visualization, PromQL, Prometheus, LogQL, Loki, log queries, error logs, log search, TraceQL, Tempo, traces, distributed tracing, span search, find slow traces, debug session traces, annotations, deployments, sharing charts, investigating alert notifications, pushing custom data (calendar, git, fitness, finance) to Grafana for visualization, pushing historical data, backfilling metrics, recording past data with timestamps, modifying dashboards, adding panels, removing panels, changing dashboard settings, updating dashboard time range, explain metric, metric trend, what is this metric, how has this changed, is this metric normal, why did my bill spike, cost visibility, security monitoring, security check, security audit, am I being attacked, is my agent compromised, suspicious activity, threat detection, prompt injection detection, set up security alerts, investigate, debug, triage, root cause, what's wrong, why is X broken, anomaly detection, RED method, USE method, alert fatigue, postmortem, incident summary, collect metrics from, monitor my database, monitor my app, scrape endpoint, set up log collection, collect Docker logs, tail log files, collect Kubernetes logs, receive OTLP, set up trace collection, data collection pipeline, Alloy pipeline, pipeline status, pipeline health, node exporter, system metrics, postgres exporter, mysql exporter, redis exporter, syslog, Grafana Alloy.\"\nmetadata:\n  {\n    \"openclaw\":\n      {\n        \"emoji\": \"🔭\",\n        \"requires\": { \"config\": [\"grafana.url\", \"grafana.apiKey\"] },\n      },\n  }\n---\n\n# Grafana Lens\n\nYou have full native Grafana access — query data, create dashboards, set alerts, receive alert notifications, annotate events, explore datasources, push custom data, and deliver visualizations inline. Works with ANY data in Grafana, not just agent metrics.\n\n## Musts\n\n- **Always call `grafana_explore_datasources` first** when you need a datasource UID — never guess UIDs\n- **Always call `grafana_search` before creating a dashboard** — avoid duplicates\n- **Always call `grafana_get_dashboard` before `grafana_share_dashboard`** — you need exact panel IDs\n- **Always call `grafana_get_dashboard` before `grafana_update_dashboard`** — you need panel IDs and current structure\n- **Prefer `grafana_query` for direct answers** over creating dashboards — \"what's my cost?\" needs a number, not a URL\n- **Prefer `grafana_query` over `grafana_create_dashboard` + `grafana_share_dashboard`** for simple data questions — a number is faster than a chart\n- **Use `grafana_query_logs` for log searches** — LogQL for logs, PromQL for metrics, TraceQL for traces. Never use `grafana_query` for Loki datasources\n- **Use `grafana_query_traces` for trace searches** — TraceQL for traces, PromQL for metrics, LogQL for logs. Never use `grafana_query` or `grafana_query_logs` for Tempo datasources\n- **All tools work with ANY Prometheus datasource** — not just `openclaw_lens_*` metrics\n- **When you see \"GRAFANA ALERTS\" in prompt context**, investigate immediately with `grafana_check_alerts` — use the `suggestedInvestigation` field to go directly to querying (it provides the tool, query, and datasource)\n- **Run `grafana_check_alerts` with action `setup` once** before alert notifications can reach the agent — this creates the webhook contact point\n- **Push data before querying or dashboarding it** — data is pushed via OTLP and available immediately\n- **Prefer `grafana_explain_metric` for \"what is this metric?\" questions** over manual `grafana_query` — it returns current value, trend, stats, and metadata in one call\n- **Use `queryNames` from push response for PromQL queries** — don't guess metric names (counters get `_total` suffix)\n- **Use `openclaw_ext_` prefix for custom metrics** — `grafana_push_metrics` auto-prepends it if missing\n- **Follow statistics-first discipline for log investigation** — always run count/rate LogQL before reading individual entries. Use `grafana_query_logs` with metric-over-logs queries (`count_over_time`, `rate`, `topk`) before switching to raw log entries\n- **Silence alerts during investigation** — use `grafana_check_alerts` with action `silence` to prevent repeat notifications while investigating\n- **Use `list_rules` for complete alert health** — `grafana_check_alerts` with action `list_rules` returns all rules with live eval state (normal/firing/pending/nodata/error), health, and lastEvaluation — no need to cross-reference with `list` action\n- **Use `dashboardUid` + `panelId` to re-run panel queries** — don't manually extract PromQL/LogQL from `get_dashboard` output. Both `grafana_query` and `grafana_query_logs` accept these params to auto-resolve the panel's query expression and datasource. The tool handles template variable replacement and datasource routing automatically\n- **Confirm with user before deleting dashboards or alert rules** — `grafana_update_dashboard` with operation `delete` and `grafana_check_alerts` with action `delete_rule` are permanent and cannot be undone\n- **Always use `alloy_pipeline` action `recipes` first** when unsure which pipeline recipe fits the user's request — because recipes provide validation, credential handling, and sample queries that raw config does not\n- **Always call `alloy_pipeline` action `status`** after creating a pipeline — because data takes 15-20s to flow through the pipeline, and components may fail silently after reload\n- **Never guess Alloy component names** — use recipes for known patterns, or raw `config` only when the user explicitly provides Alloy syntax\n- **Prefer recipes over raw config** when a recipe exists — recipes provide validation, sample queries, credential handling, dashboard templates, and automatic export target wiring\n- **Never write credentials into raw `config`** — when the user provides a connection string, DSN, password, or API key, ALWAYS use the matching recipe (which routes credentials through `sys.env()`, keeping secrets off disk). If you must use raw config, wrap sensitive values in `sys.env(\"MY_VAR_NAME\")` and tell the user to set that env var where Alloy runs\n- **Read `envVarsRequired` from every pipeline create response** — credential recipes may return `pending_credentials` status when env vars aren't set yet. Tell the user the exact var names and that they must set them where Alloy runs, then verify with action `status`\n- **Warn users before creating credential-required pipelines** — Alloy config reload is atomic: if a credential recipe's env vars aren't set, the reload failure blocks ALL managed pipelines (not just the new one) until the env vars are set or the pipeline is deleted. Always ask: \"Do you have the credentials ready to set as env vars on the Alloy host?\"\n- **Chain pipeline creation into existing tools** — after pipeline is active: `grafana_list_metrics` or `grafana_query_logs` to discover data, `grafana_create_dashboard` to visualize, `grafana_create_alert` to monitor\n- **Use `alloy_pipeline` action `diagnose`** as first step when user reports pipeline issues — because it checks Alloy connectivity, all pipeline health, config file drift, and limits in one call\n- **Confirm with user before deleting pipelines** — `alloy_pipeline` with action `delete` removes the config and data stops flowing\n- **All log recipes accept processing params** — don't create separate \"processing\" pipelines. Add `jsonExpressions`, `labelFields`, `structuredMetadata`, `tenantValue`, `matchRoutes`, etc. directly to any log recipe (docker-logs, file-logs, syslog, etc.)\n- **Use `samplingPolicies` for multi-policy tail sampling** — don't create raw config when `application-traces` can handle it. `sampleRate` is for simple probabilistic, `samplingPolicies` is for intelligent multi-policy (keep errors, keep slow, sample rest)\n- **Use log processing params for multi-tenant routing** — `tenantValue`/`tenantSource`/`matchRoutes` work on ALL log recipes. Don't create separate \"routing\" pipelines\n- **Read `references/alloy-components.md` before composing raw config** — it has copy-pasteable snippets for all common Alloy components\n\n## Quick Decision Tree\n\n- \"What is [metric]?\" / \"Why did it spike?\" → `grafana_explain_metric`\n- \"What's the current value of X?\" / complex PromQL → `grafana_query`\n- \"Find error logs\" / \"Search logs for...\" → `grafana_query_logs`\n- \"Find slow traces\" / \"Show trace for session X\" / \"Debug distributed spans\" → `grafana_query_traces`\n- \"Debug this session\" / \"Why did it fail?\" / \"What went wrong?\" → `grafana_query_traces` (search error/slow) → `grafana_query_traces` (get → follow `correlationHint`) → `grafana_query_logs` → `grafana_query` → `grafana_annotate`\n- \"Show me a chart\" / \"Visualize...\" → `grafana_search` → `grafana_get_dashboard` → `grafana_share_dashboard`\n- \"Create a dashboard for...\" → `grafana_search` (check duplicates) → `grafana_create_dashboard`\n- \"Add a panel to my dashboard\" → `grafana_get_dashboard` → `grafana_update_dashboard`\n- \"Delete this dashboard\" → `grafana_update_dashboard` with operation `delete` (confirm with user first)\n- \"Alert me when...\" → `grafana_check_alerts` (setup) → `grafana_create_alert`\n- \"List my alert rules\" / \"What alerts do I have?\" → `grafana_check_alerts` with action `list_rules`\n- \"Delete alert rule X\" → `grafana_check_alerts` with action `list_rules` → `delete_rule` with `ruleUid`\n- \"Track my [custom data]\" / \"Record my [past data]\" → `grafana_push_metrics` (with optional `timestamp` for historical data, auto-registers, returns `queryNames`) → `grafana_query` with `queryNames`\n- \"What data sources do I have?\" → `grafana_explore_datasources`\n- \"What metrics are available?\" → `grafana_list_metrics`\n- \"Set up monitoring\" / \"Monitor my agent\" / \"What dashboards should I have?\" → `grafana_search` (check existing) → `grafana_create_dashboard` with `llm-command-center` → follow `suggestedNext` chain through remaining templates\n- \"GenAI observability\" / \"OTel gen_ai metrics\" / \"Standard AI monitoring\" → `grafana_create_dashboard` with `genai-observability` template\n- \"What happened in session X?\" / \"Debug this session\" → `grafana_create_dashboard` with `session-explorer` template → paste session ID\n- \"Show me LLM traces\" / \"Show agent logs\" → `grafana_create_dashboard` with `llm-command-center` template (Loki + Tempo)\n- \"How much am I spending?\" / \"Cost analysis\" → `grafana_create_dashboard` with `cost-intelligence` template\n- \"Which tools are slow?\" / \"Tool errors\" → `grafana_create_dashboard` with `tool-performance` template\n- \"Queue health\" / \"Webhook issues\" / \"Stuck sessions\" → `grafana_create_dashboard` with `sre-operations` template\n- \"System health check\" / \"Status report\" / \"Review all dashboards\" → `grafana_explore_datasources` → `grafana_check_alerts` (list + list_rules) → `grafana_search` → `grafana_get_dashboard` (audit=true for each) → summarize\n- \"Audit my dashboard\" / \"Which panels are broken?\" → `grafana_get_dashboard` (audit=true) → review `auditSummary` + per-panel `health`\n- \"Am I being attacked?\" / \"Security check\" / \"Security status\" → `grafana_security_check`\n- \"Set up security monitoring\" → `grafana_check_alerts` (setup) → `grafana_create_dashboard` (`security-overview`) → `grafana_create_alert` (webhook error burst, cost spike, tool loops, injection signals)\n- \"Investigate security alert\" → `grafana_security_check` → `grafana_query_logs` (correlate) → `grafana_annotate` (mark investigation) → `grafana_check_alerts` (silence)\n- \"Investigate this alert\" / \"Why is X broken?\" / \"Debug this issue\" / \"Triage\" / \"Root cause\" → `grafana_investigate` (multi-signal triage) → follow `suggestedHypotheses.testWith` for deep-dives\n- \"Is this metric normal?\" / \"Is there an anomaly?\" → `grafana_explain_metric` (returns `anomaly` z-score + `seasonality` vs 1d/7d ago for 24h period)\n- \"RED analysis\" / \"What's the error rate?\" / \"Service health\" → RED Method queries (see sre-investigation.md §2)\n- \"Alert fatigue\" / \"Which alerts are noisy?\" / \"Alert health\" → `grafana_check_alerts` with action `analyze` — fatigue report\n- \"Postmortem\" / \"Incident summary\" / \"What happened?\" → `grafana_investigate` → 5-Phase methodology → postmortem template (see sre-investigation.md §9)\n- \"Compare before/after deployment\" → `grafana_annotate` (list, tags: [\"deploy\"]) → `grafana_explain_metric` (compareWith: \"previous\")\n\n### Data Collection Pipelines (Alloy)\n\n- \"Monitor a service/database/app\" → `alloy_pipeline` action `recipes` (filter by category) → select recipe → `create` → `status` → query → dashboard → alert\n- \"Scrape metrics from [endpoint]\" / \"My app exposes /metrics\" → `alloy_pipeline` with recipe `scrape-endpoint` + params `{ url }`\n- \"Monitor PostgreSQL/MySQL/Redis/MongoDB/Memcached\" → `alloy_pipeline` with recipe `[db]-exporter` + params `{ connectionString }`\n- \"Collect and parse logs with JSON extraction\" → `alloy_pipeline` (log recipe + processing params: `jsonExpressions`, `labelFields`, `structuredMetadata`)\n- \"Collect Docker logs\" / \"See container logs in Grafana\" → `alloy_pipeline` with recipe `docker-logs`\n- \"Tail log files\" / \"Collect app logs from /var/log\" → `alloy_pipeline` with recipe `file-logs` + params `{ paths }`\n- \"Accept logs via HTTP push API\" / \"Centralized log gateway\" → `alloy_pipeline` with recipe `loki-push-api`\n- \"Consume logs from Kafka\" → `alloy_pipeline` with recipe `kafka-logs` + params `{ brokers, topics }`\n- \"Set up syslog collection\" → `alloy_pipeline` with recipe `syslog`\n- \"Monitor endpoint availability\" / \"Synthetic probing\" / \"HTTP health checks\" → `alloy_pipeline` with recipe `blackbox-exporter` + params `{ targets }`\n- \"Kubernetes monitoring\" / \"Monitor my K8s cluster\" → `alloy_pipeline` with recipe `kubernetes-pods` + `kubernetes-services` + `kubernetes-logs` (3 pipelines)\n- \"Receive OTLP data\" / \"Set up trace collection\" → `alloy_pipeline` with recipe `otlp-receiver`\n- \"Generate RED metrics from traces\" / \"Span metrics\" → `alloy_pipeline` with recipe `span-metrics`\n- \"Service dependency graph from traces\" → `alloy_pipeline` with recipe `service-graph`\n- \"Monitor Alloy itself\" / \"Self-monitoring\" → `alloy_pipeline` with recipe `self-monitoring`\n- \"Redact secrets from logs\" / \"Compliance logging\" → `alloy_pipeline` with recipe `secret-filter-logs` + params `{ paths }`\n- \"Monitor Elasticsearch/Kafka\" → `alloy_pipeline` with recipe `elasticsearch-exporter` / `kafka-exporter`\n- \"System metrics\" / \"Node monitoring\" / \"CPU/memory/disk\" → `alloy_pipeline` with recipe `node-exporter`\n- \"Docker container metrics\" / \"Container resource usage\" → `alloy_pipeline` with recipe `docker-metrics`\n- \"Reduce trace costs\" / \"Keep only error traces\" / \"Smart trace sampling\" / \"Tail sampling\" → `alloy_pipeline` with recipe `application-traces` + `samplingPolicies` array (keep errors, keep slow, filter health checks, sample rest)\n- \"Multi-tenant Loki\" / \"Route logs by tenant\" / \"Different tenants for different apps\" → any log recipe + `tenantValue` or `matchRoutes` processing param\n- \"Profile my app\" / \"CPU profiling\" / \"Memory profiling\" / \"Continuous profiling\" / \"Go pprof\" → `alloy_pipeline` with recipe `continuous-profiling` + `targets`\n- \"Frontend observability\" / \"Browser RUM\" / \"Web vitals\" / \"Faro SDK\" → `alloy_pipeline` with recipe `faro-frontend`\n- \"GELF logs\" / \"Graylog\" / \"Docker GELF driver\" → `alloy_pipeline` with recipe `gelf-logs`\n- \"Custom Alloy pattern\" / \"Advanced pipeline\" → Read `references/alloy-components.md` → `alloy_pipeline` with raw `config` + optional `sampleQueries`\n- \"What data collection recipes are available?\" → `alloy_pipeline` with action `recipes`\n- \"What pipelines do I have?\" / \"Pipeline list\" → `alloy_pipeline` with action `list`\n- \"Is my pipeline working?\" / \"Pipeline health\" → `alloy_pipeline` with action `status` + name\n- \"Pipeline problems\" / \"Why isn't data showing up?\" → `alloy_pipeline` with action `diagnose` → follow remediation\n- \"Delete pipeline\" / \"Remove monitoring for...\" → `alloy_pipeline` with action `delete` + name (confirm with user first)\n\n## Working with Multiple Grafana Instances\n\nWhen several Grafana environments are configured (dev, staging, prod), every tool accepts an optional `instance` parameter. `grafana_explore_datasources` returns `availableInstances` — use the `name` values from that list.\n\n**Why this matters**: Users often need to query production metrics, create dashboards in dev, or compare environments side by side. Each tool call targets one instance.\n\n**Smart defaults**: Omitting `instance` always targets the configured default — safe and invisible for single-environment setups. Only specify `instance` when the user explicitly names a non-default environment.\n\n**Cross-environment workflows**: Each call is independent. Query prod, create dashboard in dev — just set `instance` differently on each call. No context switching needed.\n\n## Tool Inventory\n\n| Tool | What It Does |\n|------|-------------|\n| `grafana_explore_datasources` | Discover configured datasources (UIDs, types, query routing) — tells you which tool + query language to use for each datasource |\n| `grafana_list_metrics` | Discover available metrics or label values from a datasource. Use `compact: true` with `metadata: true` for minimal fields in multi-tool chains |\n| `grafana_query` | Run PromQL instant/range queries — get numbers directly |\n| `grafana_query_logs` | Run LogQL queries against Loki — search and filter logs |\n| `grafana_query_traces` | Run TraceQL queries against Tempo — search traces or get full trace by ID |\n| `grafana_create_dashboard` | Create dashboards from templates or custom JSON |\n| `grafana_update_dashboard` | Add/remove/update panels, change dashboard metadata, or delete dashboard |\n| `grafana_get_dashboard` | Get dashboard summary (panels, queries). Use `compact: true` for overview scans, `audit: true` to health-check all panels in one call |\n| `grafana_search` | Search existing dashboards by title, tags, or starred status |\n| `grafana_share_dashboard` | Render panel as image and deliver inline via messaging |\n| `grafana_create_alert` | Create Grafana-native alert rules on any metric |\n| `grafana_annotate` | Create or list annotations (events) on dashboards |\n| `grafana_check_alerts` | Check, acknowledge, list/delete rules, silence/unsilence, or set up Grafana alert webhook notifications. Use `compact: true` with `list_rules` for minimal fields |\n| `grafana_push_metrics` | Push custom data (calendar, git, fitness, finance) via OTLP |\n| `grafana_explain_metric` | Get metric context: current value, trend, stats, metadata, drill-down queries — agent interprets |\n| `grafana_security_check` | Run 6 parallel security checks and return threat-level assessment (green/yellow/red) — \"Am I being attacked?\" |\n| `grafana_investigate` | Multi-signal investigation triage — gathers metrics, logs, traces, and context in parallel, generates hypotheses with specific tool+params for follow-up |\n| `alloy_pipeline` | Create and manage Alloy data collection pipelines — 29 recipes for metrics, logs, traces, profiles from any infrastructure (databases, K8s, Docker, apps, profiling, frontend RUM) |\n\n## Tool Details\n\n### `grafana_explore_datasources`\n**When**: First step when user mentions data, metrics, or monitoring. Gets datasource UIDs needed by `grafana_query`, `grafana_query_logs`, `grafana_query_traces`, `grafana_list_metrics`, `grafana_create_alert`, and `grafana_explain_metric`.\n**Params**: `instance` (optional — target Grafana instance, omit for default).\n**Example**: `{}`\n**Example (multi-instance)**: `{ \"instance\": \"prod\" }`\n**Returns**: List of datasources with `uid`, `name`, `type`, `isDefault`, plus routing hints: `queryTool` (which agent tool to use, e.g. `\"grafana_query\"`, `\"grafana_query_logs\"`, or `\"grafana_query_traces\"`), `queryLanguage` (e.g. `\"PromQL\"`, `\"LogQL\"`, `\"TraceQL\"`), and `supported` (boolean — whether an agent tool can query this datasource). Use `queryTool` to pick the right tool for each datasource. When multiple Grafana instances are configured, also returns `instance` (which instance was queried) and `availableInstances` (list of `{ name, url, isDefault }` for all configured instances).\n\n### `grafana_list_metrics`\n**When**: User asks \"what metrics are available?\" or you need to discover metrics before querying or composing dashboards. Also when grouping metrics by function — metadata mode adds `category` to each `openclaw_*` metric. Use `purpose` when user asks about a specific concern (e.g., \"performance metrics\", \"cost metrics\").\n**Params**: `datasourceUid` (required), `prefix` (filter by prefix), `search` (targeted discovery — server-side regex, only matching metrics returned), `purpose` (`\"performance\"` | `\"cost\"` | `\"reliability\"` | `\"capacity\"` — pre-filter by intent, composable with prefix and search), `label` (list label values instead), `metadata` (boolean — enriched results with type/help/category), `compact` (boolean — with metadata, returns only name/type/category, ~60% smaller).\n**Example names**: `{ \"datasourceUid\": \"prom1\", \"prefix\": \"openclaw_lens_\" }`\n**Example search**: `{ \"datasourceUid\": \"prom1\", \"search\": \"steps\" }`\n**Example purpose**: `{ \"datasourceUid\": \"prom1\", \"purpose\": \"performance\", \"metadata\": true }`\n**Example combined**: `{ \"datasourceUid\": \"prom1\", \"prefix\": \"openclaw_ext_\", \"search\": \"fitness\" }`\n**Example metadata**: `{ \"datasourceUid\": \"prom1\", \"metadata\": true, \"prefix\": \"openclaw_\" }`\n**Example compact**: `{ \"datasourceUid\": \"prom1\", \"metadata\": true, \"compact\": true }`\n**Returns names**: `{ metrics: [\"metric1\", \"metric2\", ...] }`. Truncated at 200.\n**Returns metadata**: `{ metadataSource, categorySummary: { cost: 3, usage: 4, session: 5, ... }, metrics: [{ name, type, help, category?, source? }, ...] }`. Use this before composing custom dashboards — type tells you counter vs gauge vs histogram, category groups `openclaw_*` metrics by function. Search also matches help text. Categories: `cost`, `usage`, `session`, `queue`, `messaging`, `webhook`, `tools`, `agent`, `custom`. `categorySummary` gives counts per category for quick overview (omitted when no `openclaw_*` metrics). Purpose maps: `performance` → session + tools, `cost` → cost + usage, `reliability` → webhook + messaging + agent, `capacity` → queue + session. `metadataSource`: `\"prometheus\"` when Prometheus metadata endpoint has data, `\"synthetic\"` when OTLP-only (metadata synthesized from known metric registry — histogram sub-metrics deduplicated, type/help from Grafana Lens definitions). On OTLP stacks, includes `hint` explaining why metadata is synthetic. `source: \"synthetic\"` on individual entries from the registry; `source: \"custom\"` on entries from the custom metrics store.\n**Returns compact**: `{ metadataSource, categorySummary: {...}, metrics: [{ name, type, category? }, ...] }`. Same as metadata but drops `help`, `source`, `labelNames` — use in multi-tool chains where you need metric names and types but not full descriptions.\n**Example label**: `{ \"datasourceUid\": \"prom1\", \"label\": \"job\" }`\n**Returns label**: `{ label, count, totalCount, values: [\"value1\", \"value2\", ...] }`. Truncated at 200.\n\n### `grafana_query`\n**When**: User asks a data question that needs a direct answer, not a dashboard. Also for re-running an existing dashboard panel's query with different time ranges.\n**Params**: `datasourceUid`, `expr` (PromQL), `queryType` (`instant`/`range`), `start` (range only, required), `end` (range only, default `\"now\"`), `step` (range only, optional — auto-calculated from time range if omitted, targeting ~300 datapoints), `dashboardUid` (optional — resolve query from panel), `panelId` (optional — use with `dashboardUid`).\n**Example instant**: `{ \"datasourceUid\": \"prom1\", \"expr\": \"sum(increase(openclaw_lens_cost_by_model_total[1d])) or vector(0)\" }`\n**Example range (auto-step)**: `{ \"datasourceUid\": \"prom1\", \"expr\": \"rate(openclaw_tokens_total[5m])\", \"queryType\": \"range\", \"start\": \"now-30d\" }`\n**Example range (explicit step)**: `{ \"datasourceUid\": \"prom1\", \"expr\": \"rate(openclaw_tokens_total[5m])\", \"queryType\": \"range\", \"start\": \"now-1h\", \"end\": \"now\", \"step\": \"60\" }`\n**Example panel re-run**: `{ \"dashboardUid\": \"openclaw-command-center\", \"panelId\": 10, \"queryType\": \"range\", \"start\": \"now-7d\" }`\n**Tip**: `start`/`end` accept Unix seconds or relative expressions like `\"now-1h\"`, `\"now-7d\"`. For range queries, just set `start` — `end` defaults to `\"now\"` and `step` is auto-calculated. Override `step` only when you need specific resolution.\n**Tip (panel re-run)**: Set `dashboardUid` + `panelId` to re-run a panel's query without manually extracting PromQL. The tool auto-resolves `expr` and `datasourceUid` from the panel definition. Template variables are replaced with wildcards. You can still override `expr` or `datasourceUid` explicitly if needed. Get panel IDs from `grafana_get_dashboard`.\n**Returns instant**: `{ metrics: [{ metric: {...}, value: \"1.23\", timestamp: \"...\", healthContext?: { status, thresholds, description, direction } }], datasourceUid, resultCount, warnings?, hint? }` — `healthContext` is included for well-known `openclaw_lens_*` gauge metrics, providing SRE-grade health assessment: `status` (\"healthy\"/\"warning\"/\"critical\"), `thresholds` (warning/critical values), `description` (what the metric means), `direction` (\"higher_is_worse\"/\"lower_is_worse\"). Omitted for unknown metrics. Capped at 50 results; when exceeded includes `truncated: true`, `totalResults`, and `truncationHint` advising to narrow the query.\n**Returns range**: `{ series: [{ metric: {...}, values: [{ time, value }...] }], datasourceUid, resultCount, warnings?, hint? }` — truncated to 20 points per series and 50 series max. When series are truncated includes `truncated: true`, `totalSeries`, and `truncationHint`. When step is auto-calculated, includes `step: { value: \"288s\", display: \"5m\", auto: true }`.\n**Returns (panel re-run)**: Includes `resolvedFrom: \"panel\"`, `panelTitle`, `panelType`, `templateVarsReplaced` alongside normal query results. If the panel uses a Loki datasource, returns an error directing you to use `grafana_query_logs` instead.\n**Returns (warnings)**: When Prometheus flags a non-fatal issue (e.g., `rate()` on a gauge), `warnings: [{ cause, suggestion, example? }]` is included. Example: `rate()` on a gauge → cause says \"rate() applied to 'metric' which appears to be a gauge\", suggestion says \"use delta() or deriv() instead\", example shows the corrected query.\n**Returns (hint)**: When the query returns zero results, `hint: { cause, suggestion }` explains why (metric may not exist, label filters may not match) and suggests using `grafana_list_metrics` to verify.\n**Returns (error with guidance)**: On query failure, includes `guidance: { cause, suggestion, example? }` alongside the raw error. Pattern-matched for common PromQL mistakes: unclosed parenthesis, missing range selector, timeout, auth failure, rate on gauge, etc. Omitted when the error is unrecognized.\n**Tip (chaining)**: Both instant and range responses include `datasourceUid` — pass it directly to `grafana_create_alert` or other tools without re-calling `grafana_explore_datasources`. This enables zero-friction query→alert chains.\n\n### `grafana_query_logs`\n**When**: User asks about logs, errors, or needs to investigate issues by searching log data. Also for session debugging, OTel log investigation, and re-running existing log panel queries.\n**Params**: `datasourceUid`, `expr` (LogQL), `queryType` (`instant`/`range`, default `range`), `start`/`end` (default `now-1h`/`now`), `step` (metric queries only), `limit` (default 100), `direction` (`backward`/`forward`), `lineLimit` (max chars per log line, default 500, max 2000), `extractFields` (boolean, default false — extract structured OTel attributes into a clean `fields` object), `dashboardUid` (optional — resolve query from panel), `panelId` (optional — use with `dashboardUid`).\n**Example log search**: `{ \"datasourceUid\": \"loki1\", \"expr\": \"{job=\\\"api\\\"} |= \\\"error\\\"\" }`\n**Example with filters**: `{ \"datasourceUid\": \"loki1\", \"expr\": \"{job=\\\"api\\\"} |~ \\\"timeout|refused\\\"\", \"limit\": 50, \"direction\": \"forward\" }`\n**Example full stack traces**: `{ \"datasourceUid\": \"loki1\", \"expr\": \"{job=\\\"api\\\"} |= \\\"Exception\\\"\", \"lineLimit\": 2000 }`\n**Example session debugging**: `{ \"datasourceUid\": \"loki1\", \"expr\": \"{service_name=\\\"openclaw\\\"} | json | component=\\\"lifecycle\\\"\", \"extractFields\": true }`\n**Example metric query**: `{ \"datasourceUid\": \"loki1\", \"expr\": \"rate({job=\\\"api\\\"}[5m])\", \"queryType\": \"range\", \"start\": \"now-6h\", \"end\": \"now\", \"step\": \"60\" }`\n**Example panel re-run**: `{ \"dashboardUid\": \"openclaw-command-center\", \"panelId\": 18, \"start\": \"now-24h\", \"extractFields\": true }`\n**Returns streams**: `{ entries: [{ labels: {...}, timestamp: \"...\", line: \"...\" }], datasourceUid, totalEntries, truncated }` — capped at 100 entries, lines at 500 chars (set `lineLimit: 2000` for full stack traces).\n**Returns streams (extractFields)**: `{ entries: [{ labels: {...cleaned...}, timestamp: \"...\", line: \"...\", fields: { component, event_name, session_id, trace_id, model, duration_s, ... } }], datasourceUid }` — infrastructure noise labels removed, `openclaw_` prefix stripped from field keys, numeric values auto-converted. Also parses JSON log bodies if present.\n**Returns streams (traceCorrelation)**: When `extractFields: true` and entries contain `trace_id`, includes `traceCorrelation: { traceIds: [...], tool: \"grafana_query_traces\", tip }` — up to 5 unique trace IDs ready for `grafana_query_traces` with `queryType: \"get\"`.\n**Returns metric**: Same shape as `grafana_query` range/instant results (matrix capped at 50 series, vector capped at 50 results — includes `datasourceUid`, `truncated`, `totalSeries`/`totalResults`, and `truncationHint` when exceeded).\n**Returns (panel re-run)**: Includes `resolvedFrom: \"panel\"`, `panelTitle`, `panelType`, `templateVarsReplaced` alongside normal results. If the panel uses a Prometheus datasource, returns an error directing you to use `grafana_query` instead.\n**Returns (error with guidance)**: On query failure, includes `guidance: { cause, suggestion, example? }` alongside the raw error. Pattern-matched for common LogQL mistakes: bare text without stream selector, empty `{}`, unclosed braces, missing label matchers, auth failure, timeout. Omitted when the error is unrecognized.\n**Tip**: LogQL: `{label=\"value\"}` selects streams, `|=` substring filter, `|~` regex, `!=` exclude. Metric wrappers: `rate()`, `count_over_time()`, `bytes_rate()`. Use `extractFields: true` when investigating OTel/lifecycle logs — it surfaces `trace_id`, `session_id`, `event_name`, `model`, and other attributes as first-class fields instead of buried in raw labels.\n**Tip (panel re-run)**: Same as `grafana_query` — set `dashboardUid` + `panelId` to auto-resolve LogQL and datasource. The tool routes Prometheus panels to `grafana_query` with a helpful error.\n\n### `grafana_query_traces`\n**When**: User asks about traces, distributed tracing, slow spans, session trace hierarchies, or needs to debug request flows across services.\n**Params**: `datasourceUid`, `query` (TraceQL expression or trace ID), `queryType` (`search`/`get`, default `search`), `start`/`end` (default `now-1h`/`now`), `limit` (default 20, max 50), `minDuration`/`maxDuration` (e.g., `\"1s\"`, `\"10s\"`), `dashboardUid` (optional — resolve query from panel), `panelId` (optional — use with `dashboardUid`).\n**Example search**: `{ \"datasourceUid\": \"tempo1\", \"query\": \"{ resource.service.name = \\\"openclaw\\\" }\" }`\n**Example search slow**: `{ \"datasourceUid\": \"tempo1\", \"query\": \"{ resource.service.name = \\\"openclaw\\\" }\", \"minDuration\": \"5s\" }`\n**Example search with time**: `{ \"datasourceUid\": \"tempo1\", \"query\": \"{ span.gen_ai.system = \\\"anthropic\\\" }\", \"start\": \"now-24h\", \"limit\": 50 }`\n**Example get**: `{ \"datasourceUid\": \"tempo1\", \"query\": \"abc123def456789...\", \"queryType\": \"get\" }`\n**Example panel re-run**: `{ \"dashboardUid\": \"openclaw-session-explorer\", \"panelId\": 12, \"start\": \"now-24h\" }`\n**Returns search**: `{ traces: [{ traceId, rootServiceName, rootTraceName, startTime, durationMs, spanCount? }], datasourceUid, totalTraces, truncated?, correlationHint? }` — capped at 50 traces. When exceeded includes `truncated: true` and `truncationHint`. When traces are found, includes `correlationHint: { logQuery, tool, tip }` with a ready-to-use LogQL expression for `grafana_query_logs`.\n**Returns get**: `{ traceId, spans: [{ traceId, spanId, parentSpanId?, operationName, serviceName, startTime, durationMs, status, kind?, attributes: {...} }], datasourceUid, totalSpans, truncated? }` — flattened OTLP spans with resolved attributes (string/number/boolean). Capped at 200 spans. Sorted by start time (earliest first).\n**Returns (panel re-run)**: Includes `resolvedFrom: \"panel\"`, `panelTitle`, `panelType`, `templateVarsReplaced` alongside normal results. If the panel uses a Prometheus or Loki datasource, returns an error directing you to use the correct tool.\n**Returns (error with guidance)**: On query failure, includes `guidance: { cause, suggestion, example? }` alongside the raw error. Pattern-matched for common TraceQL mistakes: syntax errors, invalid attributes, auth failure, timeout, not-found, invalid trace ID. Omitted when the error is unrecognized.\n**Returns (no results)**: When search returns zero traces, includes `hint: { cause, suggestion }` suggesting to broaden the query or check the datasource.\n**Tip**: TraceQL: `{ }` matches all traces, `resource.service.name` for service filter, `span.http.status_code` for HTTP spans, `name` for operation name, `duration` for span duration, `status` for error/ok filtering. Use `minDuration`/`maxDuration` to find performance outliers. **Trace-to-Log**: search and get results include `correlationHint.logQuery` — pass it directly to `grafana_query_logs` to find correlated logs. **Log-to-Trace**: `grafana_query_logs` results (with `extractFields: true`) include `traceCorrelation.traceIds` — pass any ID to `grafana_query_traces` with `queryType: \"get\"`.\n**Tip (panel re-run)**: Same as `grafana_query` — set `dashboardUid` + `panelId` to auto-resolve TraceQL and datasource. The tool routes Prometheus/Loki panels to the correct tool with a helpful error.\n\n### `grafana_create_dashboard`\n**When**: User wants a persistent dashboard for ongoing monitoring.\n**Params**: `template` or `dashboard` (custom JSON) — one required. Optional: `title` (overrides template default), `folderUid` (target folder), `overwrite` (default `true`).\n**Returns**: `{ uid, url, status, message, suggestedNext?: [{ template, reason }], validation?: DashboardValidation }`. For template-based dashboards, `suggestedNext` lists complementary templates to deploy next. For custom JSON dashboards, `validation` dry-runs each panel's PromQL and reports per-panel health — check `validation.panelsError` for broken queries.\n\n**Choose the right template (3-tier SRE drill-down hierarchy):**\n\n**Tier 1 → System:** Start here for overall health.\n**Tier 2 → Session:** Click a session from Tier 1 to investigate.\n**Tier 3 → Deep Dive:** Cost, tool, or SRE details.\n\n| Template | Tier | Domain | Variables | Use When |\n|----------|------|--------|-----------|----------|\n| `llm-command-center` | **Tier 1** | System overview | `$prometheus`, `$loki`, `$tempo`, `$provider`, `$model`, `$channel` | Golden signals, session table with click-to-drill-down, cost, cache, live feeds |\n| `session-explorer` | **Tier 2** | Session debug | `$prometheus`, `$loki`, `$tempo`, `$session` (textbox) | Per-session trace hierarchy, LLM calls, tool calls, conversation flow |\n| `cost-intelligence` | Tier 3a | Cost analysis | `$prometheus`, `$loki`, `$provider`, `$model` | Spending trends, model attribution, cache savings, per-session cost table |\n| `tool-performance` | Tier 3b | Tool analytics | `$prometheus`, `$loki`, `$tempo`, `$tool` | Tool leaderboard, latency ranking, error rates, tool traces |\n| `sre-operations` | Tier 3c | SRE operations | `$prometheus`, `$loki` | Queue health, webhooks, stuck sessions, tool loops |\n| `genai-observability` | — | **OTel gen_ai standard** | `$prometheus`, `$loki`, `$tempo`, `$model`, `$provider` | Industry-standard AI monitoring: token analytics, LLM performance, traces, logs, cache efficiency. Works with any gen_ai data. |\n| `node-exporter` | — | System/DevOps | `$datasource`, `$instance` | Server CPU, memory, disk, network |\n| `http-service` | — | Web/DevOps | `$datasource`, `$job` | HTTP request rate, errors, latency (RED signals) |\n| `metric-explorer` | — | **Any domain** | `$datasource`, `$metric` | Deep-dive into any single metric from a dropdown |\n| `multi-kpi` | — | **Any domain** | `$datasource`, `$metric1`..`$metric4` | 4-metric KPI overview (business, fitness, finance, IoT) |\n| `weekly-review` | — | **Any domain** | `$datasource`, `$metric1`, `$metric2` | Weekly overview of 2 external metrics with trends + all openclaw_ext_* table |\n\nAll AI templates have Loki log-to-trace correlation via Tempo + stable UIDs for cross-dashboard navigation.\n\n**Example AI health**: `{ \"template\": \"llm-command-center\", \"title\": \"My AI Dashboard\" }`\n**Example session debug**: `{ \"template\": \"session-explorer\", \"title\": \"Session Debug\" }`\n**Example cost analysis**: `{ \"template\": \"cost-intelligence\", \"title\": \"My AI Costs\" }`\n**Example tool analytics**: `{ \"template\": \"tool-performance\", \"title\": \"Tool Health\" }`\n**Example SRE ops**: `{ \"template\": \"sre-operations\", \"title\": \"SRE Health\" }`\n**Example GenAI observability**: `{ \"template\": \"genai-observability\", \"title\": \"GenAI Observability\" }`\n**Example system**: `{ \"template\": \"node-exporter\", \"title\": \"Server Health\" }`\n**Example generic**: `{ \"template\": \"metric-explorer\", \"title\": \"Explore My Data\" }`\n**Example multi-KPI**: `{ \"template\": \"multi-kpi\", \"title\": \"Business KPIs\" }`\n**Example weekly review**: `{ \"template\": \"weekly-review\", \"title\": \"My Weekly Review\" }`\n**Example custom with validation**: `{ \"dashboard\": { \"title\": \"Model Comparison\", \"panels\": [{ \"id\": 1, \"title\": \"Cost by Model\", \"type\": \"timeseries\", \"targets\": [{ \"refId\": \"A\", \"expr\": \"sum by (model) (rate(openclaw_lens_cost_by_token_type[1h]))\", \"datasource\": { \"uid\": \"prometheus\" } }] }] } }`\n\n**Custom dashboard validation** (returned only for `dashboard` param, not templates):\n`validation: { panelsTotal: 3, panelsValid: 1, panelsNoData: 1, panelsError: 1, panelsSkipped: 0, details: [{ panelId: 1, title: \"Cost by Model\", status: \"ok\", queries: [{ refId: \"A\", expr: \"...\", valid: true, sampleValue: 0.42 }] }, { panelId: 2, title: \"Latency\", status: \"nodata\" }, { panelId: 3, title: \"Bad Query\", status: \"error\", error: \"parse error at char 5\" }] }`\nPanel statuses: `ok` (query returned data), `nodata` (valid query, no results — metric may not exist yet), `error` (PromQL syntax error or datasource issue), `skipped` (no datasource UID found). Dashboard is always created regardless — validation is informational.\n\n### `grafana_update_dashboard`\n**When**: User wants to add a panel, remove a panel, change a query, update dashboard settings, or delete a dashboard.\n**Params**: `uid` (required), `operation` (required: `add_panel`, `remove_panel`, `update_panel`, `update_metadata`, `delete`).\n**add_panel params**: `panel` (object with `title`, `type`, `targets`). Auto-layouts below existing panels.\n**remove_panel / update_panel params**: `panelId` (preferred) or `panelTitle` (case-insensitive substring fallback). `updates` (object) for update_panel.\n**update_metadata params**: `title`, `description`, `tags`, `time` (e.g., `{ \"from\": \"now-7d\", \"to\": \"now\" }`), `refresh` (e.g., `\"1m\"`).\n**delete params**: None besides `uid` — permanently removes the dashboard. Always confirm with user first.\n**Example add**: `{ \"uid\": \"abc123\", \"operation\": \"add_panel\", \"panel\": { \"title\": \"Error Rate\", \"type\": \"timeseries\", \"targets\": [{ \"refId\": \"A\", \"expr\": \"rate(errors_total[5m])\", \"datasource\": { \"uid\": \"prom1\" } }] } }`\n**Example add (no datasource)**: `{ \"uid\": \"abc123\", \"operation\": \"add_panel\", \"panel\": { \"title\": \"Latency\", \"type\": \"timeseries\", \"targets\": [{ \"refId\": \"A\", \"expr\": \"histogram_quantile(0.99, rate(http_duration_bucket[5m]))\" }] } }` — validation skipped if no datasource UID found, panel still saved.\n**Example remove**: `{ \"uid\": \"abc123\", \"operation\": \"remove_panel\", \"panelId\": 3 }`\n**Example update panel**: `{ \"uid\": \"abc123\", \"operation\": \"update_panel\", \"panelId\": 1, \"updates\": { \"title\": \"New Title\", \"targets\": [{ \"refId\": \"A\", \"expr\": \"new_query\" }] } }`\n**Example update metadata**: `{ \"uid\": \"abc123\", \"operation\": \"update_metadata\", \"title\": \"My Dashboard v2\", \"time\": { \"from\": \"now-7d\", \"to\": \"now\" }, \"refresh\": \"5m\" }`\n**Example delete**: `{ \"uid\": \"abc123\", \"operation\": \"delete\" }`\n**Returns update**: `{ status: \"updated\", uid, url, version, operation, panelCount, affectedPanel?: { id, title }, changedFields?: [...], queryValidation?: { validated, results, datasourceUid?, skippedReason? } }`.\n**Returns queryValidation**: For `add_panel` and `update_panel` (when targets change), PromQL queries are dry-run against Grafana. Each result: `{ refId, expr, valid: boolean, error?: string, sampleValue?: number }`. Panel is always saved — validation is informational. If `valid: false`, check the `error` field for PromQL syntax issues. If `skippedReason` is set, no datasource UID was found — include `datasource: { uid: \"...\" }` on targets to enable validation.\n**Returns delete**: `{ status: \"deleted\", uid, title, message }`.\n**Tip**: `targets` in update_panel replaces entirely — include all targets, not just changed ones. Include `datasource.uid` on targets for query validation feedback.\n\n### `grafana_get_dashboard`\n**When**: Need to inspect a dashboard's panels — find panel IDs for sharing, verify structure, scan multiple dashboards for an overview, or audit which panels are returning data.\n**Params**: `uid` (required). Optional: `compact` (boolean, default `false`) — return panel titles and types only, no queries or metadata (~70% smaller). `audit` (boolean, default `false`) — dry-run each panel's query and add `health` status.\n**Example (full)**: `{ \"uid\": \"abc123\" }`\n**Example (compact overview)**: `{ \"uid\": \"abc123\", \"compact\": true }`\n**Example (audit)**: `{ \"uid\": \"abc123\", \"audit\": true }`\n**Returns (full)**: `{ uid, title, description?, url, tags, time?, refresh?, panelCount, panels: [{ id, title, type, queries: [{ refId, expr }] }], folderUid, created?, updated? }`.\n**Returns (compact)**: `{ uid, title, url, tags, panelCount, panels: [{ id, title, type }] }`.\n**Returns (audit)**: Same as full, plus each panel gets `health: { status: \"ok\"|\"nodata\"|\"error\"|\"skipped\", error?, sampleValue? }` and the response includes `auditSummary: { ok, nodata, error, skipped }`. Resolves template variable datasources (`$prometheus`, `$loki`) and replaces expression template vars with wildcards.\n**Tip**: Use `audit: true` when the user asks \"which panels are broken?\" or \"audit my dashboard\" — it replaces N separate `grafana_query` calls with one tool call. Use `compact: true` for lightweight overview scans. Omit both when you need query details (before update or share).\n\n### `grafana_search`\n**When**: User mentions a dashboard by name, before creating one (check duplicates), or for reporting/audit workflows.\n**Params**: `query` (required). Optional: `tags` (array — filter by tags), `starred` (boolean — only starred), `sort` (`\"alpha-asc\"`/`\"alpha-desc\"`), `limit` (number, default 100), `enrich` (boolean — add `updatedAt` + `panelCount` per result, default false).\n**Example**: `{ \"query\": \"cost\" }`\n**Example with tags**: `{ \"query\": \"\", \"tags\": [\"production\"] }`\n**Example starred**: `{ \"query\": \"\", \"starred\": true, \"limit\": 10 }`\n**Example enriched**: `{ \"query\": \"\", \"enrich\": true }`\n**Returns**: `{ count, enriched, dashboards: [{ uid, title, url, tags, folderTitle?, folderUid?, updatedAt?, panelCount? }] }`. `folderTitle`/`folderUid` always included when dashboard is in a folder. `updatedAt` (ISO 8601) and `panelCount` only present when `enrich: true` — enables staleness detection and reporting without per-dashboard `get_dashboard` calls.\n**Tip**: Use `enrich: true` for reporting workflows (\"which dashboards are stale?\", \"give me a summary of all dashboards\"). Skip enrichment for simple lookups. After finding a dashboard, use `grafana_get_dashboard` to inspect panels, `grafana_share_dashboard` to render a chart, or `grafana_update_dashboard` to modify it.\n\n### `grafana_share_dashboard`\n**When**: User says \"show me\" or \"send me\" a chart/dashboard.\n**Params**: `dashboardUid`, `panelId` (required). Optional: `from` (default `\"now-6h\"`), `to` (default `\"now\"`), `width` (default `1000`), `height` (default `500`), `theme` (`\"light\"`/`\"dark\"`, default `\"dark\"`).\n**Example**: `{ \"dashboardUid\": \"abc123\", \"panelId\": 2, \"from\": \"now-6h\", \"to\": \"now\" }`\n**Returns**: Image rendered inline (tier 1), or snapshot URL (tier 2), or deep link (tier 3). Always delivers something. Includes `deliveryTier` (`\"image\"` | `\"snapshot\"` | `\"link\"`), `rendererAvailable` (boolean — false when Image Renderer plugin is missing), `renderFailureReason` (why image rendering failed), and `remediation` (how to fix it). Tier 3 also includes `snapshotFailureReason`.\n**Tip**: Use `grafana_get_dashboard` first to find panel IDs. If `rendererAvailable` is false, tell the user to install the grafana-image-renderer plugin.\n\n### `grafana_create_alert`\n**When**: User wants notifications when a metric crosses a threshold.\n**Params**: `title`, `datasourceUid`, `expr` (PromQL), `threshold` (all required). Optional: `evaluation` (`\"instant\"`/`\"rate\"`/`\"increase\"`, default `\"instant\"`), `evaluationWindow` (default `\"5m\"`, used with `rate`/`increase`), `condition` (`gt`/`lt`/`gte`/`lte`, default `gt`), `for` (duration, default `5m`), `folderUid`, `labels` (e.g., `{ \"severity\": \"warning\" }`), `annotations` (e.g., `{ \"summary\": \"Cost too high\" }`), `noDataState` (`NoData`/`Alerting`/`OK`, default `NoData`).\n**IMPORTANT**: For counter metrics (`*_total`), always use `evaluation: \"rate\"` (per-second rate) or `evaluation: \"increase\"` (total change over window). Raw counter values always increase and will immediately breach any threshold. Use `\"instant\"` (default) only for gauges.\n**Example gauge alert**: `{ \"title\": \"High Cost Alert\", \"datasourceUid\": \"prom1\", \"expr\": \"openclaw_lens_daily_cost_usd\", \"threshold\": 5, \"condition\": \"gt\" }`\n**Example rate alert**: `{ \"title\": \"High Error Rate\", \"datasourceUid\": \"prom1\", \"expr\": \"openclaw_lens_webhook_error_total\", \"threshold\": 0.1, \"evaluation\": \"rate\" }`\n**Example increase alert**: `{ \"title\": \"Token Burst\", \"datasourceUid\": \"prom1\", \"expr\": \"openclaw_lens_tokens_total\", \"threshold\": 10000, \"evaluation\": \"increase\", \"evaluationWindow\": \"1h\" }`\n**Returns**: `{ uid, title, status: \"created\", datasourceUid, url, evaluation?: { mode, window, evaluatedExpr }, metricValidation: { valid, error?, sampleValue? }, message }`. The `datasourceUid` echoes back which datasource the rule targets (verify correctness). `metricValidation` dry-runs the expression before creation — `valid: true` + `sampleValue` confirms data exists; `valid: false` + `error` warns of typos/missing metrics. Alert is always created regardless (metric may not have data yet). When `evaluation` is `\"rate\"` or `\"increase\"`, validation runs the wrapped expression.\n**Note**: Auto-creates a \"Grafana Lens Alerts\" folder if no `folderUid` is specified.\n\n### `grafana_annotate`\n**When**: User deploys, changes config, or wants to mark an event for correlation.\n**Params**: `action` (`\"create\"` default, or `\"list\"`).\n**Create params**: `text` (required), `tags`, `dashboardUid`, `panelId`, `time` (epoch ms or relative like `\"now-2h\"`, default now), `timeEnd` (epoch ms or relative).\n**List params**: `from`, `to` (epoch ms or relative like `\"now-7d\"`, `\"now-24h\"`, `\"now\"`), `tags`, `limit` (default `20`).\n**Time formats**: All time params accept epoch ms (e.g., `1700000000000`) OR Grafana-style relative strings (`\"now\"`, `\"now-1h\"`, `\"now-7d\"`, `\"now-30m\"`). Prefer relative strings — they're simpler and avoid arithmetic errors.\n**Example create**: `{ \"text\": \"Deployed v2.1.0\", \"tags\": [\"deploy\", \"production\"] }`\n**Example create past**: `{ \"text\": \"Incident started\", \"time\": \"now-2h\", \"timeEnd\": \"now-30m\", \"tags\": [\"incident\"] }`\n**Example list recent**: `{ \"action\": \"list\", \"from\": \"now-7d\", \"to\": \"now\", \"tags\": [\"deploy\"] }`\n**Example list**: `{ \"action\": \"list\", \"tags\": [\"deploy\"], \"limit\": 10 }`\n**Returns create**: `{ status: \"created\", id, message, time, comparisonHint: { beforeWindow: { from, to }, afterWindow: { from, to }, suggestion } }`. The `comparisonHint` provides ready-to-use ISO 8601 time ranges (30-min windows) for before/after comparison via `grafana_query` — no manual time math needed. For region annotations (with `timeEnd`), `afterWindow` starts at `timeEnd`.\n**Returns list**: `{ annotations: [{ id, text, tags, time, timeEnd?, dashboardUID?, panelId? }] }`.\n\n### `grafana_check_alerts`\n**When**: Prompt context shows \"GRAFANA ALERTS\", need to manage alert rules (list/delete), set up the alert webhook, silence alerts during investigation, or acknowledge an investigated alert.\n**Params**: `action` (`\"list\"` default, `\"acknowledge\"`, `\"list_rules\"`, `\"de\n\nFile v0.5.0:README.md\n\n# Grafana Lens\n\n**Agent-driven Grafana observability for OpenClaw — query, visualize, alert, trace, and share across 15+ messaging channels.**\n\n> **Note:** This is a community-built OpenClaw plugin, not an official Grafana Labs product. Grafana, Loki, Tempo, and Prometheus are trademarks of Grafana Labs.\n\n[OpenClaw](https://openclaw.com) is an open-source AI agent platform. Grafana Lens extends it with full Grafana integration — 18 composable tools that let your agent query metrics and logs, trace distributed requests, create dashboards, set up alerts, render charts, run security audits, investigate incidents, push custom data, and manage data collection pipelines — all through natural language conversation.\n\n---\n\n## Why Grafana Lens?\n\n| Pain Point | How Grafana Lens Helps |\n|---|---|\n| **\"Where did my budget go?\"** | Cost dashboards with model-level attribution, token tracking, and cost anomaly alerts |\n| **\"Is my agent stuck in a loop?\"** | Real-time tool loop detection, stuck session monitoring, and SRE operations dashboard |\n| **\"Am I being prompt-injected?\"** | 12-pattern prompt injection detection with security dashboard and threat-level reporting |\n| **\"I need observability but don't want another SaaS\"** | Fully self-hosted, open source, OTLP-native — runs on a free local Grafana stack |\n| **\"I can't debug multi-step agent sessions\"** | Hierarchical traces: session → LLM call → tool execution, with log-to-trace correlation |\n| **\"My alert fired — now what?\"** | `grafana_investigate` gathers metrics, logs, traces in parallel and generates hypotheses with specific tool+params for follow-up |\n| **\"I want to track my own data in Grafana\"** | Push any custom metrics (fitness, calendar, git, finance) from conversation |\n| **\"How do I get my data INTO Grafana?\"** | `alloy_pipeline` sets up data collection from databases, Docker, Kubernetes, log files, and more — 29 recipes, just describe what you want to monitor |\n\n---\n\n## Key Features\n\n- **18 Composable Agent Tools** — Query PromQL/LogQL/TraceQL, create dashboards, set alerts, share panel images, run security checks, investigate incidents, push custom metrics, manage data collection pipelines, and more\n- **SRE Investigation** — Multi-signal triage (`grafana_investigate`), anomaly scoring with z-score against 7-day baselines, seasonality comparison, and alert fatigue detection\n- **Full OTLP Observability** — Metrics → Prometheus, Logs → Loki, Traces → Tempo. Push-based with no scraping — data is available immediately\n- **Security Monitoring** — 6-check threat assessment covering prompt injection, cost anomalies, tool loops, session enumeration, webhook errors, and stuck sessions\n- **12 Pre-Built Dashboard Templates** — From LLM Command Center and Cost Intelligence to Security Overview and SRE Operations\n- **Custom Data Observatory** — Push any external data (calendar events, git commits, fitness stats, financial metrics) into Grafana via conversation\n- **Works with ANY Datasource** — Not limited to OpenClaw metrics. Query any Prometheus or Loki datasource configured in your Grafana instance\n- **Data Collection Pipeline Management** — 29 pre-built Alloy pipeline recipes across 5 categories (metrics, logs, traces, infrastructure, profiling) — describe what you want to monitor and the agent handles the Alloy configuration\n\n---\n\n## Quick Start\n\n```bash\n# 1. Start the LGTM observability stack (Grafana + Prometheus + Loki + Tempo + OTel Collector)\n#    See https://github.com/grafana/docker-otel-lgtm for more options\ndocker pull grafana/otel-lgtm:latest\ndocker run -d --name lgtm -p 3000:3000 -p 4317:4317 -p 4318:4318 -p 9090:9090 grafana/otel-lgtm:latest\n\n# 2. Install the plugin\nopenclaw plugins install openclaw-grafana-lens\n\n# 3. Configure credentials (see \"Configuration\" section below for full options)\nexport GRAFANA_URL=http://localhost:3000\nexport GRAFANA_SERVICE_ACCOUNT_TOKEN=glsa_xxxxxxxxxxxx\n\n# 4. Restart the gateway to load the plugin\nopenclaw gateway restart\n\n# Optional: Enable Alloy pipeline management (for data collection)\n# See \"Alloy Pipeline Management\" section below for full setup\nexport ALLOY_URL=http://localhost:12345\nexport ALLOY_CONFIG_DIR=/path/to/alloy/config.d\n```\n\n---\n\n## Table of Contents\n\n- [Prerequisites](#prerequisites)\n- [Setup Guide](#setup-guide)\n  - [LGTM Stack (Docker)](#1-lgtm-stack-docker)\n  - [Plugin Installation](#2-plugin-installation)\n  - [Configuration](#3-configuration)\n- [What Can You Do?](#what-can-you-do)\n- [Agent Tools](#agent-tools)\n- [Alloy Pipeline Management](#alloy-pipeline-management)\n- [Dashboard Templates](#dashboard-templates)\n- [Security Monitoring](#security-monitoring)\n- [Custom Metrics](#custom-metrics-data-observatory)\n- [Architecture](#architecture)\n- [Observability Deep Dive](#observability-deep-dive)\n- [Configuration Reference](#configuration-reference)\n- [Development](#development)\n- [License](#license)\n\n---\n\n## Prerequisites\n\n- **OpenClaw** installed and running\n- **Docker** (for the recommended LGTM stack) or an existing Grafana instance\n  - [Grafana](https://grafana.com/grafana/) is an open-source platform for monitoring, visualization, and alerting. It connects to data sources like Prometheus (metrics), Loki (logs), and Tempo (traces) to give you dashboards, alerts, and exploration tools for any data.\n- **Grafana Service Account Token** with Editor role (see setup steps below)\n\n---\n\n## Setup Guide\n\n### 1. LGTM Stack (Docker)\n\nThe recommended local setup uses the [`grafana/otel-lgtm`](https://github.com/grafana/docker-otel-lgtm) all-in-one Docker image, maintained by Grafana Labs. It bundles everything you need in a single container — no configuration files, no compose setup:\n\n| Component | Port | Purpose |\n|-----------|------|---------|\n| Grafana | 3000 | Dashboards, alerting, visualization |\n| Prometheus | 9090 | Metrics storage and querying |\n| Loki | — | Log aggregation |\n| Tempo | — | Distributed tracing |\n| OTel Collector | 4317 (gRPC), 4318 (HTTP) | Receives OTLP metrics, logs, and traces |\n\n```bash\ndocker pull grafana/otel-lgtm:latest\ndocker run -d --name lgtm -p 3000:3000 -p 4317:4317 -p 4318:4318 -p 9090:9090 grafana/otel-lgtm:latest\n```\n\n> **Tip:** You can also use the [`run-lgtm.sh`](https://github.com/grafana/docker-otel-lgtm) scripts from the official repo for additional options like volume persistence and OBI auto-instrumentation.\n\n**Default credentials:** `admin` / `admin`\n\n#### Create a Service Account Token\n\n1. Open Grafana at `http://localhost:3000`\n2. Go to **Administration → Service Accounts → Add service account**\n3. Name it (e.g., `grafana-lens`), set role to **Editor**\n4. Click **Add service account token → Generate token**\n5. Copy the token (starts with `glsa_`) — you'll need it for configuration\n\n### 2. Plugin Installation\n\n```bash\n# Install from npm\nopenclaw plugins install openclaw-grafana-lens\n\n# Restart the gateway to load the plugin\nopenclaw gateway restart\n```\n\nFor local development:\n\n```bash\n# Clone and link locally\ngit clone <repo-url> ~/workspace/grafana-lens\ncd ~/workspace/grafana-lens && npm install\nopenclaw plugins install -l ~/workspace/grafana-lens\nopenclaw gateway restart\n```\n\n### 2b. Grafana Alloy (Optional -- Data Collection Pipelines)\n\n[Grafana Alloy](https://grafana.com/docs/alloy/latest/) is Grafana Labs' open-source telemetry collector. It collects metrics, logs, traces, and profiles from your infrastructure and applications, then forwards them to backends like Prometheus, Loki, Tempo, and Pyroscope. Think of it as the \"data pipeline\" -- it gets your data INTO Grafana.\n\nGrafana Lens manages Alloy pipelines through the `alloy_pipeline` tool. Once Alloy is running, the agent can create, update, delete, and diagnose pipelines via conversation -- just describe what you want to monitor.\n\n#### Option A: Docker Compose (Full LGTM + Alloy Stack)\n\nThe [alloy-scenarios](https://github.com/grafana/alloy/tree/main/example) repository includes a `grafana-lens-test` environment with a complete LGTM + Alloy stack:\n\n```bash\ncd alloy-scenarios/grafana-lens-test\ndocker compose --env-file ../image-versions.env up -d\n\n# Verify\ncurl -sf http://localhost:3000/api/health && echo \"Grafana OK\"\ncurl -sf http://localhost:12345/-/ready && echo \"Alloy OK\"\ncurl -sf http://localhost:9090/-/ready && echo \"Prometheus OK\"\n```\n\n| Service | Port | Purpose |\n|---------|------|---------|\n| Alloy | 12345 | Telemetry collector (HTTP API + live debugging UI) |\n| Grafana | 3000 | Dashboards and visualization |\n| Prometheus | 9090 | Metrics storage |\n| Loki | -- | Log aggregation |\n| Tempo | -- | Distributed tracing |\n\n#### Option B: Add Alloy to an Existing Setup\n\nInstall Alloy standalone ([installation guide](https://grafana.com/docs/alloy/latest/set-up/install/)):\n\n```bash\n# macOS\nbrew install grafana/grafana/alloy\n\n# Linux (Debian/Ubuntu)\nsudo apt install alloy\n\n# Run in directory mode (loads all .alloy files from the directory)\nmkdir -p /path/to/alloy/config.d\nalloy run /path/to/alloy/config.d/\n```\n\nAlloy runs its HTTP API on port `12345` by default. Grafana Lens communicates with this API to reload configuration after creating or updating pipelines. No restart needed -- Alloy picks up new config files via a hot-reload API call.\n\n> **Key insight:** `alloy.url` in plugin config uses `localhost:12345` (your machine → Alloy), but `alloy.lgtm.*` URLs use Docker service names (e.g., `http://prometheus:9090`) because they are embedded in generated `.alloy` configs that run *inside* Docker. If you're running everything on bare metal, all URLs use `localhost`.\n\n### 3. Configuration\n\n#### Minimal Setup (Environment Variables)\n\nThe simplest way to configure Grafana Lens — just two environment variables:\n\n```bash\nexport GRAFANA_URL=http://localhost:3000\nexport GRAFANA_SERVICE_ACCOUNT_TOKEN=glsa_xxxxxxxxxxxx\n```\n\n#### Full Configuration\n\nFor more control, add the plugin config to `~/.openclaw/openclaw.json`:\n\n```jsonc\n{\n  \"plugins\": {\n    \"entries\": {\n      \"openclaw-grafana-lens\": {\n        \"enabled\": true,\n        \"config\": {\n          \"grafana\": {\n            \"url\": \"http://localhost:3000\",        // or set GRAFANA_URL env var\n            \"apiKey\": \"glsa_xxxxxxxxxxxx\",         // or set GRAFANA_SERVICE_ACCOUNT_TOKEN env var\n            \"orgId\": 1                             // optional, default 1\n          },\n          \"metrics\": {\n            \"enabled\": true                        // enable OTLP telemetry (default: true)\n          },\n          \"otlp\": {\n            \"endpoint\": \"http://localhost:4318/v1/metrics\",  // OTLP collector endpoint\n            \"exportIntervalMs\": 15000,             // push interval in ms (default: 15000)\n            \"logs\": true,                          // push logs to Loki (default: true)\n            \"traces\": true,                        // push traces to Tempo (default: true)\n            \"captureContent\": true,                // include prompts/completions in telemetry (default: true)\n            \"contentMaxLength\": 2000,              // truncation limit for content fields (default: 2000)\n            \"forwardAppLogs\": true,                // forward OpenClaw app logs to Loki (default: true)\n            \"appLogMinSeverity\": \"debug\",          // min severity: trace/debug/info/warn/error/fatal\n            \"redactSecrets\": true                  // auto-strip API keys/tokens before export (default: true)\n          },\n          \"proactive\": {\n            \"enabled\": false,                      // enable Grafana alert webhook handler\n            \"webhookPath\": \"/grafana-lens/alerts\", // HTTP path for webhook endpoint\n            \"costAlertThreshold\": 5.0              // daily cost alert threshold in USD\n          },\n          \"customMetrics\": {\n            \"enabled\": true,                       // allow custom metric push (default: true)\n            \"maxMetrics\": 100,                     // max metric definitions\n            \"maxLabelsPerMetric\": 5,               // max label keys per metric\n            \"maxLabelValues\": 50,                  // max unique label combos per metric\n            \"defaultTtlDays\": null                 // optional auto-expiry in days\n          },\n          \"alloy\": {\n            \"enabled\": true,                       // enable Alloy pipeline management (default: false)\n            \"url\": \"http://localhost:12345\",        // Alloy HTTP API (or set ALLOY_URL env var)\n            \"configDir\": \"/path/to/alloy/config.d\", // where pipeline configs are written (or set ALLOY_CONFIG_DIR)\n            \"filePrefix\": \"lens-\",                 // prefix for managed config files (default: \"lens-\")\n            \"maxPipelines\": 20,                    // max managed pipelines (default: 20)\n            \"lgtm\": {                              // export target URLs embedded in generated configs\n              \"prometheusRemoteWriteUrl\": \"http://prometheus:9090/api/v1/write\",\n              \"lokiUrl\": \"http://loki:3100/loki/api/v1/push\",\n              \"otlpEndpoint\": \"http://tempo:4318\",\n              \"pyroscopeUrl\": \"http://localhost:4040\"  // optional, for profiling recipes\n            }\n          }\n        }\n      }\n    }\n  }\n}\n```\n\n#### Environment Variable Fallbacks\n\n| Env Var | Config Key | Description |\n|---------|------------|-------------|\n| `GRAFANA_URL` | `grafana.url` | Grafana instance URL |\n| `GRAFANA_SERVICE_ACCOUNT_TOKEN` | `grafana.apiKey` | Service account token |\n| `OTEL_EXPORTER_OTLP_ENDPOINT` | `otlp.endpoint` | OTLP collector base URL (auto-appends `/v1/metrics`) |\n| `OTEL_EXPORTER_OTLP_HEADERS` | `otlp.headers` | Custom headers in `key=value,key2=value2` format |\n| `ALLOY_URL` | `alloy.url` | Alloy HTTP API URL (default: `http://localhost:12345`) |\n| `ALLOY_CONFIG_DIR` | `alloy.configDir` | Directory where pipeline config files are written |\n\nConfig precedence: explicit plugin config > environment variables > defaults.\n\n---\n\n## What Can You Do?\n\n### For OpenClaw Users — Bring Grafana Into Your Agent's Toolkit\n\nJust talk to your agent in natural language:\n\n- **\"Create a cost dashboard\"** — Creates a pre-built cost intelligence dashboard with model attribution, cache savings, and spending trends\n- **\"Alert me if daily spend exceeds $5\"** — Sets up a Grafana-native alert rule with PromQL condition\n- **\"Show me a chart of my token usage\"** — Renders a panel as a PNG image and delivers it inline to your chat\n- **\"Am I being attacked?\"** — Runs 6 parallel security checks and reports a threat level (green/yellow/red)\n- **\"Track my daily steps\"** — Pushes custom fitness data into Grafana via OTLP for personal dashboards\n- **\"What metrics are available?\"** — Discovers all metrics in your Prometheus datasource with descriptions\n- **\"What does openclaw_lens_daily_cost_usd mean? Why did it spike?\"** — Explains the metric with current value, trend, stats, and drill-down suggestions\n- **\"Find slow traces in the last hour\"** — Searches Tempo traces by duration, status, or span attributes using TraceQL\n- **\"Show me the trace for session abc123\"** — Retrieves full distributed trace with hierarchical span details\n- **\"Investigate this alert\"** — Gathers metrics, logs, traces, and annotations in parallel and suggests hypotheses with specific follow-up tool+params\n- **\"Is this metric anomalous?\"** — Shows z-score against 7-day baseline, seasonality comparison (vs 1 day ago, vs 7 days ago), and severity rating\n- **\"Which alerts are noisy?\"** — Detects always-firing, flapping, and error/nodata rules with optimization suggestions\n- **\"Monitor my Postgres database\"** — Creates an Alloy pipeline using the postgres-exporter recipe with credential handling and sample queries\n- **\"Collect Docker container logs\"** — Sets up a docker-logs pipeline that forwards container logs to Loki\n- **\"Set up OTLP trace collection\"** — Deploys an otlp-receiver pipeline to accept traces from your applications\n- **\"What pipelines are running?\"** — Lists all managed Alloy pipelines with status and signal type\n- **Create and customize your own unique dashboards** — Combine any data from Prometheus, Loki, or custom metrics to build personalized monitoring views\n\n### For Grafana Power Users — Let an AI Agent Manage Your Grafana\n\n- **Natural language PromQL** — Ask questions about your metrics without memorizing syntax\n- **Automated alert creation** — Describe alert conditions in plain English; the agent generates the PromQL and creates the rule\n- **Dashboard management** — Create, update, add/remove panels, or delete dashboards via conversation\n- **Panel sharing** — Render any dashboard panel as an image and deliver it to Telegram, Slack, Discord, and 15+ other channels\n- **Infrastructure monitoring** — Use pre-built templates for node-exporter and HTTP service monitoring\n- **Metric exploration** — Discover, explain, and drill into any metric with `grafana_explain_metric`\n- **Dashboard auditing** — Health-check all panels in a dashboard, find broken queries, verify datasource connectivity\n- **Search and discovery** — Find existing dashboards by title, tags, or starred status\n\n---\n\n## Agent Tools\n\nAll 18 tools are registered automatically when the plugin loads. The agent decides when to use each tool based on your request.\n\n| Tool | Description | Example Use |\n|------|-------------|-------------|\n| `grafana_explore_datasources` | Discover datasources configured in Grafana | \"What data sources do I have?\" |\n| `grafana_list_metrics` | Discover available metrics with optional metadata | \"What metrics are available?\" |\n| `grafana_query` | Run PromQL instant or range queries | \"What's my token usage today?\" |\n| `grafana_query_logs` | Run LogQL queries against Loki | \"Show me error logs from the last hour\" |\n| `grafana_create_dashboard` | Create dashboards from 12 templates or custom JSON | \"Create a cost dashboard\" |\n| `grafana_update_dashboard` | Add, remove, or update panels; change dashboard metadata | \"Add a latency panel to my dashboard\" |\n| `grafana_get_dashboard` | Get compact dashboard summary with optional health audit | \"What panels are on my overview?\" |\n| `grafana_search` | Search dashboards by title, tags, or starred status | \"Do I have a cost dashboard?\" |\n| `grafana_share_dashboard` | Render panels as PNG images for inline delivery | \"Show me a chart of my costs\" |\n| `grafana_create_alert` | Create Grafana-native alert rules with PromQL conditions | \"Alert me if daily cost > $5\" |\n| `grafana_check_alerts` | List, acknowledge, silence alerts; manage rules and webhooks | \"Are there any alerts firing?\" |\n| `grafana_annotate` | Create or query event annotations on dashboards | \"Mark that deployment on my dashboard\" |\n| `grafana_push_metrics` | Push custom data (calendar, git, fitness, finance) via OTLP | \"Track my daily steps in Grafana\" |\n| `grafana_explain_metric` | Get metric context: current value, trend, stats, metadata | \"Why did my bill spike?\" |\n| `grafana_security_check` | Run 6 parallel security checks → threat level report | \"Am I being attacked?\" |\n| `grafana_query_traces` | Run TraceQL queries against Tempo; search traces or get full trace by ID | \"Find slow traces\" / \"Show trace for session X\" |\n| `grafana_investigate` | Multi-signal investigation triage with hypothesis generation | \"Investigate this alert\" / \"What's wrong?\" / \"Root cause\" |\n| `alloy_pipeline` | Create and manage Alloy data collection pipelines (29 recipes, 7 actions) | \"Monitor my Postgres DB\" / \"Collect Docker logs\" / \"Pipeline status\" |\n\n---\n\n## Alloy Pipeline Management\n\nThe `alloy_pipeline` tool manages [Grafana Alloy](https://grafana.com/docs/alloy/latest/) data collection pipelines. Alloy is Grafana Labs' open-source telemetry collector -- it gets your data INTO Grafana by collecting metrics, logs, traces, and profiles from your infrastructure and forwarding them to the LGTM stack.\n\n### How It Works\n\n1. **Describe what you want to monitor** -- \"monitor my Postgres database\", \"collect Docker logs\", \"set up OTLP trace ingestion\"\n2. **The agent selects a recipe** -- 29 pre-built templates handle common scenarios with validation, credential management, and sample queries\n3. **Config is generated and hot-reloaded** -- An Alloy `.alloy` config file is written and Alloy picks it up via API reload (no restart needed)\n4. **Data flows automatically** -- The agent verifies data flow and provides sample queries and suggested next steps (dashboards, alerts)\n\nFor custom patterns not covered by recipes, the agent can write raw Alloy River config directly.\n\n### 7 Actions\n\n| Action | What It Does |\n|--------|-------------|\n| `create` | Deploy a new pipeline from a recipe or raw config |\n| `list` | Show all managed pipelines with status |\n| `update` | Change pipeline parameters or replace raw config |\n| `delete` | Remove a pipeline and its config file |\n| `recipes` | Browse the recipe catalog by category |\n| `status` | Check component health and data flow for a pipeline |\n| `diagnose` | Full system check: Alloy connectivity, all pipeline health, config drift, orphan files |\n\n### Recipe Categories (29 Recipes)\n\n| Category | Count | Examples |\n|----------|-------|---------|\n| **Metrics** | 11 | scrape-endpoint, node-exporter, postgres-exporter, mysql-exporter, redis-exporter, kubernetes-pods |\n| **Logs** | 10 | docker-logs, file-logs, syslog, kubernetes-logs, kafka-logs, faro-frontend |\n| **Traces** | 4 | otlp-receiver, application-traces, span-metrics, service-graph |\n| **Infrastructure** | 3 | docker-metrics, elasticsearch-exporter, kafka-exporter |\n| **Profiling** | 1 | continuous-profiling (Pyroscope) |\n\nUse `alloy_pipeline` with action `recipes` to browse the full catalog with required parameters and descriptions.\n\n### Credential Handling\n\nRecipes that require credentials (database passwords, API keys) use Alloy's `sys.env()` function -- secrets are referenced as environment variables, never written to config files. The tool response includes `envVarsRequired` listing which variables must be set where Alloy runs.\n\n> **Important:** Alloy config reload is atomic. If a credential recipe's env vars aren't set, the reload failure blocks ALL managed pipelines until the env vars are set or the pipeline is deleted. The agent will warn you and ask if you have the credentials ready before creating such pipelines.\n\n### Workflow Integration\n\nAfter creating a pipeline, the tool provides `suggestedWorkflow` with concrete next-step tool calls:\n\n1. `alloy_pipeline` action `status` -- verify data flow (data takes ~15-20s to appear)\n2. `grafana_list_metrics` or `grafana_query_logs` -- discover collected data\n3. `grafana_create_dashboard` -- visualize the data\n4. `grafana_create_alert` -- set up monitoring\n\n> **Tip:** For real-world Alloy scenario examples (Postgres monitoring, Docker logs, Kubernetes, Kafka, OTLP tracing, and more), see the [alloy-scenarios](https://github.com/grafana/alloy/tree/main/example) repository.\n\n---\n\n## Dashboard Templates\n\nGrafana Lens includes 12 pre-built dashboard templates. Create any of them with:\n\n> \"Create a \\<template-name\\> dashboard\"\n\n### AI Observability (Tier 1-3 Drill-Down Hierarchy)\n\n| Template | Purpose | Key Panels |\n|----------|---------|------------|\n| `llm-command-center` | System overview (Tier 1) | Golden signals, live sessions, cost, cache efficiency, token rates |\n| `session-explorer` | Session debugging (Tier 2) | Per-session traces, LLM calls, tool calls, conversation flow |\n| `cost-intelligence` | Cost deep-dive (Tier 3a) | Spending trends, model attribution, cache savings, per-session cost |\n| `tool-performance` | Tool analytics (Tier 3b) | Tool leaderboard, latency ranking, error rates, execution traces |\n| `sre-operations` | SRE operations (Tier 3c) | Queue health, webhook errors, stuck sessions, tool loops, context pressure |\n| `genai-observability` | Industry-standard gen_ai | OpenTelemetry gen_ai metrics: token analytics, LLM performance, traces |\n| `security-overview` | Security monitoring | Webhook errors, injection signals, session anomalies, cost spikes |\n\n### Infrastructure & Generic\n\n| Template | Purpose | Key Panels |\n|----------|---------|------------|\n| `node-exporter` | System health | CPU, memory, disk, network (Prometheus node-exporter metrics) |\n| `http-service` | Web service monitoring | HTTP request rate, error rate, latency (RED signals) |\n| `metric-explorer` | Single metric deep-dive | Interactive exploration of any metric from a dropdown |\n| `multi-kpi` | KPI overview | 4-metric side-by-side comparison |\n| `weekly-review` | Weekly trends | 2-metric trends with external data table |\n\nAll AI templates use **Grafana template variables** for dropdown selectors (`$prometheus`, `$loki`, `$tempo`, `$model`, `$provider`, etc.) and **stable UIDs** for cross-dashboard drill-down navigation.\n\n---\n\n## Security Monitoring\n\nPrompt injection is the [#1 risk in OWASP's Top 10 for LLM Applications](https://genai.owasp.org/llmrisk/llm01-prompt-injection/). Grafana Lens provides detection-only security monitoring — it observes and reports, but never blocks or terminates operations.\n\n### Security Check (`grafana_security_check`)\n\nRuns 6 PromQL queries in parallel using `Promise.allSettled` (a single failing check won't break the others):\n\n| Check | What It Detects | Green | Yellow | Red |\n|-------|----------------|-------|--------|-----|\n| `webhook_error_ratio` | Failing webhook deliveries | < 20% | 20–50% | ≥ 50% |\n| `cost_anomaly` | Unusual daily spending | < $10 | $10–$50 | ≥ $50 |\n| `tool_loops` | Agent stuck in repetitive tool calls | 0 | 1–2 | ≥ 3 |\n| `injection_signals` | Prompt injection patterns detected | 0 | 1–4 | ≥ 5 |\n| `session_enumeration` | Unusual session volume (brute-force) | < 50/hr | 50–200/hr | ≥ 200/hr |\n| `stuck_sessions` | Sessions not progressing | 0 | 1–2 | ≥ 3 |\n\nReturns an overall threat level (`green`, `yellow`, or `red`) with suggested investigation actions for each finding.\n\n**Honest limitations:** Authentication failures are invisible to this tool — OpenClaw's auth middleware emits no diagnostic events for failed auth attempts. Monitor gateway access logs separately.\n\n### Security Dashboard\n\nUse the `security-overview` template for a persistent visual dashboard with 15 panels covering all security signals.\n\n### What's Being Monitored\n\nGrafana Lens automatically collects security-relevant metrics from OpenClaw's lifecycle:\n\n- **Prompt injection detection** — 12 regex patterns scanning LLM inputs (configurable via `otlp.captureContent`)\n- **Tool error classification** — Categorizes tool failures as network, filesystem, timeout, or other\n- **Session anomalies** — Unique session sliding window (1-hour) detects enumeration attempts\n- **Gateway restarts** — Tracks infrastructure availability\n- **Session resets** — Monitors forced context wipes\n\n---\n\n## SRE Investigation\n\nGrafana Lens includes purpose-built investigation capabilities for diagnosing alerts, anomalies, and outages.\n\n### Multi-Signal Triage (`grafana_investigate`)\n\nUse as the **first step** for any \"what's wrong?\" question. The tool:\n\n1. **Auto-discovers** Prometheus, Loki, and Tempo datasources (graceful degradation if Loki/Tempo unavailable)\n2. **Gathers signals in parallel** using `Promise.allSettled`:\n   - **Metrics** — Focus metric current value + trend, anomaly z-score, RED signals (rate, error rate, p95 latency)\n   - **Logs** — Volume, severity breakdown, top error patterns, sample errors\n   - **Traces** — Error traces, slow traces (>10s)\n   - **Context** — Recent annotations, active alerts\n3. **Generates hypotheses** — Each includes an `evidence` summary, `confidence` level, and a `testWith` field with the exact tool name and parameters for follow-up\n\n### Anomaly Scoring (`grafana_explain_metric`)\n\nFor 24-hour period queries on plain metric names, `grafana_explain_metric` now includes:\n\n- **Anomaly z-score** — Current value compared against the 7-day baseline (avg ± stddev). Severity levels: `normal` (<1.5σ), `mild` (1.5–2σ), `significant` (2–3σ), `critical` (>3σ)\n- **Seasonality comparison** — Value vs 1 day ago and vs 7 days ago, with change percentages\n\n### Alert Fatigue Detection (`grafana_check_alerts`)\n\nUse `action: \"analyze\"` to audit alert rule health:\n\n- **Always-firing** — Rules firing >24 hours, suggesting thresholds need adjustment\n- **Flapping** — Rules in error/nodata state, indicating query or datasource issues\n- **Overall health** — `healthy`, `moderate_fatigue`, or `severe_fatigue` with specific suggestions\n\n---\n\n## Custom Metrics (Data Observatory)\n\nPush any external data into Grafana via the `grafana_push_metrics` tool — no code required, just ask your agent:\n\n> \"Track my daily steps in Grafana\"\n> \"Push my git commit count for today\"\n> \"Record my morning weight\"\n\n### How It Works\n\n1. **Register** a metric (or let auto-registration handle it):\n   - Metric names are auto-prefixed with `openclaw_ext_` if not already\n   - Types: `counter` (ever-increasing) or `gauge` (current value)\n\n2. **Push** values with optional labels and timestamps:\n   - Data is pushed immediately via OTLP — no scraping delay\n   - Historical data supported via the `timestamp` parameter\n\n3. **Visualize** — create a dashboard or query the metric directly\n\n### Naming Rules\n\n- All custom metrics use the `openclaw_ext_` prefix\n- Counter metrics get a `_total` suffix in Prometheus (e.g., `openclaw_ext_steps` → `openclaw_ext_steps_total`)\n- Label names follow Prometheus conventions (letters, digits, underscores only)\n\n### Cardinality Limits\n\n| Limit | Default | Purpose |\n|-------|---------|---------|\n| Max metrics | 100 | Prevents unbounded metric growth |\n| Max labels per metric | 5 | Limits label key cardinality |\n| Max label values per metric | 50 | Limits unique label combinations |\n\nDefinitions persist across restarts in `${stateDir}/custom-metrics.json`.\n\n---\n\n## Architecture\n\nGrafana Lens is a **self-contained OpenClaw plugin** — all Grafana interaction is handled by the bundled `GrafanaClient`. No external MCP servers, no additional infrastructure beyond the LGTM stack.\n\n```\n┌─────────────────────────────────────────────────────┐\n│  OpenClaw Agent                                     │\n│  (processes messages, invokes tools)                │\n│                                                     │\n│  ┌─────────────────────────────────────────────┐    │\n│  │  Grafana Lens Plugin                        │    │\n│  │  • 18 Agent Tools                           │    │\n│  │  • MetricsCollector Service                 │    │\n│  │  • AlertWebhook Service                     │    │\n│  │  • AlloyPipeline Service                    │    │\n│  │  • LifecycleTelemetry (16 hooks)            │    │\n│  │  • Bundled GrafanaClient (REST API)         │    │\n│  │  • Bundled AlloyClient (REST API)           │    │\n│  └────────────┬──────────────┬─────────────────┘    │\n│               │              │                      │\n└───────────────┼──────────────┼──────────────────────┘\n                │              │\n    ┌───────────┴──────┐  ┌────┴──────────────────┐\n    │ OTLP HTTP Push   │  │ Alloy HTTP API        │\n    │ (:4318)          │  │ (:12345)              │\n    │  /v1/metrics     │  │  /-/reload            │\n    │  /v1/logs        │  │  /api/v0/components   │\n    │  /v1/traces      │  └────┬──────────────────┘\n    └───────┬──────────┘       │\n            │       ┌──────────┴──────────────────┐\n            │       │ Grafana Alloy               │\n            │       │ (data collection)           │\n            │       │ Reads config.d/*.alloy      │\n            │       │ Hot-reload via API           │\n            │       │ Scrapes, tails, receives     │\n            │       └──────────┬──────────────────┘\n            │                  │ remote_write / loki.write / otel\n┌───────────┴──────────────────┴──────────────────────┐\n│  LGTM Stack                                         │\n│  ┌──────────────┐  ┌──────┐  ┌───────┐             │\n│  │ Prometheus   │  │ Loki │  │ Tempo │             │\n│  │ (:9090)      │  │      │  │       │             │\n│  │ Metrics      │  │ Logs │  │Traces │             │\n│  └──────┬───────┘  └──┬───┘  └──┬────┘             │\n│         └──────────────┼────────┬┘                  │\n│                   ┌────┴────────┴───┐               │\n│                   │   Grafana       │               │\n│                   │   (:3000)       │               │\n│                   │   Dashboards    │               │\n│                   └─────────────────┘               │\n└─────────────────────────────────────────────────────┘\n```\n\n### Key Design Decisions\n\n| Decision | Choice | Why |\n|----------|--------|-----|\n| Everything is tools | 18 composable tools, no background automation | Agent decides when to act; Grafana handles scheduled work |\n| Self-contained | Bundled GrafanaClient (REST API) | No external MCP servers or dependencies |\n| OTLP push | Push-based metrics, logs, traces | No scraping delay — data available immediately |\n| General-purpose | Works with ANY Grafana datasource | Not limited to `openclaw_lens_*` metrics |\n| Alert provenance | `X-Disable-Provenance` header | Agent-created alert rules remain editable in Grafana UI |\n| Panel rendering | Grafana Image Renderer plugin | Persistent dashboards + three-tier fallback (PNG → snapshot → deep link) |\n| Secret redaction | Auto-strips API keys before OTLP export | Prevents token leakage into observability pipelines |\n| Alloy hot-reload | Write config file, POST `/-/reload` | Atomic config updates; failed reloads keep previous config running |\n\n### Relationship with diagnostics-otel\n\nOpenClaw ships a built-in [`diagnostics-otel`](https://github.com/nicepkg/openclaw) extension that converts diagnostic events into basic counters and histograms (token usage, cost, webhook durations, message processing). Grafana Lens does **not** replace it — they run side by side and complement each other:\n\n| Capability | diagnostics-otel | Grafana Lens |\n|-----------|-----------------|-------------|\n| Token/cost counters | Yes | Replicated (self-sufficient templates) |\n| Session-scoped traces (gen_ai conventions) | No | Yes — hierarchical `invoke_agent → chat → execute_tool` |\n| Structured logs → Loki | No | Yes — diagnostic events, LLM I/O, app logs |\n| Log-to-trace correlation | No | Yes — `trace_id` in log records for Loki → Tempo click-through |\n| Security monitoring | No | Yes — prompt injection, tool error classification, cost anomalies |\n| Operational gauges | No | Yes — active sessions, queue depth, context pressure, stuck sessions |\n| Secret redaction | No | Yes — auto-strips tokens before OTLP export |\n| Content capture controls | No | Yes — configurable prompt/completion logging with truncation |\n\n**Why not just use diagnostics-otel?** It only subscribes to diagnostic events and emits counters — it has no lifecycle hooks, no tracing, no log forwarding, and no security signals. Grafana Lens registers 16 lifecycle hooks (`session_start`, `llm_input`, `after_tool_call`, etc.) to build session-scoped traces and rich log records that diagnostics-otel was never designed to provide.\n\n**Why not extend diagnostics-otel?** It uses `NodeSDK` with global OTel providers (`setGlobalMeterProvider()`). Grafana Lens uses local `MeterProvider`, `LoggerProvider`, and `BasicTracerProvider` instances to avoid conflicts — both extensions coexist safely.\n\n---\n\n## Observability Deep Dive\n\n### Three Pillars — OTLP Push\n\nAll telemetry is pushed via OTLP HTTP to the collector (default `localhost:4318`). No Prometheus scraping, no `/metrics` endpoint.\n\n| Signal | Destination | Content |\n|--------|-------------|---------|\n| **Metrics** | Prometheus (via OTel Collector) | Agent gauges, counters, histograms — token usage, cost, session state, security signals |\n| **Logs** | Loki | Diagnostic events, app logs, LLM inputs/outputs (with redaction), security events |\n| **Traces** | Tempo | Session-scoped hierarchical traces following gen_ai semantic conventions |\n\n### Trace Hierarchy\n\nEach agent session produces a trace tree:\n\n```\ninvoke_agent openclaw          (root span)\n├── chat claude-3-opus         (LLM call)\n├── execute_tool grafana_query (tool execution)\n├── chat claude-3-opus         (next LLM turn)\n├── execute_tool grafana_create_dashboard\n├── openclaw.compaction        (context compression)\n└── openclaw.agent.end         (session close)\n```\n\nSpan names follow the [OpenTelemetry gen_ai semantic conventions](https://opentelemetry.io/docs/specs/semconv/gen-ai/): `{operation} {model/provider}`.\n\n### Log-to-Trace Correlation\n\nSpan-producing events include a `trace_id` attribute in their log records. In Grafana, this enables click-through from Loki logs directly to the corresponding Tempo trace — useful for debugging specific LLM calls or tool failures.\n\n### Distributed Trace Queries\n\nThe `grafana_query_traces` tool queries Tempo directly using TraceQL:\n\n- **Search by attributes** — `{ span.service.name = \"openclaw\" && duration > 5s }` finds slow spans\n- **Search by status** — `{ status = error }` finds failed operations\n- **Get trace by ID** — Retrieves the full span tree for a specific trace, with hierarchical parent-child relationships\n- **Log-to-trace click-through** — Span-producing events include `trace_id` in Loki log records, enabling one-click navigation from a log line to the full Tempo trace in Grafana\n\nThe tool handles both OTLP JSON and Tempo's protobuf-JSON response formats (base64-encoded IDs, string enum kinds/status codes) transparently.\n\n### Secret Redaction\n\nWhen `otlp.redactSecrets` is enabled (default: `true`), all telemetry data is automatically scanned for sensitive tokens before OTLP export:\n\n- GitHub tokens (`ghp_*`, `github_pat_*`)\n- Slack tokens (`xoxb-*`, `xoxp-*`)\n- API keys (`sk-*`, `sk-ant-*`, `AIza*`)\n- Grafana tokens (`glsa_*`)\n- Bearer tokens, PEM private keys, and more\n\nTokens are redacted to `${first6}...${last4}` format (e.g., `glsa_Ab...x9Zq`).\n\n### Lifecycle Hooks\n\nGrafana Lens registers 16 lifecycle hooks into OpenClaw for deep observability:\n\n`session_start`, `session_end`, `llm_input`, `llm_output`, `agent_end`, `message_received`, `message_sent`, `before_compaction`, `after_compaction`, `subagent_spawned`, `subagent_ended`, `before_tool_call`, `after_tool_call`, `before_reset`, `gateway_start`, `gateway_stop`\n\nThese hooks power the session-scoped traces, gen_ai standard metrics, security signals, and SRE operational data.\n\n### Content Capture Controls\n\nFor privacy-sensitive deployments, configure what's included in telemetry:\n\n```jsonc\n{\n  \"otlp\": {\n    \"captureContent\": false,    // disable prompt/completion logging\n    \"redactSecrets\": true,      // auto-strip tokens (default: true)\n    \"contentMaxLength\": 500,    // truncate content fields\n    \"forwardAppLogs\": false     // disable app log forwarding to Loki\n  }\n}\n```\n\n---\n\n## Configuration Reference\n\n### All Config Options\n\n| Key | Type | Default | Description |\n|-----|------|---------|-------------|\n| `grafana.url` | string | — | Grafana instance URL (required) |\n| `grafana.apiKey` | string | — | Service account token (required) |\n| `grafana.orgId` | number | `1` | Organization ID |\n| `metrics.enabled` | boolean | `true` | Enable OTLP telemetry collection |\n| `otlp.endpoint` | string | `http://localhost:4318/v1/metrics` | OTLP collector endpoint |\n| `otlp.headers` | object | `{}` | Custom HTTP headers for OTLP export |\n| `otlp.exportIntervalMs` | number | `15000` | Metric export interval (ms) |\n| `otlp.logs` | boolean | `true` | Push logs to Loki via OTLP |\n| `otlp.traces` | boolean | `true` | Push traces to Tempo via OTLP |\n| `otlp.captureContent` | boolean | `true` | Include prompts/completions in telemetry |\n| `otlp.contentMaxLength` | number | `2000` | Max chars for content fields |\n| `otlp.forwardAppLogs` | boolean | `true` | Forward OpenClaw app logs to Loki |\n| `otlp.appLogMinSeverity` | string | `\"debug\"` | Min log severity (trace/debug/info/warn/error/fatal) |\n| `otlp.redactSecrets` | boolean | `true` | Auto-redact API keys/tokens before OTLP export |\n| `proactive.enabled` | boolean | `false` | Enable Grafana alert webhook handler |\n| `proactive.webhookPath` | string | `\"/grafana-lens/alerts\"` | HTTP path for alert webhooks |\n| `proactive.costAlertThreshold` | number | `5.0` | Daily cost alert threshold (USD) |\n| `customMetrics.enabled` | boolean | `true` | Allow custom metric push |\n| `customMetrics.maxMetrics` | number | `100` | Max custom metric definitions |\n| `customMetrics.maxLabelsPerMetric` | number | `5` | Max label keys per metric |\n| `customMetrics.maxLabelValues` | number | `50` | Max unique label combos per metric |\n| `customMetrics.defaultTtlDays` | number | `null` | Optional auto-expiry (days) |\n| `alloy.enabled` | boolean | `false` | Enable Alloy pipeline management |\n| `alloy.url` | string | `\"http://localhost:12345\"` | Alloy HTTP API URL |\n| `alloy.configDir` | string | -- | Directory for pipeline config files (required when enabled) |\n| `alloy.filePrefix` | string | `\"lens-\"` | Prefix for managed config file names |\n| `alloy.maxPipelines` | number | `20` | Maximum managed pipelines |\n| `alloy.lgtm.prometheusRemoteWriteUrl` | string | `\"http://localhost:9009/api/prom/push\"` | Prometheus remote write endpoint (used in generated configs) |\n| `alloy.lgtm.lokiUrl` | string | `\"http://localhost:3100/loki/api/v1/push\"` | Loki push endpoint |\n| `alloy.lgtm.otlpEndpoint` | string | (derived from `otlp.endpoint`) | OTLP endpoint for trace/metrics export |\n| `alloy.lgtm.pyroscopeUrl` | string | `\"http://localhost:4040\"` | Pyroscope endpoint (profiling recipes) |\n\n---\n\n## Development\n\n```bash\n# Install dependencies\nnpm install\n\n# Run unit tests\nnpm test\n\n# TypeScript type-check (no emit)\nnpm run typecheck\n\n# Link into OpenClaw for local development\nopenclaw plugins install -l ~/workspace/grafana-lens\n\n# Restart gateway after changes\nopenclaw gateway restart\n```\n\n### Project Structure\n\n```\ngrafana-lens/\n├── index.ts                          # Plugin entry point\n├── package.json                      # Package metadata\n├── openclaw.plugin.json              # Plugin manifest with config schema\n├── src/\n│   ├── config.ts                     # Config parsing and validation\n│   ├── grafana-client.ts             # Bundled Grafana REST API client\n│   ├── metric-definitions.ts         # Shared metric registry\n│   ├── tools/                        # 18 agent tools\n│   │   ├── query.ts                  # grafana_query\n│   │   ├── query-logs.ts             # grafana_query_logs\n│   │   ├── create-dashboard.ts       # grafana_create_dashboard\n│   │   ├── update-dashboard.ts       # grafana_update_dashboard\n│   │   ├── get-dashboard.ts          # grafana_get_dashboard\n│   │   ├── search.ts                 # grafana_search\n│   │   ├── share-dashboard.ts        # grafana_share_dashboard\n│   │   ├── create-alert.ts           # grafana_create_alert\n│   │   ├── check-alerts.ts           # grafana_check_alerts\n│   │   ├── annotate.ts               # grafana_annotate\n│   │   ├── explore-datasources.ts    # grafana_explore_datasources\n│   │   ├── list-metrics.ts           # grafana_list_metrics\n│   │   ├── push-metrics.ts           # grafana_push_metrics\n│   │   ├── explain-metric.ts         # grafana_explain_metric\n│   │   ├── security-check.ts         # grafana_security_check\n│   │   ├── query-traces.ts           # grafana_query_traces\n│   │   ├── investigate.ts            # grafana_investigate\n│   │   └── alloy-pipeline.ts         # alloy_pipeline\n│   ├── alloy/                        # Alloy pipeline management\n│   │   ├── alloy-client.ts           # Alloy HTTP API client\n│   │   ├── pipeline-store.ts         # Pipeline state persistence\n│   │   ├── pipeline-helpers.ts       # Config generation helpers\n│   │   ├── types.ts                  # Shared Alloy types\n│   │   └── recipes/                  # 29 pipeline recipe definitions\n│   │       └── catalog.ts            # Recipe registry\n│   ├── services/\n│   │   ├── metrics-collector.ts      # Diagnostic event → OTLP push\n│   │   ├── alert-webhook.ts          # Grafana alert webhook handler\n│   │   ├── alloy-service.ts          # Alloy pipeline lifecycle management\n│   │   ├── otel-metrics.ts           # MeterProvider (local, no global)\n│   │   ├── otel-logs.ts              # LoggerProvider\n│   │   ├── otel-traces.ts            # TracerProvider\n│   │   ├── lifecycle-telemetry.ts    # Session-scoped gen_ai traces + metrics\n│   │   ├── custom-metrics-store.ts   # Custom metric definitions + state\n│   │   └── redact.ts                 # Secret redaction utility\n│   └── templates/                    # 12 dashboard JSON templates\n│       ├── llm-command-center.json\n│       ├── session-explorer.json\n│       ├── cost-intelligence.json\n│       ├── tool-performance.json\n│       ├── sre-operations.json\n│       ├── genai-observability.json\n│       ├── security-overview.json\n│       ├── node-exporter.json\n│       ├── http-service.json\n│       ├── metric-explorer.json\n│       ├── multi-kpi.json\n│       └── weekly-review.json\n└── skills/\n    └── SKILL.md                      # Agent skill file\n```\n\n---\n\n## License\n\nMIT\n\n---\n\n> **Disclaimer:** Grafana Lens is a community-built OpenClaw plugin. It is not developed, maintained, or endorsed by Grafana Labs. Grafana, Prometheus, Loki, Tempo, and Mimir are trademarks of Grafana Labs.\n\nFile v0.5.0:_meta.json\n\n{\n  \"ownerId\": \"kn785ebhh3a5v4pqymw1yhz6wh8292pc\",\n  \"slug\": \"grafana-lens\",\n  \"version\": \"0.5.0\",\n  \"publishedAt\": 1775407137121\n}\n\nFile v0.5.0:references/agent-metrics.md\n\n# Agent Metrics Reference\n\nMetrics come from two sources — both push via OTLP to the same collector/Mimir instance and are queryable in Grafana.\n\n## Core Agent Telemetry (from diagnostics-otel)\n\nThese are published by OpenClaw's built-in `diagnostics-otel` extension. Grafana Lens dashboards query them but does not collect them.\n\n| Prometheus Name | Type | Labels | Source Event |\n|----------------|------|--------|-------------|\n| `openclaw_tokens_total` | counter | `openclaw_token`, `openclaw_model`, `openclaw_provider`, `openclaw_channel` | `model.usage` |\n| `openclaw_cost_usd_total` | counter | `openclaw_model`, `openclaw_provider`, `openclaw_channel` | `model.usage` |\n| `openclaw_run_duration_ms_milliseconds` | histogram | `openclaw_model`, `openclaw_provider`, `openclaw_channel` | `model.usage` |\n| `openclaw_context_tokens` | histogram | `openclaw_context` (limit/used), `openclaw_model`, `openclaw_provider`, `openclaw_channel` | `model.usage` |\n| `openclaw_message_processed_total` | counter | `openclaw_outcome`, `openclaw_channel` | `message.processed` |\n| `openclaw_message_duration_ms_milliseconds` | histogram | `openclaw_outcome`, `openclaw_channel` | `message.processed` |\n| `openclaw_message_queued_total` | counter | `openclaw_channel`, `openclaw_source` | `message.queued` |\n| `openclaw_webhook_received_total` | counter | `openclaw_channel`, `openclaw_webhook` | `webhook.received` |\n| `openclaw_webhook_error_total` | counter | `openclaw_channel`, `openclaw_webhook` | `webhook.error` |\n| `openclaw_webhook_duration_ms_milliseconds` | histogram | `openclaw_channel`, `openclaw_webhook` | `webhook.processed` |\n| `openclaw_queue_depth` | histogram | `openclaw_lane`, `openclaw_channel` | `message.queued`, `queue.lane.*`, `heartbeat` |\n| `openclaw_queue_wait_ms_milliseconds` | histogram | `openclaw_lane` | `queue.lane.dequeue` |\n| `openclaw_queue_lane_enqueue_total` | counter | `openclaw_lane` | `queue.lane.enqueue` |\n| `openclaw_queue_lane_dequeue_total` | counter | `openclaw_lane` | `queue.lane.dequeue` |\n| `openclaw_session_state_total` | counter | `openclaw_state`, `openclaw_reason` | `session.state` |\n| `openclaw_session_stuck_total` | counter | `openclaw_state` | `session.stuck` |\n| `openclaw_session_stuck_age_ms_milliseconds` | histogram | `openclaw_state` | `session.stuck` |\n| `openclaw_run_attempt_total` | counter | `openclaw_attempt` | `run.attempt` |\n\n**Label names use underscores** (OTel dots → Prometheus underscores): `openclaw.model` → `openclaw_model`.\n\n**OTel unit suffix**: Histograms declared with `unit: \"ms\"` get `_milliseconds` appended in Prometheus (OTLP-to-Prometheus translation). So the OTel instrument `openclaw_run_duration_ms` becomes `openclaw_run_duration_ms_milliseconds_bucket` in PromQL. All PromQL in this doc uses the physical Prometheus names.\n\n**Label value reference**:\n- `openclaw_token`: `input`, `output`, `cache_read`, `cache_write`, `prompt`, `total`\n- `openclaw_context`: `limit`, `used`\n- `openclaw_outcome`: `completed`, `skipped`, `error`\n\n## Operational Gauges (from grafana-lens)\n\nThese are unique to Grafana Lens — current-state snapshots and user data that diagnostics-otel doesn't provide.\n\n| Metric | Type | Labels | Source |\n|--------|------|--------|--------|\n| `openclaw_lens_sessions_active` | gauge (UpDownCounter) | `state` | `session.state` events |\n| `openclaw_lens_queue_depth` | gauge | — | `session.state` + `diagnostic.heartbeat` |\n| `openclaw_lens_context_tokens` | gauge | `type` (limit/used) | `model.usage` events |\n| `openclaw_lens_daily_cost_usd` | gauge | — | `model.usage` + midnight reset |\n| `openclaw_lens_sessions_active_snapshot` | gauge | — | `diagnostic.heartbeat` (ground-truth cross-check) |\n| `openclaw_lens_sessions_stuck` | gauge | — | `session.stuck` events |\n| `openclaw_lens_stuck_session_max_age_ms` | gauge | — | `session.stuck` events |\n| `openclaw_lens_cache_read_ratio` | gauge | — | `model.usage` (cacheRead / total input) |\n| `openclaw_lens_tool_loops_active` | gauge | `level` (warning/critical) | `tool.loop` events |\n| `openclaw_lens_queue_lane_depth` | gauge | `lane` | `queue.lane.enqueue/dequeue` |\n| `openclaw_lens_alert_webhooks_received` | gauge | `status` (firing/resolved) | Alert webhook subsystem |\n| `openclaw_lens_alert_webhooks_pending` | gauge | — | Alert webhook subsystem |\n| `openclaw_lens_custom_metrics_pushed_total` | counter | — | `grafana_push_metrics` usage |\n\nCustom metrics (`openclaw_ext_*`) — see [external-data.md](external-data.md).\n\n## Security Metrics (from grafana-lens)\n\nThese observe security-relevant signals from lifecycle hooks and diagnostic events. Detection-only — never blocks or terminates sessions.\n\n| Metric | Type | Labels | Source Hook/Event |\n|--------|------|--------|-------------------|\n| `openclaw_lens_gateway_restarts` | counter | — | `gateway_start` hook |\n| `openclaw_lens_session_resets` | counter | `reason` | `before_reset` hook |\n| `openclaw_lens_tool_error_classes` | counter | `tool`, `error_class` (network/filesystem/timeout/other) | `after_tool_call` hook (when `event.error` set) |\n| `openclaw_lens_prompt_injection_signals` | counter | `detector` (input_scan/tool_loop) | `llm_input` hook (pattern scan) + `tool.loop` diagnostic event |\n| `openclaw_lens_unique_sessions_1h` | gauge | — | `session_start` hook (1h sliding window) |\n\n**Limitation — auth failures are invisible**: OpenClaw's auth middleware (`gateway/auth.ts`) returns `{ ok: false }` but emits **zero** diagnostic events and **zero** log records. Auth failures (bad tokens, brute-force, rate-limiter lockouts) are completely silent from Grafana Lens's telemetry pipeline. Security monitoring relies on observable signals (webhook errors, cost spikes, prompt injection patterns, session anomalies) but cannot detect silent auth-layer attacks. Monitor gateway-level logs outside OpenClaw for auth visibility.\n\n**Error classification labels** (`error_class` on `tool_error_classes`):\n- `network`: ECONNREFUSED, ETIMEDOUT, ENOTFOUND, fetch failures\n- `filesystem`: ENOENT, EACCES, path/directory/traversal errors\n- `timeout`: Timeout/timed out errors\n- `other`: Everything else\n\n**Prompt injection detection** (`detector` on `prompt_injection_signals`):\n- `input_scan`: Pattern match on LLM input (only when `captureContent` config enabled — respects privacy)\n- `tool_loop`: From `tool.loop` diagnostic event (tool stuck in infinite loops — prompt injection indicator)\n\n## Self-Sufficient Counters/Histograms (from grafana-lens)\n\nThese replicate key counters/histograms from diagnostics-otel so dashboards work without the diagnostics-otel extension. Labels use clean names (no `openclaw_` prefix).\n\n| Metric | Type | Labels | Source Event |\n|--------|------|--------|-------------|\n| `openclaw_lens_tokens_total` | counter | `token` (input/output/cacheRead/cacheWrite), `provider`, `model` | `model.usage` |\n| `openclaw_lens_messages_processed_total` | counter | `outcome` (completed/skipped/error), `channel` | `message.processed` |\n| `openclaw_lens_webhook_received_total` | counter | `channel`, `update_type` | `webhook.received` |\n| `openclaw_lens_webhook_error_total` | counter | `channel`, `update_type` | `webhook.error` |\n| `openclaw_lens_webhook_duration_ms_bucket` | histogram | `channel`, `update_type` | `webhook.processed` |\n| `openclaw_lens_queue_lane_enqueue_total` | counter | `lane` | `queue.lane.enqueue` |\n| `openclaw_lens_queue_lane_dequeue_total` | counter | `lane` | `queue.lane.dequeue` |\n| `openclaw_lens_queue_wait_ms_bucket` | histogram | `lane` | `queue.lane.dequeue` |\n\n**Note**: These use clean label names (`model`, `provider`, `channel`, `outcome`) — not the `openclaw_`-prefixed labels from diagnostics-otel. Dashboard templates query these `openclaw_lens_*` metrics exclusively.\n\n## gen_ai Standard Metrics (Grafana Cloud AI Observability compatible)\n\nThese follow the [OTel gen_ai semantic conventions](https://opentelemetry.io/docs/specs/semconv/gen-ai/) and are compatible with Grafana Cloud's AI Observability dashboards out of the box.\n\n| Metric | Type | Unit | Labels | Source |\n|--------|------|------|--------|--------|\n| `gen_ai_client_token_usage` | histogram | `{token}` | `gen_ai_operation_name`, `gen_ai_provider_name`, `gen_ai_token_type` (input/output/cache_read_input/cache_creation_input), `gen_ai_request_model` | `llm_output` hook |\n| `gen_ai_client_operation_duration` | histogram | `s` | `gen_ai_operation_name`, `gen_ai_provider_name`, `gen_ai_request_model` | `llm_input` + `llm_output` paired |\n\n**Note**: OTel dotted names become underscores in Prometheus. `gen_ai.client.token.usage` → `gen_ai_client_token_usage`. Duration uses explicit bucket boundaries: `[0.01, 0.02, 0.04, 0.08, 0.16, 0.32, 0.64, 1.28, 2.56, 5.12, 10.24, 20.48, 40.96, 81.92]` seconds.\n\n## Lifecycle Metrics (from grafana-lens hook telemetry)\n\nSession, compaction, subagent, and delivery metrics from OpenClaw plugin lifecycle hooks.\n\n| Metric | Type | Labels | Source Hook |\n|--------|------|--------|-------------|\n| `openclaw_lens_sessions_started_total` | counter | `type` (new/resumed) | `session_start` | **Note**: webchat sessions always resume (`type=resumed`); `type=new` only appears for brand-new sessions (no prior conversation). |\n| `openclaw_lens_session_duration_ms` | histogram | — | `session_end` |\n| `openclaw_lens_compactions_total` | counter | — | `after_compaction` |\n| `openclaw_lens_compaction_messages_removed` | histogram | — | `after_compaction` |\n| `openclaw_lens_subagents_spawned_total` | counter | `mode` (run/session) | `subagent_spawned` |\n| `openclaw_lens_sessions_completed_total` | counter | `outcome` (success/error) | `session_end` / `agent_end` |\n| `openclaw_lens_subagent_outcomes_total` | counter | `outcome`, `mode` | `subagent_ended` |\n| `openclaw_lens_subagent_duration_ms` | histogram | `mode` | `subagent_ended` (paired with `subagent_spawned`) |\n| `openclaw_lens_message_delivery_total` | counter | `channel`, `success` | `message_sent` | **Note**: only fires for webhook-based channels (Telegram, Slack, etc.). Webchat uses a different delivery path that bypasses the `message_sent` hook. |\n| `openclaw_lens_tool_calls_total` | counter | `tool`, `status` | `after_tool_call` |\n| `openclaw_lens_tool_duration_ms` | histogram | `tool` | `after_tool_call` |\n| `openclaw_lens_cost_by_model` | counter | `model`, `provider` | `model.usage` diagnostic event |\n| `openclaw_lens_session_message_types` | counter | `type` (user/assistant/tool_call/tool_result/error) | lifecycle hooks |\n| `openclaw_lens_cost_by_token_type` | counter | `token_type` (input/output/cache_read/cache_write), `model`, `provider` | `model.usage` diagnostic event |\n| `openclaw_lens_cache_savings_usd` | gauge | — | `model.usage` (accumulated cache read savings) |\n| `openclaw_lens_session_latency_avg_ms` | gauge | — | `llm_input`/`llm_output` paired (rolling average) |\n| `openclaw_lens_cache_token_ratio` | gauge | — | `model.usage` (cache tokens / all tokens) |\n\n## gen_ai Span Attributes (OTel Semantic Convention Compliance)\n\nLifecycle telemetry emits hierarchical spans following the [OTel gen_ai semantic conventions](https://opentelemetry.io/docs/specs/semconv/gen-ai/). These are visible in Tempo and Grafana Cloud AI Observability.\n\n### invoke_agent spans (root — one per session)\n\n| Attribute | Status | Value |\n|-----------|--------|-------|\n| `gen_ai.operation.name` | Required | `\"invoke_agent\"` |\n| `gen_ai.provider.name` | Required | `\"openclaw\"` |\n| `gen_ai.agent.name` | Recommended | `\"openclaw\"` |\n| `gen_ai.agent.id` | Recommended | `\"grafana-lens\"` |\n| `gen_ai.agent.version` | Recommended | Plugin version (e.g., `\"0.1.0\"`) |\n| `gen_ai.output.type` | Recommended | `\"text\"` |\n| `gen_ai.conversation.id` | Recommended | Session ID |\n| `openclaw.session.resumed_from` | Custom | Previous session ID (omitted if not resumed) |\n| `openclaw.session.message_count` | Custom | Total messages in session (set at session end) |\n| `openclaw.session.duration_ms` | Custom | Session duration in ms (set at session end) |\n| `openclaw.session.cost_usd` | Custom | Accumulated cost for this session (set at session end) |\n| `openclaw.session.total_input_tokens` | Custom | Accumulated input tokens (set at session end) |\n| `openclaw.session.total_output_tokens` | Custom | Accumulated output tokens (set at session end) |\n| `openclaw.session.total_cache_read_tokens` | Custom | Accumulated cache read tokens (set at session end) |\n| `openclaw.session.total_cache_write_tokens` | Custom | Accumulated cache write tokens (set at session end) |\n| `openclaw.session.messages.user` | Custom | User messages in session |\n| `openclaw.session.messages.assistant` | Custom | Assistant messages in session |\n| `openclaw.session.messages.tool_calls` | Custom | Tool calls in session |\n| `openclaw.session.messages.tool_results` | Custom | Tool results in session |\n| `openclaw.session.messages.errors` | Custom | Errors in session |\n| `openclaw.session.latency.avg_ms` | Custom | Average LLM call latency in session |\n| `openclaw.session.latency.p95_ms` | Custom | P95 LLM call latency in session |\n| `openclaw.session.latency.min_ms` | Custom | Min LLM call latency in session |\n| `openclaw.session.latency.max_ms` | Custom | Max LLM call latency in session |\n| `openclaw.session.tools.unique_count` | Custom | Number of unique tools used |\n| `openclaw.session.tools.total_calls` | Custom | Total tool calls in session |\n| `openclaw.session.tools.top` | Custom | Comma-separated top tools by usage |\n| `openclaw.session.cost.input` | Custom | Estimated input token cost (USD) |\n| `openclaw.session.cost.output` | Custom | Estimated output token cost (USD) |\n| `openclaw.session.cost.cache_read` | Custom | Estimated cache read cost (USD) |\n| `openclaw.session.cost.cache_write` | Custom | Estimated cache write cost (USD) |\n| `openclaw.session.cache_hit_ratio` | Custom | Cache hit ratio (0-1) |\n| `openclaw.session.cache_savings_usd` | Custom | Estimated cache savings (USD) |\n| `gen_ai.agent.id` | Recommended | Agent ID from session context |\n| `gen_ai.conversation.id` | Recommended | Session ID |\n| `gen_ai.conversation.parent_id` | Custom | Parent session ID (set on child subagent sessions via deferred linking) |\n| `gen_ai.provider.name` | Recommended | Primary provider used in session |\n| `gen_ai.request.model` | Recommended | Primary model used in session |\n| `openclaw.parent_session_id` | Custom | Parent session ID (subagent only, set via deferred linking) |\n| `openclaw.parent_session_key` | Custom | Parent session key (subagent only) |\n| `openclaw.parent_trace_id` | Custom | Trace ID of parent agent (for cross-trace correlation) |\n| `openclaw.is_subagent` | Custom | `true` if this session is a spawned subagent |\n| `openclaw.subagent.agent_id` | Custom | Agent ID of the subagent (set on child root span) |\n| `openclaw.subagent.label` | Custom | Subagent label (set on child root span) |\n| `openclaw.subagent.mode` | Custom | Subagent mode: `\"run\"` or `\"session\"` (set on child root span) |\n\n**Span links on child root span**: Link to parent spawn span with `openclaw.link.type=parent_agent` (cross-trace).\n\n### openclaw.subagent.spawn spans (long-lived — brackets subagent lifetime)\n\n| Attribute | Status | Value |\n|-----------|--------|-------|\n| `openclaw.subagent.agent_id` | Custom | Agent ID of the spawned subagent |\n| `openclaw.subagent.mode` | Custom | `\"run\"` or `\"session\"` |\n| `openclaw.subagent.label` | Custom | Human-readable label |\n| `openclaw.subagent.child_session_key` | Custom | Session key of the child agent |\n| `openclaw.subagent.thread_requested` | Custom | Whether a thread was requested |\n| `openclaw.subagent.child_trace_id` | Custom | Trace ID of child agent (set by deferred linking) |\n| `openclaw.subagent.child_session_id` | Custom | Session ID of child agent (set by deferred linking) |\n| `openclaw.subagent.target_kind` | Custom | Kind of subagent (set at end) |\n| `openclaw.subagent.reason` | Custom | End reason (set at end) |\n| `openclaw.subagent.outcome` | Custom | End outcome (set at end) |\n\n**Span links on spawn span**: Bidirectional link to child root span with `openclaw.link.type=child_agent` (cross-trace).\n\n**Span lifecycle**: Created on `subagent_spawned`, ended on `subagent_ended`. If child linking happens (via `onLlmInput`), span is enriched with `child_trace_id` and `child_session_id`.\n\n### chat spans (one per LLM call)\n\n| Attribute | Status | Value |\n|-----------|--------|-------|\n| `gen_ai.operation.name` | Required | `\"chat\"` |\n| `gen_ai.provider.name` | Required | Provider name (e.g., `\"anthropic\"`) |\n| `gen_ai.request.model` | Recommended | Requested model name |\n| `gen_ai.response.model` | Recommended | Actual model served (may differ from request) |\n| `gen_ai.response.finish_reasons` | Recommended | Array: `[\"stop\"]`, `[\"max_tokens\"]`, `[\"tool_calls\"]`, `[\"error\"]` |\n| `gen_ai.usage.input_tokens` | Recommended | Input token count |\n| `gen_ai.usage.output_tokens` | Recommended | Output token count |\n| `gen_ai.usage.cache_creation.input_tokens` | Custom | Cache write tokens |\n| `gen_ai.usage.cache_read.input_tokens` | Custom | Cache read tokens |\n\n### execute_tool spans (one per tool call)\n\n| Attribute | Status | Value |\n|-----------|--------|-------|\n| `gen_ai.operation.name` | Required | `\"execute_tool\"` |\n| `gen_ai.provider.name` | Required | `\"openclaw\"` |\n| `gen_ai.tool.name` | Recommended | Tool name (e.g., `\"grafana_query\"`) |\n| `gen_ai.tool.call.id` | Recommended | Tool call ID |\n| `gen_ai.tool.type` | Recommended | `\"function\"` |\n| `error.type` | Required on error | `\"tool_error\"` |\n\n### Error span attributes\n\n| Span Type | `error.type` Value |\n|-----------|--------------------|\n| `openclaw.agent.end` (with error) | `\"agent_error\"` |\n| `openclaw.message.sent` (with error) | `\"delivery_error\"` |\n| `openclaw.subagent.end` (with error) | `\"subagent_error\"` |\n| `execute_tool` (with error) | `\"tool_error\"` |\n\n## Common PromQL Expressions\n\n| Question | PromQL |\n|----------|--------|\n| Daily cost so far | `sum(increase(openclaw_lens_cost_by_model_total[1d])) or vector(0)` |\n| Daily cost (gauge) | `openclaw_lens_daily_cost_usd` |\n| Total cost all time | `sum(openclaw_lens_cost_by_model_total)` |\n| Token rate (5m) | `sum(rate(openclaw_lens_tokens_total[5m]))` |\n| Cost by model | `sum by (model) (openclaw_lens_cost_by_model_total)` |\n| P95 LLM latency | `histogram_quantile(0.95, sum(rate(gen_ai_client_operation_duration_seconds_bucket[5m])) by (le))` |\n| Error rate | `rate(openclaw_lens_messages_processed_total{outcome=\"error\"}[5m])` |\n| Active sessions | `sum(openclaw_lens_sessions_active)` |\n| Queue depth | `openclaw_lens_queue_depth` |\n| Context utilization % | `openclaw_lens_context_tokens{type=\"used\"} / openclaw_lens_context_tokens{type=\"limit\"} * 100` |\n| Webhook error rate | `rate(openclaw_lens_webhook_error_total[5m])` |\n| Cache hit rate | `openclaw_lens_cache_read_ratio` |\n| Stuck sessions | `openclaw_lens_sessions_stuck` |\n| Longest stuck session | `openclaw_lens_stuck_session_max_age_ms` |\n| Tool loops active | `sum(openclaw_lens_tool_loops_active)` |\n| Queue lane depth | `openclaw_lens_queue_lane_depth` |\n| Pending alerts | `openclaw_lens_alert_webhooks_pending` |\n| Custom metrics pushed | `rate(openclaw_lens_custom_metrics_pushed_total[5m])` |\n| Context window size (p95) | `histogram_quantile(0.95, sum(rate(openclaw_context_tokens_bucket[5m])) by (le))` |\n| P95 webhook latency | `histogram_quantile(0.95, sum(rate(openclaw_lens_webhook_duration_ms_bucket[5m])) by (le))` |\n| Message inflow rate | `sum(rate(openclaw_lens_messages_processed_total[5m])) by (channel)` |\n| Queue wait p95 | `histogram_quantile(0.95, sum(rate(openclaw_lens_queue_wait_ms_bucket[5m])) by (le))` |\n| Queue throughput | `sum(rate(openclaw_lens_queue_lane_dequeue_total[5m])) by (lane)` |\n| Queue enqueue vs dequeue | `sum(rate(openclaw_lens_queue_lane_enqueue_total[5m]))` vs `sum(rate(openclaw_lens_queue_lane_dequeue_total[5m]))` |\n| Stuck session rate | `rate(openclaw_session_stuck_total[5m])` |\n| P95 stuck age | `histogram_quantile(0.95, sum(rate(openclaw_session_stuck_age_ms_milliseconds_bucket[5m])) by (le))` |\n| Retry rate | `sum(rate(openclaw_run_attempt_total{openclaw_attempt!=\"1\"}[5m]))` |\n| Queue depth distribution | `histogram_quantile(0.95, sum(rate(openclaw_queue_depth_bucket[5m])) by (le))` |\n| gen_ai token usage (input) | `sum(rate(gen_ai_client_token_usage_bucket{gen_ai_token_type=\"input\"}[5m]))` |\n| gen_ai token usage (output) | `sum(rate(gen_ai_client_token_usage_bucket{gen_ai_token_type=\"output\"}[5m]))` |\n| gen_ai P95 LLM latency | `histogram_quantile(0.95, sum(rate(gen_ai_client_operation_duration_seconds_bucket[5m])) by (le))` |\n| gen_ai token usage by model | `sum by (gen_ai_request_model) (rate(gen_ai_client_token_usage_bucket[5m]))` |\n| gen_ai cache read tokens | `sum(rate(gen_ai_client_token_usage_sum{gen_ai_token_type=\"cache_read_input\"}[5m]))` |\n| gen_ai cache write tokens | `sum(rate(gen_ai_client_token_usage_sum{gen_ai_token_type=\"cache_creation_input\"}[5m]))` |\n| Cost by token type | `sum by (token_type) (increase(openclaw_lens_cost_by_token_type[$__range]))` |\n| Cost by model & provider | `sum by (model, provider) (increase(openclaw_lens_cost_by_model_total[$__range]))` |\n| Message types rate | `sum by (type) (rate(openclaw_lens_session_message_types[$__rate_interval]))` |\n| Cache savings USD | `openclaw_lens_cache_savings_usd` |\n| Cache token ratio | `openclaw_lens_cache_token_ratio` |\n| Avg LLM latency | `openclaw_lens_session_latency_avg_ms` |\n| Sessions started rate | `rate(openclaw_lens_sessions_started_total[5m])` |\n| P50 session duration | `histogram_quantile(0.5, sum(rate(openclaw_lens_session_duration_ms_bucket[5m])) by (le))` |\n| Compaction rate | `rate(openclaw_lens_compactions_total[5m])` |\n| Subagent spawn rate | `rate(openclaw_lens_subagents_spawned_total[5m])` |\n| Session completion rate | `rate(openclaw_lens_sessions_completed_total[5m])` |\n| Session success rate | `sum(rate(openclaw_lens_sessions_completed_total{outcome=\"success\"}[5m])) / sum(rate(openclaw_lens_sessions_completed_total[5m]))` |\n| P95 subagent duration | `histogram_quantile(0.95, sum(rate(openclaw_lens_subagent_duration_ms_bucket[5m])) by (le))` |\n| Subagent duration by mode | `histogram_quantile(0.95, sum by (le, mode) (rate(openclaw_lens_subagent_duration_ms_bucket[5m])))` |\n| Message delivery success rate | `sum(rate(openclaw_lens_message_delivery_total{success=\"true\"}[5m])) / sum(rate(openclaw_lens_message_delivery_total[5m]))` |\n| Prompt injection signals (1h) | `sum(increase(openclaw_lens_prompt_injection_signals_total[1h]))` |\n| Gateway restarts (24h) | `sum(increase(openclaw_lens_gateway_restarts_total[24h]))` |\n| Webhook error ratio | `rate(openclaw_lens_webhook_error_total[5m]) / (rate(openclaw_lens_webhook_received_total[5m]) + 0.001)` |\n| Tool error rate by class | `sum by (tool, error_class) (rate(openclaw_lens_tool_error_classes_total[5m]))` |\n| Session resets by reason | `sum by (reason) (increase(openclaw_lens_session_resets_total[1h]))` |\n| Unique sessions (1h rolling) | `openclaw_lens_unique_sessions_1h` |\n| Session completion rate anomaly | `rate(openclaw_lens_sessions_completed_total[5m]) > 3 * avg_over_time(rate(openclaw_lens_sessions_completed_total[5m])[1h:5m])` |\n\n## Common LogQL Expressions\n\nUse with `grafana_query_logs` against a Loki datasource. All OpenClaw agent logs use `service_name=\"openclaw\"` (standard OTLP resource attribute mapping).\n\n| Question | LogQL |\n|----------|-------|\n| All errors | `{service_name=\"openclaw\"} \\| logfmt \\| level=\"ERROR\"` |\n| Errors and warnings | `{service_name=\"openclaw\"} \\| logfmt \\| level=~\"ERROR\\|WARN\"` |\n| All tool calls | `{service_name=\"openclaw\"} \\|= \"openclaw.tool.call\" \\| logfmt` |\n| Specific tool calls | `{service_name=\"openclaw\"} \\|= \"openclaw.tool.call\" \\| logfmt \\| gen_ai_tool_name=\"grafana_query\"` |\n| Slow LLM calls (>10s) | `{service_name=\"openclaw\"} \\|= \"openclaw.llm.output\" \\| logfmt \\| openclaw_duration_s > 10` |\n| Delivery failures | `{service_name=\"openclaw\"} \\|= \"openclaw.message.sent\" \\| logfmt \\| openclaw_success=\"false\"` |\n| Session events | `{service_name=\"openclaw\"} \\| logfmt \\| event_name=~\"session\\\\..*\"` |\n| Compaction events | `{service_name=\"openclaw\"} \\| logfmt \\| event_name=~\"compaction\\\\..*\"` |\n| Subagent events | `{service_name=\"openclaw\"} \\| logfmt \\| event_name=~\"subagent\\\\..*\"` |\n| Subagent linking events | `{service_name=\"openclaw\"} \\| logfmt \\| event_name=\"subagent.linked\"` |\n| Child sessions of a parent | `{service_name=\"openclaw\"} \\| json \\| event_name=\"usage.session_summary\" \\| openclaw_parent_session_id=\"<parent_id>\"` |\n| All subagent sessions | `{service_name=\"openclaw\"} \\| json \\| event_name=\"usage.session_summary\" \\| openclaw_is_subagent=\"true\"` |\n| Parent sessions with children | `{service_name=\"openclaw\"} \\| json \\| event_name=\"usage.session_summary\" \\| openclaw_has_children=\"true\"` |\n| Logs with trace correlation | `{service_name=\"openclaw\"} \\| logfmt \\| trace_id != \"\"` |\n| Session usage summaries | `{service_name=\"openclaw\"} \\|= \"usage.session_summary\"` |\n| High-cost sessions (>$1) | `{service_name=\"openclaw\"} \\| json \\| event_name=\"usage.session_summary\" \\| openclaw_cost_total > 1` |\n| Sessions with errors | `{service_name=\"openclaw\"} \\| json \\| event_name=\"usage.session_summary\" \\| openclaw_messages_errors > 0` |\n| Low cache efficiency | `{service_name=\"openclaw\"} \\| json \\| event_name=\"usage.session_summary\" \\| openclaw_cache_hit_ratio < 0.5` |\n| Slow sessions (avg >10s) | `{service_name=\"openclaw\"} \\| json \\| event_name=\"usage.session_summary\" \\| openclaw_latency_avg_ms > 10000` |\n| Security events (injections, gateway) | `{service_name=\"openclaw\"} \\| json \\| component=\"lifecycle\" \\| event_name=~\"prompt_injection.detected\\|gateway.start\\|gateway.stop\"` |\n| Tool loop events | `{service_name=\"openclaw\"} \\| json \\| component=\"diagnostic\" \\| event_name=\"tool.loop\"` |\n| Gateway lifecycle events | `{service_name=\"openclaw\"} \\| json \\| component=\"lifecycle\" \\| event_name=~\"gateway.*\"` |\n\n**Tip**: Log lines include `trace_id` and `span_id` attributes for click-through from Loki → Tempo in Grafana. Use the Tempo datasource in a \"Derived fields\" config on Loki to enable automatic trace links.\n\n## Common TraceQL Expressions\n\nUse in Grafana's Tempo Explore view or in Tempo trace panel search filters. All OpenClaw traces use `resource.service.name=\"openclaw\"`.\n\n| Question | TraceQL |\n|----------|---------|\n| Session traces (root spans) | `{resource.service.name=\"openclaw\" && name=~\"invoke_agent.*\"}` |\n| LLM calls | `{resource.service.name=\"openclaw\" && name=~\"chat.*\"}` |\n| All tool executions | `{resource.service.name=\"openclaw\" && span.gen_ai.operation.name=\"execute_tool\"}` |\n| Specific tool calls | `{resource.service.name=\"openclaw\" && span.gen_ai.tool.name=\"grafana_query\"}` |\n| Slow LLM calls (>10s) | `{resource.service.name=\"openclaw\" && name=~\"chat.*\" && duration > 10s}` |\n| Errored spans | `{resource.service.name=\"openclaw\" && status=error}` |\n| By model | `{resource.service.name=\"openclaw\" && span.gen_ai.request.model=~\"claude.*\"}` |\n| High token usage | `{resource.service.name=\"openclaw\" && span.gen_ai.usage.input_tokens > 50000}` |\n| Subagent root spans | `{resource.service.name=\"openclaw\" && span.openclaw.is_subagent=true && name=~\"invoke_agent.*\"}` |\n| Child traces of a parent | `{resource.service.name=\"openclaw\" && span.openclaw.parent_trace_id=\"<parent_trace_id>\" && name=~\"invoke_agent.*\"}` |\n| Long-lived subagent spawn spans | `{resource.service.name=\"openclaw\" && name=~\"openclaw.subagent.spawn.*\"}` |\n| Subagent spawn spans with links | `{resource.service.name=\"openclaw\" && name=~\"openclaw.subagent.spawn.*\" && span.openclaw.subagent.child_session_id!=\"\"}` |\n\n**Span hierarchy**: `invoke_agent openclaw` (root) → `chat {model}` (LLM call) → `execute_tool {toolName}` (tool execution). Additional spans: `openclaw.compaction`, `openclaw.subagent.spawn` (long-lived, brackets subagent lifetime), `openclaw.agent.end`, `openclaw.message.received`, `openclaw.message.sent`.\n\n**Cross-trace correlation (subagents)**: Subagent spawn spans have bidirectional span links to child root spans. Use `span.links.traceId` in TraceQL to navigate. Child root spans have `openclaw.parent_trace_id` for querying related traces. Session explorer dashboard provides clickable drill-down.\n\n## Alert-Worthy Metrics\n\n| Condition | PromQL | Suggested Threshold |\n|-----------|--------|-------------------|\n| Stuck sessions | `openclaw_lens_sessions_stuck > 0` | Any stuck session |\n| Long stuck session | `openclaw_lens_stuck_session_max_age_ms > 60000` | Stuck > 60s |\n| Tool loops detected | `sum(openclaw_lens_tool_loops_active) > 0` | Any active loop |\n| High daily cost | `openclaw_lens_daily_cost_usd > 5` | $5/day |\n| Queue backed up | `openclaw_lens_queue_depth > 20` | 20+ queued |\n| Webhook latency spike | `histogram_quantile(0.95, sum(rate(openclaw_lens_webhook_duration_ms_bucket[5m])) by (le)) > 5000` | p95 > 5s |\n| Queue wait too long | `histogram_quantile(0.95, sum(rate(openclaw_lens_queue_wait_ms_bucket[5m])) by (le)) > 30000` | p95 > 30s |\n| High stuck session rate | `rate(openclaw_lens_sessions_stuck[5m]) > 0.1` | > 0.1/s (sustained) |\n| Excessive webhook errors | `rate(openclaw_lens_webhook_error_total[5m]) > 0.1` | > 0.1 errors/s |\n| Message error rate | `rate(openclaw_lens_messages_processed_total{outcome=\"error\"}[5m]) > 0.1` | > 0.1 errors/s |\n| Queue drain stall | `sum(rate(openclaw_lens_queue_lane_enqueue_total[5m])) > 2 * sum(rate(openclaw_lens_queue_lane_dequeue_total[5m]))` | Enqueue 2x dequeue |\n| Context window near limit | `openclaw_lens_context_tokens{type=\"used\"} / openclaw_lens_context_tokens{type=\"limit\"} > 0.9` | > 90% used |\n| Slow LLM calls (gen_ai) | `histogram_quantile(0.95, sum(rate(gen_ai_client_operation_duration_seconds_bucket[5m])) by (le)) > 30` | p95 > 30s |\n| High compaction rate | `rate(openclaw_lens_compactions_total[5m]) > 0.1` | > 0.1/s (context thrashing) |\n| Message delivery failures | `rate(openclaw_lens_message_delivery_total{success=\"false\"}[5m]) > 0.1` | > 0.1 failures/s |\n| Low cache efficiency | `openclaw_lens_cache_token_ratio < 0.3` | Cache ratio < 30% |\n| High avg LLM latency | `openclaw_lens_session_latency_avg_ms > 15000` | > 15s average |\n| High session error rate | `sum(rate(openclaw_lens_sessions_completed_total{outcome=\"error\"}[5m])) / sum(rate(openclaw_lens_sessions_completed_total[5m])) > 0.2` | > 20% errors |\n| Slow subagents | `histogram_quantile(0.95, sum(rate(openclaw_lens_subagent_duration_ms_bucket[5m])) by (le)) > 120000` | p95 > 2min |\n| Prompt injection signals | `sum(increase(openclaw_lens_prompt_injection_signals_total[1h])) > 3` | 3+ signals/hour |\n| Gateway restarts | `sum(increase(openclaw_lens_gateway_restarts_total[1h])) > 2` | 2+ restarts/hour |\n| Webhook error ratio high | `rate(openclaw_lens_webhook_error_total[5m]) / (rate(openclaw_lens_webhook_received_total[5m]) + 0.001) > 0.2` | > 20% errors |\n| Session enumeration | `openclaw_lens_unique_sessions_1h > 50` | 50+ unique sessions/hour |\n| Tool error burst | `sum(rate(openclaw_lens_tool_error_classes_total[5m])) > 0.5` | > 0.5 errors/s |\n| Cost spike | `openclaw_lens_daily_cost_usd > 10` | $10/day |\n\n## Session Summary Log — Subagent Hierarchy Attributes\n\nThe `usage.session_summary` log (event_name=`usage.session_summary`) includes these additional attributes for subagent correlation:\n\n| Attribute (Loki key) | Present When | Value |\n|----------------------|-------------|-------|\n| `openclaw_is_subagent` | Session is a spawned subagent | `true` |\n| `openclaw_parent_session_id` | Session is a spawned subagent | Parent session ID (clickable in Session Explorer) |\n| `openclaw_child_session_ids` | Session spawned subagents | Comma-separated child session IDs |\n| `openclaw_child_count` | Session spawned subagents | Number of child sessions |\n| `openclaw_has_children` | Session spawned subagents | `true` |\n\n## Log Event Types — Subagent Lifecycle\n\n| Event Name | Description | Key Attributes |\n|------------|-------------|----------------|\n| `subagent.spawn` | Subagent spawned by parent | `openclaw_agent_id`, `openclaw_mode`, `openclaw_child_session_key` |\n| `subagent.linked` | Deferred linking completed — child matched to parent | `openclaw_session_id` (child), `openclaw_parent_session_id`, `openclaw_parent_trace_id`, `openclaw_subagent_agent_id` |\n| `subagent.end` | Subagent finished | `openclaw_target_session_key`, `openclaw_reason`, `openclaw_outcome` |\n\n## SRE Investigation Patterns\n\nFor advanced investigation patterns — anomaly detection (z-score, predict_linear),\nRED/USE method compositions, SLI/SLO burn rates, and multi-signal investigation workflows\n— see [sre-investigation.md](sre-investigation.md).\n\nFile v0.5.0:references/alloy-components.md\n\n# Alloy Component Reference — Escape Hatch Companion\n\nWhen no recipe fits, compose raw Alloy configs using these component patterns.\nUse `alloy_pipeline` with `config` param + optional `sampleQueries` for data verification.\n\n## Table of Contents\n\n- [Log Sources](#log-sources)\n- [Log Processing](#log-processing)\n- [Metrics Exporters](#metrics-exporters)\n- [OTel Processors](#otel-processors)\n- [OTel Connectors](#otel-connectors)\n- [Profiling](#profiling)\n- [Frontend](#frontend)\n- [Wiring Patterns](#wiring-patterns)\n\n---\n\n## Log Sources\n\n### loki.source.gelf — GELF UDP Log Source\n\nReceives GELF (Graylog Extended Log Format) logs over UDP. Common with Graylog, Docker GELF driver.\n\n```alloy\nloki.source.gelf \"my_gelf\" {\n  forward_to = [loki.write.default.receiver]\n}\n```\n\nDefault: listens on `0.0.0.0:12201` (UDP). Override with `use_incoming_timestamp = true`.\nLabels: Auto-extracts `__gelf_message_host`, `__gelf_message_level`, `__gelf_message_facility`.\nUse `loki.relabel` to promote `__gelf_*` labels.\n\n**Sample queries**: `{source=\"gelf\"}`, `{source=\"gelf\"} |= \"error\"`\n\n### loki.source.api — Loki Push API Endpoint\n\nAccepts logs via Loki-compatible HTTP push API. Use for centralized log gateways, TCP JSON ingestion.\n\n```alloy\nloki.source.api \"push\" {\n  http {\n    listen_address = \"0.0.0.0\"\n    listen_port    = 3500\n  }\n  forward_to = [loki.process.parse.receiver]\n}\n```\n\n**Sample queries**: `{source=\"push-api\"}`, `rate({source=\"push-api\"}[5m])`\n\n### loki.source.kafka — Kafka Log Consumer\n\nConsumes log messages from Apache Kafka topics.\n\n```alloy\nloki.source.kafka \"logs\" {\n  brokers       = [\"kafka:9092\"]\n  topics        = [\"app-logs\"]\n  consumer_group = \"alloy\"\n  forward_to    = [loki.process.parse.receiver]\n}\n```\n\nAuthentication: Add `authentication { type = \"sasl\" ... }` block for SASL/SCRAM.\n**Sample queries**: `{source=\"kafka\"}`, `{source=\"kafka\", topic=\"app-logs\"}`\n\n### loki.source.windowsevent — Windows Event Logs\n\nCollects Windows Event Log entries. Windows-only.\n\n```alloy\nloki.source.windowsevent \"events\" {\n  eventlog_name = \"Application\"\n  forward_to    = [loki.process.parse.receiver]\n}\n```\n\nCommon event logs: `\"Application\"`, `\"System\"`, `\"Security\"`.\n**Sample queries**: `{source=\"windowsevent\"}`, `{source=\"windowsevent\"} |= \"error\"`\n\n---\n\n## Log Processing\n\nInsert `loki.process` between source and `loki.write` for parsing, enrichment, and routing.\n\n### stage.json — JSON Field Extraction\n\n```alloy\nloki.process \"parse\" {\n  stage.json {\n    expressions = {\n      \"timestamp\"  = \"\",\n      \"level\"      = \"\",\n      \"message\"    = \"\",\n      \"request_id\" = \"context.request_id\",\n    }\n  }\n  forward_to = [loki.write.default.receiver]\n}\n```\n\nEmpty string `\"\"` extracts the top-level key matching the name. Dotted paths extract nested fields.\n\n### stage.labels — Promote Fields to Labels\n\n```alloy\n  stage.labels {\n    values = {\n      \"level\" = \"\",\n      \"service\" = \"\",\n    }\n  }\n```\n\nPromotes extracted fields to Loki index labels. Use sparingly — high-cardinality labels hurt performance.\n\n### stage.structured_metadata — High-Cardinality Fields\n\n```alloy\n  stage.structured_metadata {\n    values = {\n      \"request_id\" = \"\",\n      \"user_id\"    = \"\",\n    }\n  }\n```\n\nFor high-cardinality data. Queryable via `| request_id=\"abc\"` but not indexed as labels.\n\n### stage.timestamp — Parse Log Timestamps\n\n```alloy\n  stage.timestamp {\n    source = \"timestamp\"\n    format = \"RFC3339\"\n  }\n```\n\nFormats: `\"RFC3339\"`, `\"RFC3339Nano\"`, `\"Unix\"`, `\"UnixMs\"`, or Go time layout strings.\n\n### stage.static_labels — Add Fixed Labels\n\n```alloy\n  stage.static_labels {\n    values = {\n      \"environment\" = \"production\",\n      \"service_name\" = \"my-app\",\n    }\n  }\n```\n\n### stage.output — Set Log Line Content\n\n```alloy\n  stage.output {\n    source = \"message\"\n  }\n```\n\nReplaces the log line with the extracted `message` field.\n\n### stage.regex — Regex Extraction\n\n```alloy\n  stage.regex {\n    expression = \"^(?P<timestamp>\\\\S+) (?P<level>\\\\w+) (?P<message>.*)$\"\n  }\n```\n\nNamed capture groups become available for subsequent stages.\n\n### stage.match — Conditional Processing\n\n```alloy\n  stage.match {\n    selector = '{app=\"frontend\"}'\n    stage.json { expressions = { \"url\" = \"\" } }\n    stage.labels { values = { \"url\" = \"\" } }\n  }\n```\n\nApply stages only to logs matching the LogQL selector.\n\n### stage.tenant — Multi-Tenant Routing\n\n```alloy\n  stage.tenant {\n    source = \"tenant_id\"\n  }\n```\n\nRoutes logs to different Loki tenants based on extracted field.\n\n### loki.secretfilter — Secret Redaction\n\n```alloy\nloki.secretfilter \"redact\" {\n  forward_to = [loki.write.default.receiver]\n}\n```\n\nUses built-in Gitleaks patterns to redact secrets. Optional: `redact_with = \"<REDACTED:$SECRET_NAME>\"`, `types = \"all\"`.\n\n### Full Processing Pipeline Example\n\n```alloy\nloki.source.api \"push\" {\n  http { listen_port = 3500 }\n  forward_to = [loki.process.parse.receiver]\n}\n\nloki.process \"parse\" {\n  stage.json {\n    expressions = { \"timestamp\" = \"\", \"level\" = \"\", \"message\" = \"\", \"user_id\" = \"\" }\n  }\n  stage.timestamp {\n    source = \"timestamp\"\n    format = \"RFC3339\"\n  }\n  stage.labels {\n    values = { \"level\" = \"\" }\n  }\n  stage.structured_metadata {\n    values = { \"user_id\" = \"\" }\n  }\n  stage.output {\n    source = \"message\"\n  }\n  forward_to = [loki.write.default.receiver]\n}\n\nloki.write \"default\" {\n  endpoint {\n    url = \"http://loki:3100/loki/api/v1/push\"\n  }\n}\n```\n\n---\n\n## Metrics Exporters\n\nAll follow: `prometheus.exporter.X → prometheus.scrape → prometheus.remote_write`.\n\n### prometheus.exporter.blackbox — Synthetic HTTP Probing\n\n```alloy\nprometheus.exporter.blackbox \"probes\" {\n  config = \"{ modules: { http_2xx: { prober: http, timeout: 5s } } }\"\n\n  target {\n    name    = \"web\"\n    address = \"http://myapp:8080\"\n    module  = \"http_2xx\"\n  }\n}\n\nprometheus.scrape \"blackbox\" {\n  targets    = prometheus.exporter.blackbox.probes.targets\n  forward_to = [prometheus.remote_write.default.receiver]\n}\n```\n\nModules: `http_2xx` (HTTP), `tcp_connect` (TCP), `icmp` (ICMP/ping).\n**Sample queries**: `probe_success{instance=\"...\"}`, `probe_http_duration_seconds`, `probe_http_status_code`\n\n### prometheus.exporter.memcached — Memcached Metrics\n\n```alloy\nprometheus.exporter.memcached \"cache\" {\n  address = \"memcached:11211\"\n}\n\nprometheus.scrape \"memcached\" {\n  targets    = prometheus.exporter.memcached.cache.targets\n  forward_to = [prometheus.remote_write.default.receiver]\n}\n```\n\n**Sample queries**: `memcached_up`, `memcached_current_bytes`, `memcached_current_connections`\n\n### prometheus.exporter.snmp — SNMP Device Monitoring\n\n```alloy\nprometheus.exporter.snmp \"devices\" {\n  config_file = \"/etc/alloy/snmp.yml\"\n\n  target \"switch\" {\n    address = \"192.168.1.1\"\n    module  = \"if_mib\"\n  }\n}\n```\n\nRequires external `snmp.yml` config file (MIB definitions).\n**Sample queries**: `ifOperStatus`, `ifInOctets`, `ifOutOctets`\n\n### prometheus.exporter.windows — Windows System Metrics\n\n```alloy\nprometheus.exporter.windows \"system\" {\n  enabled_collectors = [\"cpu\", \"cs\", \"logical_disk\", \"net\", \"os\", \"system\"]\n}\n```\n\nWindows-only. Collectors: `cpu`, `cs`, `logical_disk`, `memory`, `net`, `os`, `process`, `system`.\n**Sample queries**: `windows_cpu_time_total`, `windows_logical_disk_free_bytes`\n\n### prometheus.exporter.self — Alloy Self-Monitoring\n\n```alloy\nprometheus.exporter.self \"alloy\" {}\n\nprometheus.scrape \"self\" {\n  targets    = prometheus.exporter.self.alloy.targets\n  forward_to = [prometheus.remote_write.default.receiver]\n}\n```\n\n**Sample queries**: `alloy_build_info`, `rate(alloy_component_evaluation_slow_seconds_count[5m])`, `alloy_component_controller_running_components`\n\n---\n\n## OTel Processors\n\n### otelcol.processor.tail_sampling — Trace Sampling Policies\n\n```alloy\notelcol.processor.tail_sampling \"smart\" {\n  decision_wait = \"10s\"\n  num_traces    = 100\n\n  // Always keep error traces\n  policy {\n    name = \"errors\"\n    type = \"status_code\"\n    status_code { status_codes = [\"ERROR\"] }\n  }\n\n  // Keep slow traces (>5s)\n  policy {\n    name = \"latency\"\n    type = \"latency\"\n    latency { threshold_ms = 5000 }\n  }\n\n  // Drop health checks\n  policy {\n    name = \"drop-health\"\n    type = \"string_attribute\"\n    string_attribute {\n      key          = \"http.url\"\n      values       = [\"/health\", \"/metrics\", \"/ready\"]\n      invert_match = true\n    }\n  }\n\n  // Sample 10% of remaining\n  policy {\n    name = \"probabilistic\"\n    type = \"probabilistic\"\n    probabilistic { sampling_percentage = 10 }\n  }\n\n  output {\n    traces = [otelcol.processor.batch.default.input]\n  }\n}\n```\n\nPolicy types: `status_code`, `latency`, `probabilistic`, `string_attribute`, `numeric_attribute`, `always_sample`, `rate_limiting`.\n\n### otelcol.processor.transform — Attribute Transformation\n\n```alloy\notelcol.processor.transform \"enrich\" {\n  metric_statements {\n    context    = \"datapoint\"\n    statements = [\n      \"set(attributes[\\\"environment\\\"], \\\"production\\\")\",\n    ]\n  }\n  output {\n    metrics = [otelcol.exporter.otlphttp.default.input]\n  }\n}\n```\n\nUses OTTL (OpenTelemetry Transformation Language). Contexts: `resource`, `scope`, `span`, `spanevent`, `metric`, `datapoint`, `log`.\n\n### otelcol.processor.resourcedetection — Auto-Detect Host Metadata\n\n```alloy\notelcol.processor.resourcedetection \"env\" {\n  detectors = [\"env\", \"system\"]\n  system {\n    hostname_sources = [\"os\"]\n  }\n  output {\n    traces = [otelcol.processor.batch.default.input]\n  }\n}\n```\n\nDetectors: `env`, `system`, `docker`, `gcp`, `aws`, `azure`.\n\n---\n\n## OTel Connectors\n\nConnectors consume one signal type and produce another.\n\n### otelcol.connector.spanmetrics — RED Metrics from Traces\n\n```alloy\notelcol.connector.spanmetrics \"red\" {\n  histogram {\n    explicit {}\n  }\n  dimension { name = \"http.method\" }\n  dimension { name = \"http.status_code\" }\n  metrics_flush_interval = \"5s\"\n\n  output {\n    metrics = [otelcol.exporter.otlphttp.prometheus.input]\n  }\n}\n```\n\nProduces: `traces_spanmetrics_calls_total`, `traces_spanmetrics_duration_milliseconds_bucket`.\nWire: `batch.output.traces = [spanmetrics.input, exporter.input]` (dual output).\n\n### otelcol.connector.servicegraph — Service Dependency Graphs\n\n```alloy\notelcol.connector.servicegraph \"graph\" {\n  metrics_flush_interval = \"10s\"\n  dimensions             = [\"service.name\", \"http.method\"]\n  store {\n    max_items = 5000\n    ttl       = \"30s\"\n  }\n  output {\n    metrics = [otelcol.exporter.otlphttp.prometheus.input]\n  }\n}\n```\n\nProduces: `traces_service_graph_request_total`, `traces_service_graph_request_server_seconds_bucket`.\nSame dual-output wiring as spanmetrics.\n\n### otelcol.connector.count — Count Signals\n\n```alloy\notelcol.connector.count \"req\" {\n  spans {\n    \"request.count\" { description = \"Total request count\" }\n  }\n  output {\n    metrics = [otelcol.exporter.otlphttp.prometheus.input]\n  }\n}\n```\n\nDerives count metrics from traces, logs, or spans.\n\n---\n\n## Profiling\n\n### pyroscope.scrape + pyroscope.write — Continuous Profiling\n\n```alloy\npyroscope.scrape \"profiles\" {\n  targets = [\n    { \"__address__\" = \"myapp:6060\", \"service_name\" = \"myapp\" },\n  ]\n  profiling_config {\n    profile.process_cpu { enabled = true }\n    profile.memory      { enabled = true }\n    profile.goroutine   { enabled = true }\n  }\n  forward_to = [pyroscope.write.default.receiver]\n}\n\npyroscope.write \"default\" {\n  endpoint {\n    url = \"http://pyroscope:4040\"\n  }\n}\n```\n\nProfiles: `process_cpu`, `memory`, `goroutine`, `mutex`, `block`. Go apps expose at `:6060/debug/pprof`.\n\n---\n\n## Frontend\n\n### faro.receiver — Browser RUM/Web Vitals\n\n```alloy\nfaro.receiver \"frontend\" {\n  server {\n    listen_address = \"0.0.0.0\"\n    listen_port    = 12347\n    cors_allowed_origins = [\"*\"]\n  }\n  output {\n    logs = [loki.write.default.receiver]\n  }\n}\n```\n\nCollects frontend telemetry via Grafana Faro SDK. Signals: logs, traces, measurements (web vitals).\n\n---\n\n## Wiring Patterns\n\n### Source\n\nArchive v0.4.0: 97 files, 438251 bytes\n\nFiles: CLAUDE.md (28720b), index.ts (13228b), llms.txt (5728b), openclaw.plugin.json (7484b), package.json (2507b), README.md (36330b), ROADMAP.md (9096b), SKILL.md (78760b), skills/references/agent-metrics.md (32885b), skills/references/dashboard-composition.md (14256b), skills/references/external-data.md (4673b), skills/references/sre-investigation.md (18777b), skills/SKILL.md (78760b), src/config.test.ts (12398b), src/config.ts (10691b), src/grafana-client-registry.test.ts (4420b), src/grafana-client-registry.ts (2312b), src/grafana-client.test.ts (29539b), src/grafana-client.ts (39052b), src/metric-definitions.test.ts (9869b), src/metric-definitions.ts (14841b), src/sdk-compat.ts (5458b), src/services/alert-webhook.test.ts (8274b), src/services/alert-webhook.ts (8155b), src/services/custom-metrics-store.test.ts (29065b), src/services/custom-metrics-store.ts (21458b), src/services/lifecycle-telemetry.test.ts (169949b), src/services/lifecycle-telemetry.ts (95525b), src/services/metrics-collector.test.ts (66879b), src/services/metrics-collector.ts (59418b), src/services/model-pricing.test.ts (2423b), src/services/model-pricing.ts (4406b), src/services/otel-logs.test.ts (3069b), src/services/otel-logs.ts (2165b), src/services/otel-metrics.test.ts (2981b), src/services/otel-metrics.ts (2084b), src/services/otel-traces.test.ts (2915b), src/services/otel-traces.ts (2240b), src/services/otlp-json-writer.test.ts (7426b), src/services/otlp-json-writer.ts (4320b), src/services/redact.test.ts (5622b), src/services/redact.ts (4071b), src/templates/cost-intelligence.json (36078b), src/templates/genai-observability.json (27459b), src/templates/http-service.json (6198b), src/templates/llm-command-center.json (53130b), src/templates/metric-explorer.json (4892b), src/templates/multi-kpi.json (5614b), src/templates/node-exporter.json (6863b), src/templates/security-overview.json (14081b), src/templates/session-explorer.json (55151b), src/templates/sre-operations.json (25353b), src/templates/tool-performance.json (17315b), src/templates/weekly-review.json (4839b), src/tools/annotate.test.ts (11339b), src/tools/annotate.ts (7481b), src/tools/check-alerts.test.ts (40899b), src/tools/check-alerts.ts (25369b), src/tools/create-alert.test.ts (15775b), src/tools/create-alert.ts (12557b), src/tools/create-dashboard.test.ts (27884b), src/tools/create-dashboard.ts (17908b), src/tools/explain-metric.test.ts (56003b), src/tools/explain-metric.ts (20809b), src/tools/explore-datasources.test.ts (4423b), src/tools/explore-datasources.ts (3544b), src/tools/get-dashboard.test.ts (22795b), src/tools/get-dashboard.ts (11952b), src/tools/health-context.test.ts (6040b), src/tools/health-context.ts (3771b), src/tools/instance-param.ts (848b), src/tools/investigate.test.ts (29531b), src/tools/investigate.ts (25178b), src/tools/list-metrics.test.ts (49209b), src/tools/list-metrics.ts (18734b), src/tools/push-metrics.test.ts (23306b), src/tools/push-metrics.ts (12851b), src/tools/query-guidance.test.ts (5291b), src/tools/query-guidance.ts (16582b), src/tools/query-logs.test.ts (31716b)\n\nArchive v0.3.0: 57 files, 280497 bytes\n\nFiles: index.ts (12822b), openclaw.plugin.json (5784b), package.json (2652b), README.md (36330b), SKILL.md (77628b), skills/references/agent-metrics.md (32885b), skills/references/dashboard-composition.md (14256b), skills/references/external-data.md (4673b), skills/references/sre-investigation.md (18777b), skills/SKILL.md (77628b), src/config.ts (6832b), src/grafana-client.ts (38928b), src/metric-definitions.ts (14841b), src/services/alert-webhook.ts (8155b), src/services/custom-metrics-store.ts (21458b), src/services/lifecycle-telemetry.ts (95525b), src/services/metrics-collector.ts (58710b), src/services/model-pricing.ts (4406b), src/services/otel-logs.ts (2165b), src/services/otel-metrics.ts (2084b), src/services/otel-traces.ts (2240b), src/services/otlp-json-writer.ts (4320b), src/services/redact.ts (4071b), src/templates/cost-intelligence.json (36078b), src/templates/genai-observability.json (27459b), src/templates/http-service.json (6198b), src/templates/llm-command-center.json (53130b), src/templates/metric-explorer.json (4892b), src/templates/multi-kpi.json (5614b), src/templates/node-exporter.json (6863b), src/templates/security-overview.json (14081b), src/templates/session-explorer.json (55151b), src/templates/sre-operations.json (25353b), src/templates/tool-performance.json (17315b), src/templates/weekly-review.json (4839b), src/tools/annotate.ts (7465b), src/tools/check-alerts.ts (25240b), src/tools/create-alert.ts (12459b), src/tools/create-dashboard.ts (17872b), src/tools/explain-metric.ts (20788b), src/tools/explore-datasources.ts (3163b), src/tools/get-dashboard.ts (11916b), src/tools/health-context.ts (3771b), src/tools/investigate.ts (25050b), src/tools/list-metrics.ts (18713b), src/tools/push-metrics.ts (12845b), src/tools/query-guidance.ts (16582b), src/tools/query-logs.ts (15776b), src/tools/query-traces.ts (16877b), src/tools/query.ts (11927b), src/tools/resolve-panel.ts (6434b), src/tools/search.ts (6600b), src/tools/security-check.ts (11924b), src/tools/share-dashboard.ts (10498b), src/tools/update-dashboard.ts (17266b), tsconfig.json (463b), _meta.json (131b)\n\nArchive v0.2.0: 87 files, 383849 bytes\n\nFiles: _meta.json (131b), index.ts (12670b), openclaw.plugin.json (5784b), package.json (2302b), README.md (33577b), SKILL.md (72563b), skills/references/agent-metrics.md (32624b), skills/references/dashboard-composition.md (14256b), skills/references/external-data.md (4673b), skills/SKILL.md (72563b), src/config.test.ts (6516b), src/config.ts (6832b), src/grafana-client.test.ts (29539b), src/grafana-client.ts (38928b), src/metric-definitions.test.ts (9869b), src/metric-definitions.ts (14841b), src/services/alert-webhook.test.ts (8171b), src/services/alert-webhook.ts (8155b), src/services/custom-metrics-store.test.ts (29065b), src/services/custom-metrics-store.ts (21458b), src/services/lifecycle-telemetry.test.ts (169949b), src/services/lifecycle-telemetry.ts (95525b), src/services/metrics-collector.test.ts (66717b), src/services/metrics-collector.ts (58710b), src/services/model-pricing.test.ts (2423b), src/services/model-pricing.ts (4406b), src/services/otel-logs.test.ts (3069b), src/services/otel-logs.ts (2165b), src/services/otel-metrics.test.ts (2981b), src/services/otel-metrics.ts (2084b), src/services/otel-traces.test.ts (2915b), src/services/otel-traces.ts (2240b), src/services/otlp-json-writer.test.ts (7426b), src/services/otlp-json-writer.ts (4320b), src/services/redact.test.ts (5622b), src/services/redact.ts (4071b), src/templates/cost-intelligence.json (36078b), src/templates/genai-observability.json (27459b), src/templates/http-service.json (6198b), src/templates/llm-command-center.json (53130b), src/templates/metric-explorer.json (4892b), src/templates/multi-kpi.json (5614b), src/templates/node-exporter.json (6863b), src/templates/security-overview.json (14081b), src/templates/session-explorer.json (55151b), src/templates/sre-operations.json (25353b), src/templates/tool-performance.json (17315b), src/templates/weekly-review.json (4839b), src/tools/annotate.test.ts (10985b), src/tools/annotate.ts (7465b), src/tools/check-alerts.test.ts (34054b), src/tools/check-alerts.ts (19339b), src/tools/create-alert.test.ts (15413b), src/tools/create-alert.ts (12459b), src/tools/create-dashboard.test.ts (27499b), src/tools/create-dashboard.ts (17872b), src/tools/explain-metric.test.ts (47578b), src/tools/explain-metric.ts (16659b), src/tools/explore-datasources.test.ts (4087b), src/tools/explore-datasources.ts (3163b), src/tools/get-dashboard.test.ts (22431b), src/tools/get-dashboard.ts (11916b), src/tools/health-context.test.ts (6040b), src/tools/health-context.ts (3771b), src/tools/list-metrics.test.ts (48787b), src/tools/list-metrics.ts (18713b), src/tools/push-metrics.test.ts (22960b), src/tools/push-metrics.ts (12845b), src/tools/query-guidance.test.ts (5291b), src/tools/query-guidance.ts (16582b), src/tools/query-logs.test.ts (31328b), src/tools/query-logs.ts (15776b), src/tools/query-traces.test.ts (21945b), src/tools/query-traces.ts (16877b), src/tools/query.test.ts (27462b), src/tools/query.ts (11927b), src/tools/resolve-panel.test.ts (7588b), src/tools/resolve-panel.ts (6434b), src/tools/search.test.ts (8733b), src/tools/search.ts (6600b)","readmeExcerpt":"Skill: Grafana Lens Owner: awsome-o Summary: Grafana tools for data visualization, monitoring, alerting, security, SRE investigation, and data collection pipeline management via Alloy. Use grafana_query... Tags: latest:0.5.0 Version history: v0.5.0 | 2026-04-05T16:38:57.121Z | user Major update: Adds Alloy pipeline management, new recipes, and greatly expands data collection capabilities. - Introduced Alloy pipeline ","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"# 1. Start the LGTM observability stack (Grafana + Prometheus + Loki + Tempo + OTel Collector)\n#    See https://github.com/grafana/docker-otel-lgtm for more options\ndocker pull grafana/otel-lgtm:latest\ndocker run -d --name lgtm -p 3000:3000 -p 4317:4317 -p 4318:4318 -p 9090:9090 grafana/otel-lgtm:latest\n\n# 2. Install the plugin\nopenclaw plugins install openclaw-grafana-lens\n\n# 3. Configure credentials (see \"Configuration\" section below for full options)\nexport GRAFANA_URL=http://localhost:3000\nexport GRAFANA_SERVICE_ACCOUNT_TOKEN=glsa_xxxxxxxxxxxx\n\n# 4. Restart the gateway to load the plugin\nopenclaw gateway restart\n\n# Optional: Enable Alloy pipeline management (for data collection)\n# See \"Alloy Pipeline Management\" section below for full setup\nexport ALLOY_URL=http://localhost:12345\nexport ALLOY_CONFIG_DIR=/path/to/alloy/config.d"},{"language":"bash","snippet":"docker pull grafana/otel-lgtm:latest\ndocker run -d --name lgtm -p 3000:3000 -p 4317:4317 -p 4318:4318 -p 9090:9090 grafana/otel-lgtm:latest"},{"language":"bash","snippet":"# Install from npm\nopenclaw plugins install openclaw-grafana-lens\n\n# Restart the gateway to load the plugin\nopenclaw gateway restart"},{"language":"bash","snippet":"# Clone and link locally\ngit clone <repo-url> ~/workspace/grafana-lens\ncd ~/workspace/grafana-lens && npm install\nopenclaw plugins install -l ~/workspace/grafana-lens\nopenclaw gateway restart"},{"language":"bash","snippet":"curl -sf http://localhost:3000/api/health && echo \"Grafana OK\""},{"language":"bash","snippet":"curl -sf http://localhost:12345/-/ready && echo \"Alloy OK\""}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: grafana-lens\ndescription: \"Grafana tools for data visualization, monitoring, alerting, security, SRE investigation, and data collection pipeline management via Alloy. Use grafana_query, grafana_query_logs, grafana_query_traces, grafana_create_dashboard, grafana_update_dashboard, grafana_create_alert, grafana_share_dashboard, grafana_annotate, grafana_explore_datasources, grafana_list_metrics, grafana_search, grafana_get_dashboard, grafana_check_alerts, grafana_push_metrics, grafana_explain_metric, grafana_security_check, grafana_investigate, and alloy_pipeline. Trigger when asked about metrics, dashboards, monitoring, alerts, costs, token usage, data visualization, PromQL, Prometheus, LogQL, Loki, log queries, error logs, log search, TraceQL, Tempo, traces, distributed tracing, span search, find slow traces, debug session traces, annotations, deployments, sharing charts, investigating alert notifications, pushing custom data (calendar, git, fitness, finance) to Grafana for visualization, pushing historical data, backfilling metrics, recording past data with timestamps, modifying dashboards, adding panels, removing panels, changing dashboard settings, updating dashboard time range, explain metric, metric trend, what is this metric, how has this changed, is this metric normal, why did my bill spike, cost visibility, security monitoring, security check, security audit, am I being attacked, is my agent compromised, suspicious activity, threat detection, prompt injection detection, set up security alerts, investigate, debug, triage, root cause, what's wrong, why is X broken, anomaly detection, RED method, USE method, alert fatigue, postmortem, incident summary, collect metrics from, monitor my database, monitor my app, scrape endpoint, set up log collection, collect Docker logs, tail log files, collect Kubernetes logs, receive OTLP, set up trace collection, data collection pipeline, Alloy pipeline, pipeline status, pipeline health, node exporter, system metrics, postgres exporter, mysql exporter, redis exporter, syslog, Grafana Alloy.\"\nmetadata:\n  {\n    \"openclaw\":\n      {\n        \"emoji\": \"🔭\",\n        \"requires\": { \"config\": [\"grafana.url\", \"grafana.apiKey\"] },\n      },\n  }\n---\n\n# Grafana Lens\n\nYou have full native Grafana access — query data, create dashboards, set alerts, receive alert notifications, annotate events, explore datasources, push custom data, and deliver visualizations inline. Works with ANY data in Grafana, not just agent metrics.\n\n## Musts\n\n- **Always call `grafana_explore_datasources` first** when you need a datasource UID — never guess UIDs\n- **Always call `grafana_search` before creating a dashboard** — avoid duplicates\n- **Always call `grafana_get_dashboard` before `grafana_share_dashboard`** — you need exact panel IDs\n- **Always call `grafana_get_dashboard` before `grafana_update_dashboard`** — you need panel IDs and current structure\n- **Prefer `grafana_query` for direct answers** over creating dashboards — \"what's m"},{"path":"README.md","content":"# Grafana Lens\n\n**Agent-driven Grafana observability for OpenClaw — query, visualize, alert, trace, and share across 15+ messaging channels.**\n\n> **Note:** This is a community-built OpenClaw plugin, not an official Grafana Labs product. Grafana, Loki, Tempo, and Prometheus are trademarks of Grafana Labs.\n\n[OpenClaw](https://openclaw.com) is an open-source AI agent platform. Grafana Lens extends it with full Grafana integration — 18 composable tools that let your agent query metrics and logs, trace distributed requests, create dashboards, set up alerts, render charts, run security audits, investigate incidents, push custom data, and manage data collection pipelines — all through natural language conversation.\n\n---\n\n## Why Grafana Lens?\n\n| Pain Point | How Grafana Lens Helps |\n|---|---|\n| **\"Where did my budget go?\"** | Cost dashboards with model-level attribution, token tracking, and cost anomaly alerts |\n| **\"Is my agent stuck in a loop?\"** | Real-time tool loop detection, stuck session monitoring, and SRE operations dashboard |\n| **\"Am I being prompt-injected?\"** | 12-pattern prompt injection detection with security dashboard and threat-level reporting |\n| **\"I need observability but don't want another SaaS\"** | Fully self-hosted, open source, OTLP-native — runs on a free local Grafana stack |\n| **\"I can't debug multi-step agent sessions\"** | Hierarchical traces: session → LLM call → tool execution, with log-to-trace correlation |\n| **\"My alert fired — now what?\"** | `grafana_investigate` gathers metrics, logs, traces in parallel and generates hypotheses with specific tool+params for follow-up |\n| **\"I want to track my own data in Grafana\"** | Push any custom metrics (fitness, calendar, git, finance) from conversation |\n| **\"How do I get my data INTO Grafana?\"** | `alloy_pipeline` sets up data collection from databases, Docker, Kubernetes, log files, and more — 29 recipes, just describe what you want to monitor |\n\n---\n\n## Key Features\n\n- **18 Composable Agent Tools** — Query PromQL/LogQL/TraceQL, create dashboards, set alerts, share panel images, run security checks, investigate incidents, push custom metrics, manage data collection pipelines, and more\n- **SRE Investigation** — Multi-signal triage (`grafana_investigate`), anomaly scoring with z-score against 7-day baselines, seasonality comparison, and alert fatigue detection\n- **Full OTLP Observability** — Metrics → Prometheus, Logs → Loki, Traces → Tempo. Push-based with no scraping — data is available immediately\n- **Security Monitoring** — 6-check threat assessment covering prompt injection, cost anomalies, tool loops, session enumeration, webhook errors, and stuck sessions\n- **12 Pre-Built Dashboard Templates** — From LLM Command Center and Cost Intelligence to Security Overview and SRE Operations\n- **Custom Data Observatory** — Push any external data (calendar events, git commits, fitness stats, financial metrics) into Grafana via conversation\n- **Works with ANY Datasource** — Not limited "},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn785ebhh3a5v4pqymw1yhz6wh8292pc\",\n  \"slug\": \"grafana-lens\",\n  \"version\": \"0.5.0\",\n  \"publishedAt\": 1775407137121\n}"},{"path":"references/agent-metrics.md","content":"# Agent Metrics Reference\n\nMetrics come from two sources — both push via OTLP to the same collector/Mimir instance and are queryable in Grafana.\n\n## Core Agent Telemetry (from diagnostics-otel)\n\nThese are published by OpenClaw's built-in `diagnostics-otel` extension. Grafana Lens dashboards query them but does not collect them.\n\n| Prometheus Name | Type | Labels | Source Event |\n|----------------|------|--------|-------------|\n| `openclaw_tokens_total` | counter | `openclaw_token`, `openclaw_model`, `openclaw_provider`, `openclaw_channel` | `model.usage` |\n| `openclaw_cost_usd_total` | counter | `openclaw_model`, `openclaw_provider`, `openclaw_channel` | `model.usage` |\n| `openclaw_run_duration_ms_milliseconds` | histogram | `openclaw_model`, `openclaw_provider`, `openclaw_channel` | `model.usage` |\n| `openclaw_context_tokens` | histogram | `openclaw_context` (limit/used), `openclaw_model`, `openclaw_provider`, `openclaw_channel` | `model.usage` |\n| `openclaw_message_processed_total` | counter | `openclaw_outcome`, `openclaw_channel` | `message.processed` |\n| `openclaw_message_duration_ms_milliseconds` | histogram | `openclaw_outcome`, `openclaw_channel` | `message.processed` |\n| `openclaw_message_queued_total` | counter | `openclaw_channel`, `openclaw_source` | `message.queued` |\n| `openclaw_webhook_received_total` | counter | `openclaw_channel`, `openclaw_webhook` | `webhook.received` |\n| `openclaw_webhook_error_total` | counter | `openclaw_channel`, `openclaw_webhook` | `webhook.error` |\n| `openclaw_webhook_duration_ms_milliseconds` | histogram | `openclaw_channel`, `openclaw_webhook` | `webhook.processed` |\n| `openclaw_queue_depth` | histogram | `openclaw_lane`, `openclaw_channel` | `message.queued`, `queue.lane.*`, `heartbeat` |\n| `openclaw_queue_wait_ms_milliseconds` | histogram | `openclaw_lane` | `queue.lane.dequeue` |\n| `openclaw_queue_lane_enqueue_total` | counter | `openclaw_lane` | `queue.lane.enqueue` |\n| `openclaw_queue_lane_dequeue_total` | counter | `openclaw_lane` | `queue.lane.dequeue` |\n| `openclaw_session_state_total` | counter | `openclaw_state`, `openclaw_reason` | `session.state` |\n| `openclaw_session_stuck_total` | counter | `openclaw_state` | `session.stuck` |\n| `openclaw_session_stuck_age_ms_milliseconds` | histogram | `openclaw_state` | `session.stuck` |\n| `openclaw_run_attempt_total` | counter | `openclaw_attempt` | `run.attempt` |\n\n**Label names use underscores** (OTel dots → Prometheus underscores): `openclaw.model` → `openclaw_model`.\n\n**OTel unit suffix**: Histograms declared with `unit: \"ms\"` get `_milliseconds` appended in Prometheus (OTLP-to-Prometheus translation). So the OTel instrument `openclaw_run_duration_ms` becomes `openclaw_run_duration_ms_milliseconds_bucket` in PromQL. All PromQL in this doc uses the physical Prometheus names.\n\n**Label value reference**:\n- `openclaw_token`: `input`, `output`, `cache_read`, `cache_write`, `prompt`, `total`\n- `openclaw_context`: `limit`, `used`\n- `openclaw_outcome`: `co"},{"path":"references/alloy-components.md","content":"# Alloy Component Reference — Escape Hatch Companion\n\nWhen no recipe fits, compose raw Alloy configs using these component patterns.\nUse `alloy_pipeline` with `config` param + optional `sampleQueries` for data verification.\n\n## Table of Contents\n\n- [Log Sources](#log-sources)\n- [Log Processing](#log-processing)\n- [Metrics Exporters](#metrics-exporters)\n- [OTel Processors](#otel-processors)\n- [OTel Connectors](#otel-connectors)\n- [Profiling](#profiling)\n- [Frontend](#frontend)\n- [Wiring Patterns](#wiring-patterns)\n\n---\n\n## Log Sources\n\n### loki.source.gelf — GELF UDP Log Source\n\nReceives GELF (Graylog Extended Log Format) logs over UDP. Common with Graylog, Docker GELF driver.\n\n```alloy\nloki.source.gelf \"my_gelf\" {\n  forward_to = [loki.write.default.receiver]\n}\n```\n\nDefault: listens on `0.0.0.0:12201` (UDP). Override with `use_incoming_timestamp = true`.\nLabels: Auto-extracts `__gelf_message_host`, `__gelf_message_level`, `__gelf_message_facility`.\nUse `loki.relabel` to promote `__gelf_*` labels.\n\n**Sample queries**: `{source=\"gelf\"}`, `{source=\"gelf\"} |= \"error\"`\n\n### loki.source.api — Loki Push API Endpoint\n\nAccepts logs via Loki-compatible HTTP push API. Use for centralized log gateways, TCP JSON ingestion.\n\n```alloy\nloki.source.api \"push\" {\n  http {\n    listen_address = \"0.0.0.0\"\n    listen_port    = 3500\n  }\n  forward_to = [loki.process.parse.receiver]\n}\n```\n\n**Sample queries**: `{source=\"push-api\"}`, `rate({source=\"push-api\"}[5m])`\n\n### loki.source.kafka — Kafka Log Consumer\n\nConsumes log messages from Apache Kafka topics.\n\n```alloy\nloki.source.kafka \"logs\" {\n  brokers       = [\"kafka:9092\"]\n  topics        = [\"app-logs\"]\n  consumer_group = \"alloy\"\n  forward_to    = [loki.process.parse.receiver]\n}\n```\n\nAuthentication: Add `authentication { type = \"sasl\" ... }` block for SASL/SCRAM.\n**Sample queries**: `{source=\"kafka\"}`, `{source=\"kafka\", topic=\"app-logs\"}`\n\n### loki.source.windowsevent — Windows Event Logs\n\nCollects Windows Event Log entries. Windows-only.\n\n```alloy\nloki.source.windowsevent \"events\" {\n  eventlog_name = \"Application\"\n  forward_to    = [loki.process.parse.receiver]\n}\n```\n\nCommon event logs: `\"Application\"`, `\"System\"`, `\"Security\"`.\n**Sample queries**: `{source=\"windowsevent\"}`, `{source=\"windowsevent\"} |= \"error\"`\n\n---\n\n## Log Processing\n\nInsert `loki.process` between source and `loki.write` for parsing, enrichment, and routing.\n\n### stage.json — JSON Field Extraction\n\n```alloy\nloki.process \"parse\" {\n  stage.json {\n    expressions = {\n      \"timestamp\"  = \"\",\n      \"level\"      = \"\",\n      \"message\"    = \"\",\n      \"request_id\" = \"context.request_id\",\n    }\n  }\n  forward_to = [loki.write.default.receiver]\n}\n```\n\nEmpty string `\"\"` extracts the top-level key matching the name. Dotted paths extract nested fields.\n\n### stage.labels — Promote Fields to Labels\n\n```alloy\n  stage.labels {\n    values = {\n      \"level\" = \"\",\n      \"service\" = \"\",\n    }\n  }\n```\n\nPromotes extracted fields to Loki index labels. Use sparingly — high-cardin"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":null,"editorialQuality":{"score":100,"threshold":65,"status":"thin","wordCount":1938,"uniquenessScore":44,"reasons":["uniqueness-below-45"]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-10T05:57:39.886Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-10T05:57:39.886Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T08:16:49.611Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}