diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json
index d40d5c6..aba1385 100644
--- a/.claude-plugin/marketplace.json
+++ b/.claude-plugin/marketplace.json
@@ -1,31 +1,31 @@
{
"plugins": [
{
- "version": "0.2.4",
+ "version": "0.3.0",
"name": "elastic-elasticsearch",
"source": "./plugins/elasticsearch",
"description": "Elasticsearch skills — ES|QL queries, data ingestion, security (authn, authz, audit), and troubleshooting"
},
{
- "version": "0.2.4",
+ "version": "0.3.0",
"name": "elastic-kibana",
"source": "./plugins/kibana",
"description": "Kibana skills — dashboards, alerting rules, connectors, Vega visualizations, Agent Builder, streams, and audit logging"
},
{
- "version": "0.2.4",
+ "version": "0.3.0",
"name": "elastic-observability",
"source": "./plugins/observability",
"description": "Elastic Observability skills — OpenTelemetry instrumentation and migration (.NET, Java, Python), LLM observability, log search, SLOs, and service health"
},
{
- "version": "0.2.4",
+ "version": "0.3.0",
"name": "elastic-security",
"source": "./plugins/security",
"description": "Elastic Security skills — alert triage, case management, detection rule management, and sample data generation"
},
{
- "version": "0.2.4",
+ "version": "0.3.0",
"name": "elastic-cloud",
"source": "./plugins/cloud",
"description": "Elastic Cloud skills — project setup, access management, network security, and Serverless project lifecycle"
diff --git a/.github/.release-manifest.json b/.github/.release-manifest.json
index 26b257d..2e73de4 100644
--- a/.github/.release-manifest.json
+++ b/.github/.release-manifest.json
@@ -1,7 +1,7 @@
{
- "version": "0.2.4",
+ "version": "0.3.0",
"source_repo": "elastic/agent-skills",
- "sandbox_sha": "b1798448ff8f0fd7ea371606a0e0b95d33a1d70a",
+ "sandbox_sha": "092aaa6a6c68a4f0553145c90c86ea157721ae2e",
"skills": {
"skills/cloud/access-management": {
"name": "cloud-access-management",
@@ -37,7 +37,7 @@
},
"skills/elasticsearch/elasticsearch-esql": {
"name": "elasticsearch-esql",
- "version": "0.1.1"
+ "version": "0.3.0"
},
"skills/elasticsearch/elasticsearch-file-ingest": {
"name": "elasticsearch-file-ingest",
@@ -59,6 +59,10 @@
"name": "kibana-alerting-rules",
"version": "0.1.0"
},
+ "skills/kibana/kibana-anomaly-detection": {
+ "name": "kibana-anomaly-detection",
+ "version": "0.2.0"
+ },
"skills/kibana/kibana-audit": {
"name": "kibana-audit",
"version": "0.1.0"
@@ -69,7 +73,7 @@
},
"skills/kibana/kibana-dashboards": {
"name": "kibana-dashboards",
- "version": "0.1.1"
+ "version": "0.1.2"
},
"skills/kibana/kibana-vega": {
"name": "kibana-vega",
@@ -103,6 +107,10 @@
"name": "observability-edot-python-migrate",
"version": "0.1.0"
},
+ "skills/observability/k8s-investigation": {
+ "name": "observability-k8s-investigation",
+ "version": "0.2.0"
+ },
"skills/observability/llm-obs": {
"name": "observability-llm-obs",
"version": "0.1.0"
diff --git a/.github/plugin/marketplace.json b/.github/plugin/marketplace.json
index 06a9bc0..d78d94b 100644
--- a/.github/plugin/marketplace.json
+++ b/.github/plugin/marketplace.json
@@ -11,31 +11,31 @@
"name": "elastic-elasticsearch",
"source": "./plugins/elasticsearch",
"description": "Elasticsearch skills — ES|QL queries, data ingestion, security (authn, authz, audit), and troubleshooting",
- "version": "0.2.4"
+ "version": "0.3.0"
},
{
"name": "elastic-kibana",
"source": "./plugins/kibana",
"description": "Kibana skills — dashboards, alerting rules, connectors, Vega visualizations, Agent Builder, streams, and audit logging",
- "version": "0.2.4"
+ "version": "0.3.0"
},
{
"name": "elastic-observability",
"source": "./plugins/observability",
"description": "Elastic Observability skills — OpenTelemetry instrumentation and migration (.NET, Java, Python), LLM observability, log search, SLOs, and service health",
- "version": "0.2.4"
+ "version": "0.3.0"
},
{
"name": "elastic-security",
"source": "./plugins/security",
"description": "Elastic Security skills — alert triage, case management, detection rule management, and sample data generation",
- "version": "0.2.4"
+ "version": "0.3.0"
},
{
"name": "elastic-cloud",
"source": "./plugins/cloud",
"description": "Elastic Cloud skills — project setup, access management, network security, and Serverless project lifecycle",
- "version": "0.2.4"
+ "version": "0.3.0"
}
]
}
diff --git a/CHANGELOG.md b/CHANGELOG.md
index 05f1872..4f4ea2a 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -1,5 +1,22 @@
# Changelog
+## v0.3.0
+
+### New Skills
+
+- `skills/kibana/kibana-anomaly-detection` (v0.2.0)
+- `skills/observability/k8s-investigation` (v0.2.0)
+
+### Updated Skills
+
+- `skills/elasticsearch/elasticsearch-esql` (v0.1.1 → v0.3.0)
+- `skills/elasticsearch/elasticsearch-onboarding` (v0.1.0)
+- `skills/kibana/kibana-dashboards` (v0.1.1 → v0.1.2)
+
+### Generated Artifacts
+
+- Regenerated README skill table
+
## v0.2.4
### Updated Skills
diff --git a/README.md b/README.md
index 5197a1c..0bd3b94 100644
--- a/README.md
+++ b/README.md
@@ -59,30 +59,31 @@ Skills in this repository focus on:
| [elasticsearch-audit](skills/elasticsearch/elasticsearch-audit/SKILL.md) | Enable, configure, and query Elasticsearch security audit logs. Use when the task involves audit logging setup, event filtering, or investigating security incidents like failed logins. | 0.1.0 | elastic |
| [elasticsearch-authn](skills/elasticsearch/elasticsearch-authn/SKILL.md) | Authenticate to Elasticsearch using native, file-based, LDAP/AD, SAML, OIDC, Kerberos, JWT, or certificate realms. Use when connecting with credentials, choosing a realm, or managing API keys. Assumes the target realms are already configured. | 0.1.0 | elastic |
| [elasticsearch-authz](skills/elasticsearch/elasticsearch-authz/SKILL.md) | Manage Elasticsearch RBAC: native users, roles, role mappings, document- and field-level security. Use when creating users or roles, assigning privileges, or mapping external realms like LDAP/SAML. | 0.1.1 | elastic |
-| [elasticsearch-esql](skills/elasticsearch/elasticsearch-esql/SKILL.md) | Execute ES\|QL (Elasticsearch Query Language) queries, use when the user wants to query Elasticsearch data, analyze logs, aggregate metrics, explore data, or create charts and dashboards from ES\|QL results. | 0.1.1 | elastic |
+| [elasticsearch-esql](skills/elasticsearch/elasticsearch-esql/SKILL.md) | Execute ES\|QL (Elasticsearch Query Language) queries, use when the user wants to query Elasticsearch data, analyze logs, aggregate metrics, explore data, or create charts and dashboards from ES\|QL results. | 0.3.0 | elastic |
| [elasticsearch-file-ingest](skills/elasticsearch/elasticsearch-file-ingest/SKILL.md) | Ingest and transform data files (CSV/JSON/Parquet/Arrow IPC) into Elasticsearch with stream processing and custom transforms. Use when loading files or batch importing data — not for reindexing, general ingest pipeline design, or bulk API patterns. | 0.2.0 | elastic |
-| [elasticsearch-onboarding](skills/elasticsearch/elasticsearch-onboarding/SKILL.md) | Help developers new to Elasticsearch get from zero to a working search experience. Guide them through understanding their intent, mapping their data, and building a search experience with best practices baked in. Use this when developers are new to Elasticsearch and need help getting started with their search use case. | 0.1.0 | elastic |
+| [elasticsearch-onboarding](skills/elasticsearch/elasticsearch-onboarding/SKILL.md) | Help developers new to Elasticsearch get from zero to a working search experience. Guide them through understanding their intent, mapping their data, and building a search experience with best practices baked in. Use this when the user shows intent to build search-related functionality, asks about Elasticsearch-related concepts for their use case, or expresses the need for help getting started with Elasticsearch. | 0.1.0 | elastic |
| [elasticsearch-security-troubleshooting](skills/elasticsearch/elasticsearch-security-troubleshooting/SKILL.md) | Diagnose and resolve Elasticsearch security errors: 401/403 failures, TLS problems, expired API keys, role mapping mismatches, and Kibana login issues. Use when the user reports a security error. | 0.1.0 | elastic |
-Kibana (7)
+Kibana (8)
| Skill | Description | Version | Author |
| ----- | ----------- | ------- | ------ |
| [kibana-agent-builder](skills/kibana/agent-builder/SKILL.md) | Create and manage Agent Builder agents and custom tools in Kibana. Use when asked to create, update, delete, test, or inspect agents or tools in Agent Builder. | 0.2.0 | elastic |
| [kibana-alerting-rules](skills/kibana/kibana-alerting-rules/SKILL.md) | Create and manage Kibana alerting rules via REST API or Terraform. Use when creating, updating, or managing rule lifecycle (enable, disable, mute, snooze) or rules-as-code workflows. | 0.1.0 | elastic |
+| [kibana-anomaly-detection](skills/kibana/kibana-anomaly-detection/SKILL.md) | Elastic ML anomaly detection skill — investigation/RCA, score explanation, job operations (create, datafeed, start/stop, results), and troubleshooting (missing docs, memory limits, datafeed health, lifecycle). Operates against Kibana Agent Builder MCP tools (`ad_*`) on `.ml-anomalies-*`, `.ml-config`, `.ml-notifications-*`, `.ml-annotations-*`. Use when answering "what broke?"/"which entity?"/RCA, "why is score high/low?"/renormalization, "datafeed stopped"/"memory limit", or any request to set up or configure an ML anomaly detection job. | 0.2.0 | elastic |
| [kibana-audit](skills/kibana/kibana-audit/SKILL.md) | Enable and configure Kibana audit logging for saved object access, logins, and space operations. Use when setting up Kibana audit, filtering events, or correlating Kibana and ES audit logs. | 0.1.0 | elastic |
| [kibana-connectors](skills/kibana/kibana-connectors/SKILL.md) | Create and manage Kibana connectors for Slack, PagerDuty, Jira, webhooks, and more via REST API or Terraform. Use when configuring third-party integrations or managing connectors as code. | 0.1.1 | elastic |
-| [kibana-dashboards](skills/kibana/kibana-dashboards/SKILL.md) | Create and manage Kibana Dashboards and visualizations. Use when you need to define dashboards and visualizations declaratively, version control them, or automate their deployment. | 0.1.1 | elastic |
+| [kibana-dashboards](skills/kibana/kibana-dashboards/SKILL.md) | Create and manage Kibana Dashboards and visualizations. Use when you need to define dashboards and visualizations declaratively, version control them, or automate their deployment. | 0.1.2 | elastic |
| [kibana-vega](skills/kibana/kibana-vega/SKILL.md) | Create Vega and Vega-Lite visualizations with ES\|QL data sources in Kibana. Use when building custom charts, dashboards, or programmatic panel layouts beyond standard Lens charts. | 0.1.0 | elastic |
| [kibana-streams](skills/kibana/streams/SKILL.md) | List, inspect, enable, disable, and resync Kibana Streams via the REST API. Use when the user needs stream details, ingest/query settings, queries, significant events, or attachments. | 0.1.0 | elastic |
-Observability (10)
+Observability (11)
| Skill | Description | Version | Author |
| ----- | ----------- | ------- | ------ |
@@ -92,6 +93,7 @@ Skills in this repository focus on:
| [observability-edot-java-migrate](skills/observability/edot-java-migrate/SKILL.md) | Migrate a Java application from the classic Elastic APM Java agent to the EDOT Java agent. Use when switching from elastic-apm-agent.jar to elastic-otel-javaagent.jar. | 0.1.1 | elastic |
| [observability-edot-python-instrument](skills/observability/edot-python-instrument/SKILL.md) | Instrument a Python application with the Elastic Distribution of OpenTelemetry (EDOT) Python agent for automatic tracing, metrics, and logs. Use when adding observability to a Python service that has no existing APM agent. | 0.1.0 | elastic |
| [observability-edot-python-migrate](skills/observability/edot-python-migrate/SKILL.md) | Migrate a Python application from the classic Elastic APM Python agent to the EDOT Python agent. Use when switching from elastic-apm to elastic-opentelemetry. | 0.1.0 | elastic |
+| [observability-k8s-investigation](skills/observability/k8s-investigation/SKILL.md) | Investigate Kubernetes workload, node, and control-plane issues using OTel telemetry (EDOT). Use when diagnosing pod failures (CrashLoopBackOff, OOMKilled, Error), node pressure, resource exhaustion, image pull failures, admission rejections, autoscaling anomalies, or correlating K8s state with application signals. OTel ingest path only — the legacy ECS Kubernetes integration shape is out of scope. | 0.2.0 | elastic |
| [observability-llm-obs](skills/observability/llm-obs/SKILL.md) | Monitor LLMs and agentic apps: performance, token/cost, response quality, and workflow orchestration. Use when the user asks about LLM monitoring, GenAI observability, or AI cost/quality. | 0.1.0 | elastic |
| [observability-logs-search](skills/observability/logs-search/SKILL.md) | Search and filter Observability logs using ES\|QL. Use when investigating log spikes, errors, or anomalies; getting volume and trends; or drilling into services or containers during incidents. | 0.2.0 | elastic |
| [observability-manage-slos](skills/observability/manage-slos/SKILL.md) | Create and manage SLOs in Elastic Observability using the Kibana API. Use when defining SLIs, setting error budgets, or managing SLO lifecycle. | 0.2.0 | elastic |
diff --git a/package.json b/package.json
index a80c778..9296b3c 100644
--- a/package.json
+++ b/package.json
@@ -1,5 +1,5 @@
{
- "version": "0.2.4",
+ "version": "0.3.0",
"name": "elastic-agent-skills",
"description": "Official Elastic agent skills for Elasticsearch, Kibana, Observability, Security, and Cloud",
"license": "Apache-2.0",
diff --git a/plugins/cloud/.claude-plugin/plugin.json b/plugins/cloud/.claude-plugin/plugin.json
index 94d0a40..0d7cb10 100644
--- a/plugins/cloud/.claude-plugin/plugin.json
+++ b/plugins/cloud/.claude-plugin/plugin.json
@@ -1,5 +1,5 @@
{
- "version": "0.2.4",
+ "version": "0.3.0",
"name": "elastic-cloud",
"description": "Elastic Cloud skills - project setup, access management, network security, and Serverless project lifecycle",
"author": {
diff --git a/plugins/cloud/plugin.json b/plugins/cloud/plugin.json
index 4fd4d78..4fe79ba 100644
--- a/plugins/cloud/plugin.json
+++ b/plugins/cloud/plugin.json
@@ -1,7 +1,7 @@
{
"name": "elastic-cloud",
"description": "Elastic Cloud skills - project setup, access management, network security, and Serverless project lifecycle",
- "version": "0.2.4",
+ "version": "0.3.0",
"author": {
"name": "Elastic",
"url": "https://www.elastic.co"
@@ -9,10 +9,5 @@
"repository": "https://github.com/elastic/agent-skills",
"homepage": "https://github.com/elastic/agent-skills/tree/main/plugins/cloud",
"license": "Apache-2.0",
- "keywords": [
- "cloud",
- "serverless",
- "elastic-cloud",
- "elastic"
- ]
+ "keywords": ["cloud", "serverless", "elastic-cloud", "elastic"]
}
diff --git a/plugins/elasticsearch/.claude-plugin/plugin.json b/plugins/elasticsearch/.claude-plugin/plugin.json
index 57313fa..9d65e4c 100644
--- a/plugins/elasticsearch/.claude-plugin/plugin.json
+++ b/plugins/elasticsearch/.claude-plugin/plugin.json
@@ -1,5 +1,5 @@
{
- "version": "0.2.4",
+ "version": "0.3.0",
"name": "elastic-elasticsearch",
"description": "Elasticsearch skills - ES|QL queries, data ingestion, security (authn, authz, audit), and troubleshooting",
"author": {
diff --git a/plugins/elasticsearch/plugin.json b/plugins/elasticsearch/plugin.json
index 09fb57b..51af6bd 100644
--- a/plugins/elasticsearch/plugin.json
+++ b/plugins/elasticsearch/plugin.json
@@ -1,7 +1,7 @@
{
"name": "elastic-elasticsearch",
"description": "Elasticsearch skills for GitHub Copilot — translate natural language to ES|QL, ingest data, manage security (authn, authz, audit), and troubleshoot clusters.",
- "version": "0.2.4",
+ "version": "0.3.0",
"author": {
"name": "Elastic",
"url": "https://www.elastic.co"
diff --git a/plugins/elasticsearch/skills/elasticsearch-esql/SKILL.md b/plugins/elasticsearch/skills/elasticsearch-esql/SKILL.md
index 175d901..4d03d81 100644
--- a/plugins/elasticsearch/skills/elasticsearch-esql/SKILL.md
+++ b/plugins/elasticsearch/skills/elasticsearch-esql/SKILL.md
@@ -6,7 +6,7 @@ description: >
charts and dashboards from ES|QL results.
metadata:
author: elastic
- version: 0.1.1
+ version: 0.3.0
---
# Elasticsearch ES|QL
@@ -120,11 +120,24 @@ node scripts/esql.js test
curl -s "$ELASTICSEARCH_URL//_settings/index.mode" -H "Authorization: ApiKey $ELASTICSEARCH_API_KEY"
```
+ For TSDS indices on 9.4+, prefer the in-language discovery commands `METRICS_INFO` and `TS_INFO` (both GA) over
+ inspecting mappings — they enumerate the metric catalogue and the dimension labels of each time series directly. Both
+ must follow `TS` and must precede `STATS`/`SORT`/`LIMIT`. See
+ [Time Series Queries](references/time-series-queries.md#metric-and-time-series-discovery).
+
+ ```bash
+ node scripts/esql.js raw "TS metrics-tsds | METRICS_INFO | SORT metric_name" --tsv
+ node scripts/esql.js raw "TS metrics-tsds | TS_INFO | KEEP metric_name, dimensions | SORT metric_name" --tsv
+ ```
+
3. **Choose the right ES|QL feature for the task**: Before writing queries, match the user's intent to the most
appropriate ES|QL feature. Prefer a single advanced query over multiple basic ones.
- "find patterns," "categorize," "group similar messages" → `CATEGORIZE(field)`
- "spike," "dip," "anomaly," "when did X change" → `CHANGE_POINT value ON key`
- "trend over time," "time series" → `STATS ... BY BUCKET(@timestamp, interval)` or `TS` for TSDB
+ - "PromQL", "Prometheus query/dashboard/alert", `sum by (instance) (...)`, label matchers like `{cluster="prod"}` →
+ `PROMQL` source command (9.4+ preview); see [PROMQL Command](references/promql-command.md). Prefer `TS` for native
+ ES|QL phrasing.
- "search," "find documents matching" → `MATCH` (default), `QSTR` (advanced boolean), `KQL` (Kibana migration). For
content/document relevance search, follow the [ES|QL Search Strategy](references/esql-search-strategy.md)
- "count," "average," "breakdown" → `STATS` with aggregation functions
@@ -134,6 +147,8 @@ node scripts/esql.js test
CIDR_MATCH), common templates, and ambiguity handling
- [Time Series Queries](references/time-series-queries.md) - **read before any TS query**: inner/outer aggregation
model, TBUCKET syntax, RATE constraints
+ - [PROMQL Command](references/promql-command.md) — **read before any PROMQL query**: options, output schema,
+ limitations, and `PROMQL` vs `TS` decision matrix (9.4+ preview)
- [ES|QL Complete Reference](references/esql-reference.md) - full syntax for all commands and functions
- [ES|QL Search Strategy](references/esql-search-strategy.md) — for content/document relevance search (retrieve →
fuse → rerank)
@@ -281,6 +296,25 @@ TS metrics-tsds
| SORT bucket
```
+**Time series with PromQL syntax (9.4+ preview):** Use the `PROMQL` source command when the user explicitly asks for
+PromQL, references Prometheus syntax (`sum by (instance) (...)`, label matchers like `{cluster="prod"}`), or is
+migrating a Prometheus dashboard or alert. The `PROMQL` command accepts standard PromQL with optional `index`, `step`,
+`buckets`, `start`, `end`, and `scrape_interval` options, and produces a table that the rest of the ES|QL pipeline can
+process. Range selectors are optional — when omitted, the window is `max(step, scrape_interval)`. Otherwise prefer `TS`
+(GA in 9.4). `PROMQL` does **not** support group modifiers, set operators (`or`/`and`/`unless`), or functions like
+`histogram_quantile`, `predict_linear`, and `label_join` — fall back to `TS` for those. See
+[PROMQL Command](references/promql-command.md) for the full reference.
+
+```esql
+// Adaptive Kibana query — date picker drives time range and step
+PROMQL index=metrics-* sum by (instance) (rate(http_requests_total))
+
+// Named result, post-processed with ES|QL
+PROMQL index=k8s step=1h bytes=(max by (cluster) (network.bytes_in))
+| STATS max_bytes = MAX(bytes) BY cluster
+| SORT cluster
+```
+
**Data enrichment with LOOKUP JOIN:** The basic `ON` clause matches fields by name in both indices
(`LOOKUP JOIN idx ON field_name`). When the join key has a different name in the source, use `RENAME` first to align
names. 9.2+ tech preview also supports expression predicates (`ON expr == expr`); see
@@ -351,6 +385,7 @@ For complete ES|QL syntax including all commands, functions, and operators, read
- [Query Patterns](references/query-patterns.md) - Natural language to ES|QL translation
- [Generation Tips](references/generation-tips.md) - Best practices for query generation
- [Time Series Queries](references/time-series-queries.md) - TS command, time series aggregation functions, TBUCKET
+- [PROMQL Command](references/promql-command.md) - PromQL source command for TSDS indices (9.4+ preview)
- [DSL to ES|QL Migration](references/dsl-to-esql-migration.md) - Convert Query DSL to ES|QL
- [Environment Setup](references/environment-setup.md) - Connection configuration options
diff --git a/plugins/elasticsearch/skills/elasticsearch-esql/references/dsl-to-esql-migration.md b/plugins/elasticsearch/skills/elasticsearch-esql/references/dsl-to-esql-migration.md
index e9ae400..e138a91 100644
--- a/plugins/elasticsearch/skills/elasticsearch-esql/references/dsl-to-esql-migration.md
+++ b/plugins/elasticsearch/skills/elasticsearch-esql/references/dsl-to-esql-migration.md
@@ -958,28 +958,29 @@ FROM sales
Features not available in ES|QL as of version 9.3:
-| Feature | Query DSL | ES\|QL |
-| ---------------------------- | --------- | ----------------------------------- |
-| Highlighting | ✅ | ❌ |
-| Nested queries | ✅ | ❌ |
-| Parent-child queries | ✅ | ❌ |
-| Scroll/pagination beyond 10k | ✅ | ❌ |
-| Percolate queries | ✅ | ❌ |
-| Complex boosting | ✅ | Limited |
-| Geo distance sorting | ✅ | ❌ |
-| Runtime fields | ✅ | Use EVAL |
-| Suggest API | ✅ | ❌ |
-| Collapse (field collapsing) | ✅ | ❌ |
-| Inner hits | ✅ | ❌ |
-| Timezone in date functions | ✅ | ❌ (UTC only) |
-| JOIN (non-lookup) | N/A | ❌ (only LEFT JOIN on lookup index) |
+| Feature | Query DSL | ES\|QL |
+| ---------------------------- | --------- | ----------------------------------------- |
+| Highlighting | ✅ | ❌ |
+| Nested queries | ✅ | ❌ |
+| Parent-child queries | ✅ | ❌ |
+| Scroll/pagination beyond 10k | ✅ | ❌ |
+| Percolate queries | ✅ | ❌ |
+| Complex boosting | ✅ | Limited |
+| Geo distance sorting | ✅ | ❌ |
+| Runtime fields | ✅ | Use EVAL |
+| Suggest API | ✅ | ❌ |
+| Collapse (field collapsing) | ✅ | ❌ |
+| Inner hits | ✅ | ❌ |
+| Timezone support | ✅ | ✅ `SET time_zone` (Serverless GA) |
+| JOIN (non-lookup) | N/A | ❌ (only LEFT JOIN on lookup index) |
+| Subqueries / UNION ALL | N/A | ✅ `FROM` subqueries (Serverless preview) |
### Unsupported Field Types in ES|QL
- `nested`
- `binary`
- `completion`
-- `flattened`
+- `flattened` (use `METADATA _source` + `JSON_EXTRACT` to access sub-keys)
- Range types (`date_range`, `integer_range`, etc.)
- `rank_feature`, `rank_features`
- `search_as_you_type`
diff --git a/plugins/elasticsearch/skills/elasticsearch-esql/references/esql-reference.md b/plugins/elasticsearch/skills/elasticsearch-esql/references/esql-reference.md
index 662323a..3597fac 100644
--- a/plugins/elasticsearch/skills/elasticsearch-esql/references/esql-reference.md
+++ b/plugins/elasticsearch/skills/elasticsearch-esql/references/esql-reference.md
@@ -52,23 +52,26 @@ Query directives modify the behavior of an ES|QL query. They appear before the s
### SET (9.3+, tech preview)
-Controls query-level settings.
+Controls query-level settings. Every `SET` directive must end with a semicolon before the source command.
**Syntax:**
```esql
-SET setting = value; [SET settingN = valueN;]
+SET setting = "value"; [SET setting = "value";]
source-command
| processing-commands
```
**`unmapped_fields`** (9.3+ preview) -- controls how unmapped fields are treated:
-- `FAIL` (default) -- the query fails if it references unmapped fields
-- `NULLIFY` -- treats unmapped fields as null values
+- `"default"` / `"fail"` -- the query fails if it references unmapped fields
+- `"nullify"` -- treats unmapped fields as null values
+- `"load"` -- loads unmapped fields dynamically. **Limitation:** `"load"` is incompatible with subqueries and views. Use
+ `"nullify"` when composing subqueries or querying views.
**`time_zone`** (Serverless GA; self-managed planned) -- sets the default timezone for the query, overriding UTC
-default.
+default. Accepts any IANA timezone string or UTC offset. Applies to all date/time operations: `DATE_TRUNC`,
+`DATE_FORMAT`, `NOW()`, etc.
**Examples:**
@@ -79,6 +82,12 @@ FROM employees
| SORT emp_no
| LIMIT 1
+SET time_zone = "America/Los_Angeles";
+FROM error_triage
+| EVAL hour = DATE_TRUNC(1 hour, @timestamp)
+| STATS errors = COUNT(*) BY hour, service
+| SORT hour DESC
+
SET time_zone = "+05:00";
TS k8s
| WHERE @timestamp == "2024-05-10T00:04:49.000Z"
@@ -86,7 +95,11 @@ TS k8s
```
> **When to use:** `unmapped_fields` is useful when querying across multiple indices where some indices may not have all
-> fields mapped. `time_zone` shifts date functions and display to a non-UTC zone.
+> fields mapped. `time_zone` shifts date functions and display to a non-UTC zone. There is no per-function timezone
+> argument — `DATE_TRUNC(1 hour, @timestamp, "America/Los_Angeles")` does **not** work.
+>
+> **Restriction:** `SET` directives cannot be used inside view definitions. The caller must apply `SET` when querying
+> the view.
---
@@ -123,6 +136,33 @@ FROM
FROM cluster_one:logs-*, cluster_two:logs-*
```
+**Subqueries (Serverless tech preview):** `FROM` supports parenthesized subqueries with UNION ALL semantics. Each branch
+is a complete ES|QL pipeline. Columns present in one branch but not another are filled with `null`.
+
+```esql
+// Combine logs from different indices with independent pipelines
+FROM
+ (FROM web_logs
+ | WHERE status_code >= 500
+ | KEEP @timestamp, message, service.name),
+ (FROM app_logs
+ | WHERE level == "error"
+ | KEEP @timestamp, message, service.name)
+| STATS errors = COUNT(*) BY service.name
+
+// Mix bare index patterns and subqueries
+FROM raw_index, (FROM other_index | WHERE active == true | KEEP id, name)
+```
+
+**Subquery constraints:**
+
+- Non-correlated only — branches cannot reference columns from the outer query
+- Columns with the same name must have compatible types across branches
+- `FORK` cannot be used inside or after subqueries
+- `SET unmapped_fields="load"` is incompatible with subqueries
+
+**Subqueries vs FORK:** Different data sources → subqueries. Same data, different analyses → FORK.
+
**Note:** Without explicit `LIMIT`, queries default to 1000 rows (or whatever the cluster setting
esql.query.result_truncation_default_size is set to).
@@ -147,7 +187,7 @@ ROW greeting = "hello", pi = 3.14159
### TS
Retrieves data from time series data streams (TSDS). Similar to `FROM` but enables time series aggregation functions in
-`STATS` and targets only time series indices. Available since 9.2.
+`STATS` and targets only time series indices. **Preview from 9.2 to 9.3, GA since 9.4**; GA on Elastic Cloud Serverless.
**Syntax:**
@@ -161,6 +201,11 @@ TS index_pattern [METADATA fields]
- Time series functions are evaluated per time series first, then aggregated by group using an outer function
- If no inner time series function is specified, `LAST_OVER_TIME()` is assumed implicitly
- Cannot be combined with `FORK` before `STATS` is applied
+- When the query has no `STATS`, `TS` returns rows sorted by `@timestamp` descending by default
+- When the first `STATS` after `TS` uses a **bare** time series function (not wrapped in an outer aggregation like
+ `AVG()` / `SUM()`), results are implicitly grouped by every dimension and include a `_timeseries` JSON column. Use
+ `BY WITHOUT(dim, ...)` (GA in 9.4) to narrow this grouping. Bare dimension columns in `BY` are rejected; only grouping
+ functions (`TBUCKET`, `WITHOUT`) are allowed alongside a bare time series function.
**Examples:**
@@ -177,6 +222,11 @@ TS metrics
// Average of per-time-series averages (explicit inner function)
TS metrics
| STATS AVG(AVG_OVER_TIME(memory_usage))
+
+// Bare time series function — group by every dimension except `pod` (9.4+ GA)
+TS k8s
+| STATS total_cost = SUM(network.cost) BY WITHOUT(pod)
+| SORT total_cost
```
**Best practices:**
@@ -185,6 +235,69 @@ TS metrics
- Use `TS` instead of `FROM` for aggregations on time series data
- Avoid aggregating metrics with different dimensional cardinalities in the same query
+### PROMQL
+
+Queries time series data streams (TSDS) using **Prometheus Query Language (PromQL)** instead of ES|QL syntax. Like `TS`,
+it produces a table that the rest of the ES|QL pipeline can process. Available since **9.4 (preview)** and on Elastic
+Cloud Serverless. See [promql-command.md](promql-command.md) for the full reference.
+
+**Syntax:**
+
+```esql
+PROMQL [ ... ] [ = ] ( )
+```
+
+**Options:**
+
+- `index` — indices/streams/aliases (default `metrics-*`)
+- `step` — query resolution step width
+- `buckets` — target bucket count for auto-step (default `100`, mutually exclusive with `step`)
+- `start`, `end` — explicit time range (defaults to Kibana date picker, otherwise unrestricted)
+- `scrape_interval` — expected metric collection interval (default `1m`); used for the implicit range selector window
+- `=( ... )` — name the metric output column
+
+**Output columns:**
+
+- The PromQL expression (or ``) as `double` — the metric value
+- `step` (`date`) — timestamp for each evaluation step
+- One `keyword` column per `by`/`without` grouping label, or a single `_timeseries` JSON column when there is no
+ cross-series aggregation
+
+**Examples:**
+
+```esql
+// Fully adaptive Kibana query — date picker drives time range and step
+PROMQL index=metrics-* sum by (instance) (rate(http_requests_total))
+
+// Explicit range query with a named result column
+PROMQL index=k8s step=1h cost=(max by (cluster) (network.total_bytes_in{cluster!="prod"}))
+| SORT cluster
+
+// Post-process with ES|QL after the PROMQL stage
+PROMQL index=k8s step=1h bytes=(max by (cluster) (network.bytes_in))
+| STATS max_bytes = MAX(bytes) BY cluster
+| SORT cluster
+
+// Enrich PromQL results with a lookup index
+PROMQL index=metrics-*
+ http_rate=(sum by (instance) (rate(http_requests_total)))
+| LOOKUP JOIN instance_metadata ON instance
+```
+
+**Implicit range selectors:** Range vector functions can omit the range selector (`rate(http_requests_total)` instead of
+`rate(http_requests_total[5m])`); the engine uses `max(step, scrape_interval)` as the window. This makes the query scale
+with the date picker.
+
+**Limitations (9.4 preview):**
+
+- Group modifiers (`on(...) group_left(...)`) are not supported
+- Set operators (`or`, `and`, `unless`) are not supported
+- Some PromQL functions are not available, including `histogram_quantile`, `predict_linear`, and `label_join`
+- Time buckets align to fixed calendar boundaries rather than the query start time, which can cause slight differences
+ from native Prometheus for short ranges or large step sizes
+
+When any of these are required, use the [`TS` command](#ts) and express the equivalent computation in ES|QL.
+
### SHOW
Returns information about the deployment.
@@ -464,12 +577,13 @@ FROM data
### LIMIT
-Limits the number of rows returned.
+Limits the number of rows returned. Supports optional grouped top-N with `BY` (Serverless).
**Syntax:**
```esql
LIMIT number
+LIMIT number BY field
```
**Examples:**
@@ -478,8 +592,16 @@ LIMIT number
FROM logs-*
| SORT @timestamp DESC
| LIMIT 100
+
+// Grouped top-N: keep top 3 rows per service after sorting
+FROM app_logs
+| STATS cnt = COUNT(*) BY service, level
+| SORT cnt DESC
+| LIMIT 3 BY service
```
+> **Note:** In `LIMIT n BY field`, the number comes **before** `BY`. `LIMIT BY field n` does not parse.
+
### DISSECT
Extracts structured fields from a string using a pattern.
@@ -857,65 +979,242 @@ FROM data
| STATS count = COUNT(*) BY tags
```
-### URI_PARTS (Planned)
+### METRICS_INFO
+
+Returns one row per distinct metric available in the targeted time series data stream(s), with applicable dimensions and
+metadata. Use it to discover the metric catalogue without inspecting index mappings or calling the field capabilities
+API. **GA since 9.4** (and on Elastic Cloud Serverless).
+
+**Syntax:**
+
+```esql
+METRICS_INFO
+```
+
+Takes no parameters.
+
+**Output columns** (all `keyword`):
+
+- `metric_name` — the metric field name (single-valued)
+- `data_stream` — data stream(s) containing this metric (multi-valued when several streams align on
+ unit/metric_type/field_type)
+- `unit` — declared unit from field mapping (e.g., `bytes`, `packets`); may be `null` or multi-valued
+- `metric_type` — `counter`, `gauge`, etc. (multi-valued when definitions differ across backing indices)
+- `field_type` — Elasticsearch field type (e.g., `long`, `double`, `integer`)
+- `dimension_fields` — union of dimension field names across all time series for that metric
+
+**Restrictions:**
+
+- Can only be used after a `TS` source command — `FROM | METRICS_INFO` is rejected.
+- Must appear before pipeline-breaking commands (`STATS`, `SORT`, `LIMIT`).
+- The output replaces the original table — downstream commands operate on the metadata rows, not the raw documents.
+
+**Examples:**
+
+```esql
+// List every metric in a TSDS, alphabetically
+TS k8s
+| METRICS_INFO
+| SORT metric_name
+
+// Narrow to metrics that have data matching a filter, then keep only key columns
+TS k8s
+| WHERE cluster == "prod"
+| METRICS_INFO
+| KEEP metric_name, metric_type
+| SORT metric_name
+
+// Count metrics by type
+TS k8s
+| METRICS_INFO
+| STATS metric_count = COUNT(*) BY metric_type
+| SORT metric_type
+
+// Find metrics matching a name pattern
+TS k8s
+| METRICS_INFO
+| WHERE metric_name LIKE "network.eth0*"
+| SORT metric_name
+```
+
+### TS_INFO
+
+Returns one row per (metric, time series) combination in the targeted TSDS, including the dimension key/value pairs that
+identify each series. Use it to enumerate the actual time series — and their labels — that exist for each metric. **GA
+since 9.4** (and on Elastic Cloud Serverless).
+
+**Syntax:**
+
+```esql
+TS_INFO
+```
+
+Takes no parameters.
+
+**Output columns** (all `keyword`):
+
+- All columns from `METRICS_INFO` (`metric_name`, `data_stream`, `unit`, `metric_type`, `field_type`,
+ `dimension_fields`)
+- `dimensions` — JSON-encoded object with the dimension key/value pairs identifying the time series, e.g.
+ `{"job":"elasticsearch","instance":"instance_1"}`. Single-valued.
+
+**Restrictions:**
+
+- Can only be used after a `TS` source command — `FROM | TS_INFO` is rejected.
+- Must appear before pipeline-breaking commands (`STATS`, `SORT`, `LIMIT`).
+- The output replaces the original table — downstream commands operate on the metadata rows, not the raw documents.
+
+**Examples:**
+
+```esql
+// Every (metric, time series) pair in a TSDS
+TS k8s
+| TS_INFO
+| SORT metric_name, dimensions
+
+// Restrict to series with data matching a filter, keep only key columns
+TS k8s
+| WHERE cluster == "prod"
+| TS_INFO
+| KEEP metric_name, dimensions
+| SORT metric_name, dimensions
+
+// Filter by metadata after TS_INFO
+TS k8s
+| TS_INFO
+| WHERE metric_type == "gauge"
+| SORT metric_name, dimensions
+
+// Count distinct time series per metric
+TS k8s
+| TS_INFO
+| STATS series_count = COUNT(*) BY metric_name
+| SORT metric_name
+
+// Count distinct metrics per time series — useful to spot under- or over-reporting series
+TS k8s
+| TS_INFO
+| STATS metric_count = COUNT_DISTINCT(metric_name) BY dimensions
+| SORT dimensions
+```
+
+> **`METRICS_INFO` vs `TS_INFO`:** `METRICS_INFO` returns one row **per distinct metric**; `TS_INFO` returns one row
+> **per (metric, time series) combination** and adds a `dimensions` column with the labels identifying each series. Use
+> `METRICS_INFO` to enumerate _what_ is being measured, and `TS_INFO` to enumerate _which_ time series exist.
+
+### URI_PARTS (Serverless)
+
+Pipe command that parses a URI string into structured columns. A target prefix is **required**.
+
+**Syntax:**
+
+```esql
+URI_PARTS target = field
+```
+
+**Output columns:** `target.domain`, `target.path`, `target.scheme`, `target.extension`, `target.port`, `target.query`,
+`target.fragment`, `target.user_info`, `target.username`, `target.password`.
-Parses a URI string and extracts its components (domain, path, port, query, scheme, etc.) into new columns. Not yet
-released.
+**Example:**
+
+```esql
+FROM web_logs
+| WHERE http.response.status_code >= 400
+| URI_PARTS parts = url.full
+| STATS errors = COUNT(*) BY parts.domain, parts.path
+| SORT errors DESC
+```
+
+### USER_AGENT (Serverless)
+
+Pipe command that parses a user agent string into structured columns. A target prefix is **required**.
**Syntax:**
```esql
-URI_PARTS prefix = expression
+USER_AGENT target = field
```
+**Output columns:** `target.name`, `target.version`, `target.os.name`, `target.os.version`, `target.os.full`,
+`target.device.name`.
+
**Example:**
```esql
FROM web_logs
-| URI_PARTS url_parts = request_url
-| KEEP url_parts.domain, url_parts.path, url_parts.query
+| USER_AGENT ua = user_agent.original
+| STATS cnt = COUNT(*) BY ua.name, ua.version
+```
+
+### REGISTERED_DOMAIN (Serverless)
+
+Pipe command that extracts the registered domain, top-level domain, and subdomain from a hostname. A target prefix is
+**required**.
+
+**Syntax:**
+
+```esql
+REGISTERED_DOMAIN target = field
```
+**Output columns:** `target.domain` (full input), `target.registered_domain`, `target.top_level_domain`,
+`target.subdomain`.
+
+**Example:**
+
+```esql
+FROM dns_logs
+| REGISTERED_DOMAIN rd = dns.question.name
+| STATS queries = COUNT(*) BY rd.registered_domain
+| SORT queries DESC
+```
+
+> **Note:** `URI_PARTS`, `USER_AGENT`, and `REGISTERED_DOMAIN` are **pipe commands** (like `DISSECT`/`GROK`), not scalar
+> functions. The syntax `URI_PARTS(field)` does not work — use `| URI_PARTS target = field`.
+
---
## Aggregate Functions
Used with STATS command.
-| Function | Description | Example |
-| ---------------------------------- | ------------------------------------------------- | ------------------------------------------------ |
-| `COUNT(*)` | Count all rows | `STATS n = COUNT(*)` |
-| `COUNT(field)` | Count non-null values | `STATS n = COUNT(status)` |
-| `COUNT_DISTINCT(field)` | Count unique values | `STATS unique = COUNT_DISTINCT(user_id)` |
-| `SUM(field)` | Sum of values | `STATS total = SUM(amount)` |
-| `AVG(field)` | Average | `STATS avg_price = AVG(price)` |
-| `MIN(field)` | Minimum value | `STATS min_temp = MIN(temperature)` |
-| `MAX(field)` | Maximum value | `STATS max_score = MAX(score)` |
-| `MEDIAN(field)` | Median value | `STATS med = MEDIAN(response_time)` |
-| `PERCENTILE(field, p)` | Percentile | `STATS p95 = PERCENTILE(latency, 95)` |
-| `STD_DEV(field)` | Standard deviation | `STATS sd = STD_DEV(values)` |
-| `VARIANCE(field)` | Variance | `STATS var = VARIANCE(values)` |
-| `VALUES(field)` | Collect all values | `STATS all_tags = VALUES(tag)` |
-| `TOP(field, n, order)` | Top N values | `STATS top3 = TOP(score, 3, "desc")` |
-| `WEIGHTED_AVG(val, weight)` | Weighted average | `STATS wavg = WEIGHTED_AVG(score, weight)` |
-| `MEDIAN_ABSOLUTE_DEVIATION(field)` | Robust variability measure | `STATS mad = MEDIAN_ABSOLUTE_DEVIATION(latency)` |
-| `ABSENT(field)` | True if no non-null values (9.2+) | `STATS is_absent = ABSENT(error_code)` |
-| `PRESENT(field)` | True if any non-null values (9.2+) | `STATS has_data = PRESENT(metric)` |
-| `SAMPLE(field, n)` | Collect n sample values (8.19/9.1+) | `STATS examples = SAMPLE(message, 5)` |
-| `FIRST(field, sort_field)` | Earliest value by sort field (Serverless preview) | `STATS earliest = FIRST(message, @timestamp)` |
-| `LAST(field, sort_field)` | Latest value by sort field (Serverless preview) | `STATS latest = LAST(message, @timestamp)` |
-| `ST_CENTROID_AGG(field)` | Spatial centroid of points | `STATS center = ST_CENTROID_AGG(location)` |
-| `ST_EXTENT_AGG(field)` | Bounding box of geometries (8.18/9.0+, preview) | `STATS bbox = ST_EXTENT_AGG(location)` |
+| Function | Description | Example |
+| ---------------------------------- | ----------------------------------------------- | ------------------------------------------------ |
+| `COUNT(*)` | Count all rows | `STATS n = COUNT(*)` |
+| `COUNT(field)` | Count non-null values | `STATS n = COUNT(status)` |
+| `COUNT_DISTINCT(field)` | Count unique values | `STATS unique = COUNT_DISTINCT(user_id)` |
+| `SUM(field)` | Sum of values | `STATS total = SUM(amount)` |
+| `AVG(field)` | Average | `STATS avg_price = AVG(price)` |
+| `MIN(field)` | Minimum value | `STATS min_temp = MIN(temperature)` |
+| `MAX(field)` | Maximum value | `STATS max_score = MAX(score)` |
+| `MEDIAN(field)` | Median value | `STATS med = MEDIAN(response_time)` |
+| `PERCENTILE(field, p)` | Percentile | `STATS p95 = PERCENTILE(latency, 95)` |
+| `STD_DEV(field)` | Standard deviation | `STATS sd = STD_DEV(values)` |
+| `VARIANCE(field)` | Variance | `STATS var = VARIANCE(values)` |
+| `VALUES(field)` | Collect all values (GA) | `STATS all_tags = VALUES(tag)` |
+| `TOP(field, n, order)` | Top N values | `STATS top3 = TOP(score, 3, "desc")` |
+| `WEIGHTED_AVG(val, weight)` | Weighted average | `STATS wavg = WEIGHTED_AVG(score, weight)` |
+| `MEDIAN_ABSOLUTE_DEVIATION(field)` | Robust variability measure | `STATS mad = MEDIAN_ABSOLUTE_DEVIATION(latency)` |
+| `ABSENT(field)` | True if no non-null values (9.2+) | `STATS is_absent = ABSENT(error_code)` |
+| `PRESENT(field)` | True if any non-null values (9.2+) | `STATS has_data = PRESENT(metric)` |
+| `SAMPLE(field, n)` | Collect n sample values (8.19/9.1+) | `STATS examples = SAMPLE(message, 5)` |
+| `FIRST(field, sort_field)` | Earliest value by sort field (Serverless GA) | `STATS earliest = FIRST(message, @timestamp)` |
+| `LAST(field, sort_field)` | Latest value by sort field (Serverless GA) | `STATS latest = LAST(message, @timestamp)` |
+| `EARLIEST(field)` | Earliest value (single-arg; Serverless GA) | `STATS e = EARLIEST(@timestamp)` |
+| `LATEST(field)` | Latest value (single-arg; Serverless GA) | `STATS l = LATEST(@timestamp)` |
+| `ST_CENTROID_AGG(field)` | Spatial centroid of points | `STATS center = ST_CENTROID_AGG(location)` |
+| `ST_EXTENT_AGG(field)` | Bounding box of geometries (8.18/9.0+, preview) | `STATS bbox = ST_EXTENT_AGG(location)` |
### Grouping Functions
Used in the `BY` clause of `STATS` and `INLINE STATS` to create dynamic groups.
-| Function | Description | Example |
-| --------------------- | ------------------------------------------------- | ----------------------------------------------------- |
-| `BUCKET(field, size)` | Create fixed-size buckets for numbers or dates | `STATS count = COUNT(*) BY b = BUCKET(price, 10)` |
-| `TBUCKET(interval)` | Time-based bucketing (9.2+, for use with `TS`) | `STATS SUM(RATE(reqs)) BY TBUCKET(1 hour)` |
-| `CATEGORIZE(field)` | Auto-categorize text values (8.18/9.0+, Platinum) | `STATS count = COUNT(*) BY cat = CATEGORIZE(message)` |
+| Function | Description | Example |
+| --------------------- | -------------------------------------------------------------------- | ----------------------------------------------------- |
+| `BUCKET(field, size)` | Create fixed-size buckets for numbers or dates | `STATS count = COUNT(*) BY b = BUCKET(price, 10)` |
+| `TBUCKET(interval)` | Time-based bucketing (preview 9.2-9.3, GA in 9.4) | `STATS SUM(RATE(reqs)) BY TBUCKET(1 hour)` |
+| `WITHOUT(dim, ...)` | Group time series by every dimension except those listed (GA in 9.4) | `STATS total = SUM(network.cost) BY WITHOUT(pod)` |
+| `CATEGORIZE(field)` | Auto-categorize text values (8.18/9.0+, Platinum) | `STATS count = COUNT(*) BY cat = CATEGORIZE(message)` |
**CATEGORIZE options (9.2+):**
@@ -951,7 +1250,18 @@ FROM logs-*
Used with the `STATS` command after a `TS` source command. These functions evaluate per time series first, then
aggregate by group using an outer function (e.g., `SUM`, `AVG`). An optional second argument specifies a sliding time
-window. Available since 9.2.
+window.
+
+**Availability:** **All** time series aggregation functions are **GA since 9.4** — both the 9.2-introduced set (`RATE`,
+`IRATE`, `INCREASE`, `DELTA`, `IDELTA`, `AVG_OVER_TIME`, `SUM_OVER_TIME`, `MIN_OVER_TIME`, `MAX_OVER_TIME`,
+`FIRST_OVER_TIME`, `LAST_OVER_TIME`, `COUNT_OVER_TIME`, `COUNT_DISTINCT_OVER_TIME`, `PRESENT_OVER_TIME`,
+`ABSENT_OVER_TIME`) and the 9.3-introduced set (`DERIV`, `PERCENTILE_OVER_TIME`, `STDDEV_OVER_TIME`,
+`VARIANCE_OVER_TIME`). On clusters in 9.2-9.3 these functions are still in tech preview.
+
+**Sliding window parameter (second argument):** in 9.2-9.3 (preview) the window must be a multiple of the `TBUCKET`
+interval; **9.4+ (GA)** accepts arbitrary durations, with performance optimizations when the window is a multiple of the
+bucket interval. Within a single query, you cannot mix windows smaller than the bucket interval for one metric with
+windows larger than the bucket interval for another metric.
| Function | Description | Metric Types |
| ----------------------------------- | ------------------------------ | -------------- |
@@ -998,40 +1308,53 @@ TS metrics
## String Functions
-| Function | Description | Example |
-| --------------------------- | --------------------------------------------------- | -------------------------------------------------------------------- |
-| `LENGTH(s)` | String length | `EVAL len = LENGTH(name)` |
-| `CONCAT(s1, s2, ...)` | Concatenate strings | `EVAL full = CONCAT(first, " ", last)` |
-| `SUBSTRING(s, start, len)` | Extract substring | `EVAL sub = SUBSTRING(text, 1, 10)` |
-| `LEFT(s, n)` | Left n characters | `EVAL l = LEFT(text, 5)` |
-| `RIGHT(s, n)` | Right n characters | `EVAL r = RIGHT(text, 5)` |
-| `TRIM(s)` | Remove whitespace | `EVAL clean = TRIM(input)` |
-| `LTRIM(s)` | Trim left | `EVAL clean = LTRIM(input)` |
-| `RTRIM(s)` | Trim right | `EVAL clean = RTRIM(input)` |
-| `TO_UPPER(s)` | Uppercase | `EVAL upper = TO_UPPER(name)` |
-| `TO_LOWER(s)` | Lowercase | `EVAL lower = TO_LOWER(name)` |
-| `REPLACE(s, old, new)` | Replace text | `EVAL fixed = REPLACE(msg, "err", "error")` |
-| `SPLIT(s, delim)` | Split into array | `EVAL parts = SPLIT(path, "/")` |
-| `STARTS_WITH(s, prefix)` | Check prefix | `WHERE STARTS_WITH(url, "https")` |
-| `ENDS_WITH(s, suffix)` | Check suffix | `WHERE ENDS_WITH(file, ".log")` |
-| `CONTAINS(s, substr)` | Check contains | `WHERE CONTAINS(message, "error")` |
-| `LOCATE(substr, s)` | Find position | `EVAL pos = LOCATE("@", email)` |
-| `REVERSE(s)` | Reverse string | `EVAL rev = REVERSE(text)` |
-| `REPEAT(s, n)` | Repeat string | `EVAL sep = REPEAT("-", 10)` |
-| `SPACE(n)` | N spaces | `EVAL spaces = SPACE(5)` |
-| `BIT_LENGTH(s)` | Bit length (8.17+) | `EVAL bits = BIT_LENGTH(name)` |
-| `BYTE_LENGTH(s)` | Byte length (8.17+) | `EVAL bytes = BYTE_LENGTH(name)` |
-| `CHUNK(field, settings)` | Split text into chunks (9.3+, preview) | `EVAL chunks = CHUNK(body, {"strategy":"word","max_chunk_size":50})` |
-| `HASH(alg, s)` | Hash string (8.18/9.0+) | `EVAL h = HASH("SHA-256", msg)` |
-| `MD5(s)` | MD5 hash (8.18/9.0+) | `EVAL h = MD5(content)` |
-| `SHA1(s)` | SHA-1 hash (8.18/9.0+) | `EVAL h = SHA1(content)` |
-| `SHA256(s)` | SHA-256 hash (8.18/9.0+) | `EVAL h = SHA256(content)` |
-| `FROM_BASE64(s)` | Decode base64 | `EVAL decoded = FROM_BASE64(encoded)` |
-| `TO_BASE64(s)` | Encode to base64 | `EVAL encoded = TO_BASE64(data)` |
-| `URL_DECODE(s)` | URL-decode (9.2+) | `EVAL decoded = URL_DECODE(url)` |
-| `URL_ENCODE(s)` | URL-encode (9.2+) | `EVAL encoded = URL_ENCODE(text)` |
-| `URL_ENCODE_COMPONENT(s)` | URL-encode for URI components (9.2+) | `EVAL encoded = URL_ENCODE_COMPONENT(text)` |
-| `JSON_EXTRACT(field, path)` | Extract value from JSON string (Serverless preview) | `EVAL name = JSON_EXTRACT(raw, "$.user.name")` |
+| Function | Description | Example |
+| --------------------------- | ---------------------------------------------- | -------------------------------------------------------------------- |
+| `LENGTH(s)` | String length | `EVAL len = LENGTH(name)` |
+| `CONCAT(s1, s2, ...)` | Concatenate strings | `EVAL full = CONCAT(first, " ", last)` |
+| `SUBSTRING(s, start, len)` | Extract substring | `EVAL sub = SUBSTRING(text, 1, 10)` |
+| `LEFT(s, n)` | Left n characters | `EVAL l = LEFT(text, 5)` |
+| `RIGHT(s, n)` | Right n characters | `EVAL r = RIGHT(text, 5)` |
+| `TRIM(s)` | Remove whitespace | `EVAL clean = TRIM(input)` |
+| `LTRIM(s)` | Trim left | `EVAL clean = LTRIM(input)` |
+| `RTRIM(s)` | Trim right | `EVAL clean = RTRIM(input)` |
+| `TO_UPPER(s)` | Uppercase | `EVAL upper = TO_UPPER(name)` |
+| `TO_LOWER(s)` | Lowercase | `EVAL lower = TO_LOWER(name)` |
+| `REPLACE(s, old, new)` | Replace text | `EVAL fixed = REPLACE(msg, "err", "error")` |
+| `SPLIT(s, delim)` | Split into array | `EVAL parts = SPLIT(path, "/")` |
+| `STARTS_WITH(s, prefix)` | Check prefix | `WHERE STARTS_WITH(url, "https")` |
+| `ENDS_WITH(s, suffix)` | Check suffix | `WHERE ENDS_WITH(file, ".log")` |
+| `CONTAINS(s, substr)` | Check contains | `WHERE CONTAINS(message, "error")` |
+| `LOCATE(substr, s)` | Find position | `EVAL pos = LOCATE("@", email)` |
+| `REVERSE(s)` | Reverse string | `EVAL rev = REVERSE(text)` |
+| `REPEAT(s, n)` | Repeat string | `EVAL sep = REPEAT("-", 10)` |
+| `SPACE(n)` | N spaces | `EVAL spaces = SPACE(5)` |
+| `BIT_LENGTH(s)` | Bit length (8.17+) | `EVAL bits = BIT_LENGTH(name)` |
+| `BYTE_LENGTH(s)` | Byte length (8.17+) | `EVAL bytes = BYTE_LENGTH(name)` |
+| `CHUNK(field, settings)` | Split text into chunks (9.3+, preview) | `EVAL chunks = CHUNK(body, {"strategy":"word","max_chunk_size":50})` |
+| `HASH(alg, s)` | Hash string (8.18/9.0+) | `EVAL h = HASH("SHA-256", msg)` |
+| `MD5(s)` | MD5 hash (8.18/9.0+) | `EVAL h = MD5(content)` |
+| `SHA1(s)` | SHA-1 hash (8.18/9.0+) | `EVAL h = SHA1(content)` |
+| `SHA256(s)` | SHA-256 hash (8.18/9.0+) | `EVAL h = SHA256(content)` |
+| `FROM_BASE64(s)` | Decode base64 | `EVAL decoded = FROM_BASE64(encoded)` |
+| `TO_BASE64(s)` | Encode to base64 | `EVAL encoded = TO_BASE64(data)` |
+| `URL_DECODE(s)` | URL-decode (9.2+) | `EVAL decoded = URL_DECODE(url)` |
+| `URL_ENCODE(s)` | URL-encode (9.2+) | `EVAL encoded = URL_ENCODE(text)` |
+| `URL_ENCODE_COMPONENT(s)` | URL-encode for URI components (9.2+) | `EVAL encoded = URL_ENCODE_COMPONENT(text)` |
+| `JSON_EXTRACT(field, path)` | Extract value from JSON string (Serverless GA) | `EVAL name = JSON_EXTRACT(raw, "$.user.name")` |
+
+**JSON_EXTRACT with \_source — flattened field workaround:**
+
+ES|QL does not natively access `flattened` field sub-keys. Use `METADATA _source` with `JSON_EXTRACT` to reach inside
+flattened objects. `_source` can be passed directly to `JSON_EXTRACT` — do not wrap it with `TO_STRING()`.
+
+```esql
+FROM logs-* METADATA _source
+| EVAL provider = JSON_EXTRACT(_source, "$.cloud.provider")
+| STATS count = COUNT(*) BY provider
+```
+
+This also works for any field that exists in the raw document but has no explicit mapping.
---
@@ -1423,15 +1746,15 @@ FROM logs-*
Access document metadata with the `METADATA` directive on the `FROM` command. Once enabled, metadata fields behave like
regular index fields.
-| Field | Type | Description |
-| ------------- | ------- | --------------------------------------------------------------- |
-| `_id` | keyword | Unique document ID |
-| `_index` | keyword | Index name |
-| `_version` | long | Document version number |
-| `_score` | float | Query relevance score (updated by full-text search functions) |
-| `_ignored` | keyword | Fields that were ignored when the document was indexed |
-| `_index_mode` | keyword | Index mode (`standard`, `lookup`, `logsdb`, `time_series` etc.) |
-| `_source` | special | Original JSON document body (not supported by functions) |
+| Field | Type | Description |
+| ------------- | ------- | -------------------------------------------------------------------------------------- |
+| `_id` | keyword | Unique document ID |
+| `_index` | keyword | Index name |
+| `_version` | long | Document version number |
+| `_score` | float | Query relevance score (updated by full-text search functions) |
+| `_ignored` | keyword | Fields that were ignored when the document was indexed |
+| `_index_mode` | keyword | Index mode (`standard`, `lookup`, `logsdb`, `time_series` etc.) |
+| `_source` | special | Original JSON document body. Use `JSON_EXTRACT` to access flattened or unmapped fields |
```esql
FROM logs METADATA _id, _index, _version
diff --git a/plugins/elasticsearch/skills/elasticsearch-esql/references/esql-version-history.md b/plugins/elasticsearch/skills/elasticsearch-esql/references/esql-version-history.md
index 19ceab0..13ef9a1 100644
--- a/plugins/elasticsearch/skills/elasticsearch-esql/references/esql-version-history.md
+++ b/plugins/elasticsearch/skills/elasticsearch-esql/references/esql-version-history.md
@@ -27,51 +27,58 @@ determine compatibility when writing queries for specific Elasticsearch deployme
## Version Timeline Overview
-| Version | Release | Status | Key Additions |
-| ------- | -------- | ------------ | ------------------------------------------------------------------------------------------------ |
-| 8.11 | Nov 2023 | Tech Preview | Initial ES\|QL release |
-| 8.12 | Jan 2024 | Tech Preview | Spatial types, PROFILE |
-| 8.13 | Mar 2024 | Tech Preview | Async queries, cross-cluster ENRICH |
-| 8.14 | May 2024 | **GA** | Spatial functions, regex optimization |
-| 8.15 | Aug 2024 | GA | Type casting (`::`), Arrow output |
-| 8.16 | Oct 2024 | GA | Per-aggregation WHERE, new math/string functions |
-| 8.17 | Dec 2024 | GA | MATCH, QSTR full-text functions |
-| 8.18 | Feb 2025 | GA | LOOKUP JOIN (preview), scoring, KQL |
-| 8.19 | Apr 2025 | GA | MATCH_PHRASE, FORK, CHANGE_POINT (preview) |
-| 9.0 | Feb 2025 | GA | Released with 8.18 features |
-| 9.1 | Jun 2025 | GA | Full-text functions GA, FORK (preview) |
-| 9.2 | Oct 2025 | GA | Multi-field joins, TS, INLINE STATS (preview), CHANGE_POINT GA, FUSE (preview), RERANK (preview) |
-| 9.3 | Jan 2026 | GA | INLINE STATS GA, SET directive (preview), Lucene-pushable JOIN predicates |
+| Version | Release | Status | Key Additions |
+| ------- | -------- | ------------ | ------------------------------------------------------------------------------------------------------- |
+| 8.11 | Nov 2023 | Tech Preview | Initial ES\|QL release |
+| 8.12 | Jan 2024 | Tech Preview | Spatial types, PROFILE |
+| 8.13 | Mar 2024 | Tech Preview | Async queries, cross-cluster ENRICH |
+| 8.14 | May 2024 | **GA** | Spatial functions, regex optimization |
+| 8.15 | Aug 2024 | GA | Type casting (`::`), Arrow output |
+| 8.16 | Oct 2024 | GA | Per-aggregation WHERE, new math/string functions |
+| 8.17 | Dec 2024 | GA | MATCH, QSTR full-text functions |
+| 8.18 | Feb 2025 | GA | LOOKUP JOIN (preview), scoring, KQL |
+| 8.19 | Apr 2025 | GA | MATCH_PHRASE, FORK, CHANGE_POINT (preview) |
+| 9.0 | Feb 2025 | GA | Released with 8.18 features |
+| 9.1 | Jun 2025 | GA | Full-text functions GA, FORK (preview) |
+| 9.2 | Oct 2025 | GA | Multi-field joins, TS, INLINE STATS (preview), CHANGE_POINT GA, FUSE (preview), RERANK (preview) |
+| 9.3 | Jan 2026 | GA | INLINE STATS GA, SET directive (preview), Lucene-pushable JOIN predicates |
+| 9.4 | May 2026 | GA | TS GA, time series functions GA, WITHOUT/METRICS_INFO/TS_INFO GA, PROMQL (preview), MV_EXPAND/VALUES GA |
## Feature Availability by Version
### Commands
-| Command | Introduced | GA | Notes |
-| -------------- | ---------- | -------- | ------------------------------------------- |
-| `FROM` | 8.11 | 8.14 | Source command |
-| `WHERE` | 8.11 | 8.14 | Filtering |
-| `EVAL` | 8.11 | 8.14 | Computed columns |
-| `STATS ... BY` | 8.11 | 8.14 | Aggregations with grouping |
-| `SORT` | 8.11 | 8.14 | Ordering results |
-| `LIMIT` | 8.11 | 8.14 | Result set size |
-| `KEEP` | 8.11 | 8.14 | Column selection |
-| `DROP` | 8.11 | 8.14 | Column removal |
-| `RENAME` | 8.11 | 8.14 | Column renaming |
-| `DISSECT` | 8.11 | 8.14 | Pattern extraction |
-| `GROK` | 8.11 | 8.14 | Log parsing |
-| `ENRICH` | 8.11 | 8.14 | Data enrichment |
-| `MV_EXPAND` | 8.11 | 8.14 | Multi-value expansion |
-| `SHOW` | 8.11 | 8.14 | Metadata display |
-| `ROW` | 8.11 | 8.14 | Literal row creation |
-| `LOOKUP JOIN` | 8.18/9.0 | 8.19/9.1 | SQL-style LEFT JOIN with lookup indices |
-| `INLINE STATS` | 9.2 | 9.3 | Inline aggregations (like window functions) |
-| `FORK` | 8.19/9.1 | Preview | Multiple execution branches |
-| `FUSE` | 9.2 | Preview | Combine results from FORK branches |
-| `TS` | 9.2 | 9.2 | Time series mode |
-| `RERANK` | 9.2 | Preview | Re-score results with inference |
-| `COMPLETION` | 9.2 | 9.2 | LLM text generation |
-| `SAMPLE` | 8.19/9.1 | Preview | Random sampling |
+| Command | Introduced | GA | Notes |
+| -------------- | ---------- | -------- | ----------------------------------------------- |
+| `FROM` | 8.11 | 8.14 | Source command |
+| `WHERE` | 8.11 | 8.14 | Filtering |
+| `EVAL` | 8.11 | 8.14 | Computed columns |
+| `STATS ... BY` | 8.11 | 8.14 | Aggregations with grouping |
+| `SORT` | 8.11 | 8.14 | Ordering results |
+| `LIMIT` | 8.11 | 8.14 | Result set size |
+| `KEEP` | 8.11 | 8.14 | Column selection |
+| `DROP` | 8.11 | 8.14 | Column removal |
+| `RENAME` | 8.11 | 8.14 | Column renaming |
+| `DISSECT` | 8.11 | 8.14 | Pattern extraction |
+| `GROK` | 8.11 | 8.14 | Log parsing |
+| `ENRICH` | 8.11 | 8.14 | Data enrichment |
+| `MV_EXPAND` | 8.11 | 9.4 | Multi-value expansion (GA) |
+| `SHOW` | 8.11 | 8.14 | Metadata display |
+| `ROW` | 8.11 | 8.14 | Literal row creation |
+| `LOOKUP JOIN` | 8.18/9.0 | 8.19/9.1 | SQL-style LEFT JOIN with lookup indices |
+| `INLINE STATS` | 9.2 | 9.3 | Inline aggregations (like window functions) |
+| `FORK` | 8.19/9.1 | Preview | Multiple execution branches |
+| `FUSE` | 9.2 | Preview | Combine results from FORK branches |
+| `TS` | 9.2 | 9.4 | Time series source command |
+| `PROMQL` | 9.4 | Preview | Source command using PromQL syntax on TSDS |
+| `METRICS_INFO` | 9.4 | 9.4 | TSDS metric catalogue (after `TS`) |
+| `TS_INFO` | 9.4 | 9.4 | Per-(metric, time series) metadata (after `TS`) |
+| `RERANK` | 9.2 | Preview | Re-score results with inference |
+| `COMPLETION` | 9.2 | 9.2 | LLM text generation |
+| `SAMPLE` | 8.19/9.1 | Preview | Random sampling |
+| `URI_PARTS` | Srvless | Srvless | Parse URI into structured columns |
+| `USER_AGENT` | Srvless | Srvless | Parse user agent into structured columns |
+| `REG_DOMAIN` | Srvless | Srvless | `REGISTERED_DOMAIN`: extract from hostname |
### Full-Text Search Functions
@@ -157,19 +164,22 @@ determine compatibility when writing queries for specific Elasticsearch deployme
| `MEDIAN`, `MEDIAN_ABSOLUTE_DEVIATION` | 8.11 | Statistical |
| `PERCENTILE` | 8.11 | Percentile calculation |
| `TOP` | 8.15 | Top N values |
-| `VALUES` | 8.14 | Collect unique values |
+| `VALUES` | 8.14 | Unique values (GA in 9.4) |
| `ST_EXTENT_AGG` | 8.18/9.0 | Spatial bounding box |
| `WEIGHTED_AVG` | 8.16 | Weighted average |
| `STD_DEV` | 8.18/9.0 | Standard deviation |
| `VARIANCE` | 8.18/9.0 | Variance |
+| `FIRST` / `EARLIEST` | Serverless | Earliest value by sort field |
+| `LAST` / `LATEST` | Serverless | Latest value by sort field |
### Grouping Functions
-| Function | Introduced | Notes |
-| ------------ | ------------- | ------------------------------------------------- |
-| `BUCKET` | 8.11 | Numeric/date bucketing in `BY` clause |
-| `CATEGORIZE` | 8.18/9.0 | Auto-categorization of text in `BY` clause |
-| `TBUCKET` | 9.2 (preview) | Time bucketing from `@timestamp`; preferred in TS |
+| Function | Introduced | Notes |
+| ------------ | ---------- | ---------------------------------------------------------------- |
+| `BUCKET` | 8.11 | Numeric/date bucketing in `BY` clause |
+| `CATEGORIZE` | 8.18/9.0 | Auto-categorization of text in `BY` clause |
+| `TBUCKET` | 9.2 | Time bucketing from `@timestamp`; preferred in TS (GA in 9.4) |
+| `WITHOUT` | 9.4 | Group time series by every dimension except the listed ones (GA) |
### Per-Aggregation WHERE
@@ -189,31 +199,39 @@ Available since 8.16. Allows filtering individual aggregations without affecting
### Time Series Aggregation Functions
-Available under `TS ... | STATS`. See [time-series-queries.md](time-series-queries.md) for full reference.
-
-| Function | Introduced | Notes |
-| -------------------------- | ------------- | ----------------------------------------------- |
-| `RATE` | 9.2 (preview) | Per-second rate of counter increase |
-| `IRATE` | 9.2 (preview) | Instant rate (last two data points) |
-| `INCREASE` | 9.2 (preview) | Absolute counter increase in window |
-| `DELTA` | 9.2 (preview) | Absolute change of a gauge |
-| `IDELTA` | 9.2 (preview) | Change between last two data points |
-| `AVG_OVER_TIME` | 9.2 (preview) | Average value over time |
-| `SUM_OVER_TIME` | 9.2 (preview) | Sum of values over time |
-| `MIN_OVER_TIME` | 9.2 (preview) | Minimum value over time |
-| `MAX_OVER_TIME` | 9.2 (preview) | Maximum value over time |
-| `FIRST_OVER_TIME` | 9.2 (preview) | Earliest value by `@timestamp` |
-| `LAST_OVER_TIME` | 9.2 (preview) | Latest value by `@timestamp` (implicit default) |
-| `COUNT_OVER_TIME` | 9.2 (preview) | Count of values over time |
-| `COUNT_DISTINCT_OVER_TIME` | 9.2 (preview) | Count of distinct values over time |
-| `PRESENT_OVER_TIME` | 9.2 (preview) | `true` if field has values in window |
-| `ABSENT_OVER_TIME` | 9.2 (preview) | `true` if field has no values in window |
-| `DERIV` | 9.3 (preview) | Derivative via linear regression |
-| `PERCENTILE_OVER_TIME` | 9.3 (preview) | Percentile of values over time |
-| `STDDEV_OVER_TIME` | 9.3 (preview) | Population standard deviation over time |
-| `VARIANCE_OVER_TIME` | 9.3 (preview) | Population variance over time |
-
-Sliding window parameter (second argument) available since 9.3 preview.
+Available under `TS ... | STATS`. See [time-series-queries.md](time-series-queries.md) for full reference. All time
+series aggregation functions in this table — both the 9.2-introduced set and the 9.3-introduced set (`DERIV`,
+`PERCENTILE_OVER_TIME`, `STDDEV_OVER_TIME`, `VARIANCE_OVER_TIME`) — are **GA since 9.4**.
+
+| Function | Introduced | Status | Notes |
+| -------------------------- | ------------- | -------- | ----------------------------------------------- |
+| `RATE` | 9.2 (preview) | GA (9.4) | Per-second rate of counter increase |
+| `IRATE` | 9.2 (preview) | GA (9.4) | Instant rate (last two data points) |
+| `INCREASE` | 9.2 (preview) | GA (9.4) | Absolute counter increase in window |
+| `DELTA` | 9.2 (preview) | GA (9.4) | Absolute change of a gauge |
+| `IDELTA` | 9.2 (preview) | GA (9.4) | Change between last two data points |
+| `AVG_OVER_TIME` | 9.2 (preview) | GA (9.4) | Average value over time |
+| `SUM_OVER_TIME` | 9.2 (preview) | GA (9.4) | Sum of values over time |
+| `MIN_OVER_TIME` | 9.2 (preview) | GA (9.4) | Minimum value over time |
+| `MAX_OVER_TIME` | 9.2 (preview) | GA (9.4) | Maximum value over time |
+| `FIRST_OVER_TIME` | 9.2 (preview) | GA (9.4) | Earliest value by `@timestamp` |
+| `LAST_OVER_TIME` | 9.2 (preview) | GA (9.4) | Latest value by `@timestamp` (implicit default) |
+| `COUNT_OVER_TIME` | 9.2 (preview) | GA (9.4) | Count of values over time |
+| `COUNT_DISTINCT_OVER_TIME` | 9.2 (preview) | GA (9.4) | Count of distinct values over time |
+| `PRESENT_OVER_TIME` | 9.2 (preview) | GA (9.4) | `true` if field has values in window |
+| `ABSENT_OVER_TIME` | 9.2 (preview) | GA (9.4) | `true` if field has no values in window |
+| `DERIV` | 9.3 (preview) | GA (9.4) | Derivative via linear regression |
+| `PERCENTILE_OVER_TIME` | 9.3 (preview) | GA (9.4) | Percentile of values over time |
+| `STDDEV_OVER_TIME` | 9.3 (preview) | GA (9.4) | Population standard deviation over time |
+| `VARIANCE_OVER_TIME` | 9.3 (preview) | GA (9.4) | Population variance over time |
+
+**Sliding window parameter (second argument):**
+
+- 9.2-9.3 (preview) — accepted window values are limited to multiples of the `TBUCKET` interval in the `BY` clause; if
+ no window is specified, the bucket interval is used implicitly.
+- 9.4+ (GA) — all window values are accepted, with performance optimizations when the window is a multiple of the
+ `TBUCKET` interval. Mixing windows that are smaller than the time bucket for one metric with windows larger than the
+ time bucket for another metric in the same query is not allowed.
### Conditional Functions
@@ -252,22 +270,32 @@ ES|QL **does not support cursor-based pagination** like the Search API's `search
- Use `STATS` to aggregate at query time
- For exports, use Search API with `search_after` instead
-### Time Zone Support (Limited)
+### Time Zone Support (Limited before Serverless / 9.4)
-ES|QL has **limited timezone support**.
+ES|QL has **limited timezone support** on self-managed clusters prior to 9.4. All dates are processed in UTC internally
+and there is no per-function timezone argument.
-**Current limitations:**
+On **Serverless**, ES|QL supports query-wide timezone via the `SET time_zone` directive (GA on Serverless). This accepts
+IANA timezone strings and UTC offsets, and applies to all date/time operations including `DATE_TRUNC`, `DATE_FORMAT`,
+`NOW()`, bucketing, and display.
-- `DATE_FORMAT` and `DATE_PARSE` do not support timezone parameters
-- All dates processed in UTC internally
-- Kibana charts may show timezone inconsistencies
+```esql
+SET time_zone = "America/New_York";
+FROM logs-*
+| STATS errors = COUNT(*) BY hour = DATE_TRUNC(1 hour, @timestamp)
+| SORT hour DESC
+```
+
+**Remaining limitations (all versions):**
+
+- No per-function timezone argument — `DATE_TRUNC(1 hour, @timestamp, "America/New_York")` does **not** work
+- `DATE_FORMAT` and `DATE_PARSE` do not accept timezone parameters directly; use `SET time_zone` instead
- GitHub tracking issue: [#107560](https://github.com/elastic/elasticsearch/issues/107560)
-**Workarounds:**
+**Self-managed before 9.4:**
-- Store timezone offset in a separate field
-- Convert to UTC before querying
-- Use `EVAL` to add/subtract hours manually:
+- `SET time_zone` only accepts UTC offsets (`"+05:00"`), not IANA timezone strings
+- Workaround: use `EVAL` to add/subtract hours manually:
```esql
| EVAL local_time = timestamp + 1 hour
@@ -285,16 +313,16 @@ returned at all** — they are silently omitted from results.
These field types are not supported or have limitations:
-| Type | Status |
-| -------------- | ---------------------------- |
-| `nested` | Not supported - returns null |
-| `flattened` | Not supported |
-| `join` | Not supported |
-| `date_range` | Not supported |
-| `binary` | Not supported |
-| `completion` | Not supported |
-| `rank_feature` | Not supported |
-| `histogram` | Not supported |
+| Type | Status |
+| -------------- | ---------------------------------------------------------------------------------- |
+| `nested` | Not supported - returns null |
+| `flattened` | Not natively supported; use `METADATA _source` + `JSON_EXTRACT` for sub-key access |
+| `join` | Not supported |
+| `date_range` | Not supported |
+| `binary` | Not supported |
+| `completion` | Not supported |
+| `rank_feature` | Not supported |
+| `histogram` | Not supported |
### JOIN Limitations
@@ -317,15 +345,25 @@ These field types are not supported or have limitations:
- Lucene-pushable predicates: `MATCH`, `QSTR`, `KQL`, `CIDR_MATCH` in join conditions
- Further performance gains for filtered joins
-### No Subqueries
+### Subqueries (Limited)
-ES|QL does not support:
+ES|QL supports **subqueries in `FROM`** (Serverless tech preview) for combining results from multiple pipelines (UNION
+ALL semantics). These are non-correlated — each branch is independent.
-- Subqueries in WHERE clauses
-- Nested SELECT statements
-- CTEs (Common Table Expressions)
+```esql
+FROM
+ (FROM web_logs | WHERE status >= 500 | KEEP @timestamp, message, service.name),
+ (FROM app_logs | WHERE level == "error" | KEEP @timestamp, message, service.name)
+| SORT @timestamp DESC
+```
+
+**Not supported:**
-Use `INLINE STATS` (9.2+) for some subquery-like patterns.
+- Subqueries in `WHERE` clauses (no `WHERE field IN (FROM ...)`)
+- Correlated subqueries (branches cannot reference outer columns)
+- Nested SELECT / CTEs (Common Table Expressions)
+
+Use `INLINE STATS` (9.2+) for per-row vs. aggregate comparison patterns.
## Cross-Cluster Query Support
@@ -378,17 +416,48 @@ Use `INLINE STATS` (9.2+) for some subquery-like patterns.
### 9.2+
-- Use `TS` with `RATE`, `AVG_OVER_TIME`, etc. for time series metrics aggregations
-- Use `TBUCKET` for time bucketing in TS queries
+- Use `TS` with `RATE`, `AVG_OVER_TIME`, etc. for time series metrics aggregations (preview in 9.2-9.3, GA in 9.4)
+- Use `TBUCKET` for time bucketing in TS queries (GA in 9.4)
- Multi-field `LOOKUP JOIN` for complex correlations
- `FUSE` for hybrid search scoring
### 9.3+
- Use `TRANGE` instead of manual `WHERE @timestamp` filters
-- Sliding window parameter for time series functions (e.g. `RATE(field, 10m)`)
+- Sliding window parameter for time series functions (e.g. `RATE(field, 10m)`); in 9.2-9.3 the window must be a multiple
+ of the `TBUCKET` interval, this restriction is lifted in 9.4
- `CLAMP`, `CLAMP_MIN`, `CLAMP_MAX` for bounding metric values
+### 9.4+
+
+- `TS` source command and **all** time series aggregation functions are now **GA** — both the 9.2-introduced set
+ (`RATE`, `IRATE`, `INCREASE`, `DELTA`, `IDELTA`, `*_OVER_TIME`, `PRESENT_OVER_TIME`, `ABSENT_OVER_TIME`) and the
+ 9.3-introduced set (`DERIV`, `PERCENTILE_OVER_TIME`, `STDDEV_OVER_TIME`, `VARIANCE_OVER_TIME`).
+- `TBUCKET` grouping function is **GA**.
+- New `WITHOUT(...)` grouping function (GA) for time series queries: `BY WITHOUT(dim1, ...)` groups by every dimension
+ except the listed ones; `BY WITHOUT()` (no args) is equivalent to the implicit "group by all dimensions" behavior.
+- New `METRICS_INFO` and `TS_INFO` processing commands (both **GA**) for discovering the metric catalogue and dimension
+ labels of TSDS data without inspecting index mappings. Both must come after a `TS` source command and must appear
+ before pipeline-breaking commands (`STATS`/`SORT`/`LIMIT`). `METRICS_INFO` returns one row per distinct metric
+ signature; `TS_INFO` returns one row per (metric, time series) combination with the identifying dimension labels.
+- Sliding window parameter (`RATE(field, 10m)`) accepts arbitrary durations — no longer limited to multiples of the
+ `TBUCKET` interval. Note: a single query cannot mix windows smaller than the bucket for one metric with windows larger
+ than the bucket for another metric.
+- New `PROMQL` source command (preview) to run Prometheus Query Language directly against TSDS indices, with implicit
+ range selectors and a Kibana-aware `step`/`buckets` model. See [promql-command.md](promql-command.md). Prefer `PROMQL`
+ only when the user explicitly thinks in PromQL or is migrating Prometheus dashboards/alerts; otherwise prefer `TS`.
+- `MV_EXPAND` is GA
+- `VALUES` aggregation is GA
+
+### Serverless (latest)
+
+- `SET time_zone` with IANA timezone strings for query-wide timezone support (GA)
+- `LIMIT n BY field` for grouped top-N queries
+- `URI_PARTS`, `USER_AGENT`, `REGISTERED_DOMAIN` pipe commands for parsing structured strings
+- `FROM` subqueries for combining results from multiple pipelines (tech preview)
+- `EARLIEST`/`LATEST` aliases for `FIRST`/`LAST` aggregations
+- `JSON_EXTRACT` on `METADATA _source` for accessing flattened field sub-keys
+
## Version Detection
To check ES|QL availability and version:
diff --git a/plugins/elasticsearch/skills/elasticsearch-esql/references/generation-tips.md b/plugins/elasticsearch/skills/elasticsearch-esql/references/generation-tips.md
index dea929c..e4192a0 100644
--- a/plugins/elasticsearch/skills/elasticsearch-esql/references/generation-tips.md
+++ b/plugins/elasticsearch/skills/elasticsearch-esql/references/generation-tips.md
@@ -147,7 +147,7 @@ FROM my-index-2024.* // Dated indices
```
For time series data streams (TSDS), use `TS` instead of `FROM` to enable time series aggregation functions like `RATE`,
-`AVG_OVER_TIME`, etc. (9.2+):
+`AVG_OVER_TIME`, etc. (preview from 9.2 to 9.3, **GA since 9.4**):
```esql
TS metrics-* // Time series source — enables RATE, AVG_OVER_TIME, etc.
@@ -503,6 +503,12 @@ TS metrics-tsds
See [Time Series Queries](time-series-queries.md) for the full inner/outer aggregation model.
+**Version status:** `TS`, `TBUCKET`, the new `WITHOUT(...)` grouping function, the new `METRICS_INFO` / `TS_INFO`
+discovery commands, and **all** time series aggregation functions are **GA since 9.4** — including the 9.2-introduced
+set (`RATE`, `IRATE`, `INCREASE`, `DELTA`, `IDELTA`, all `*_OVER_TIME`, `PRESENT_OVER_TIME`, `ABSENT_OVER_TIME`) and the
+9.3-introduced set (`DERIV`, `PERCENTILE_OVER_TIME`, `STDDEV_OVER_TIME`, `VARIANCE_OVER_TIME`). On clusters in 9.2-9.3
+these features are tech preview. `TRANGE` remains in preview.
+
**Pre-9.2 limitation:** The `TS` command, `RATE()`, `TBUCKET()`, and `AVG_OVER_TIME()` all require Elasticsearch
**9.2+**. On older clusters, counter fields (`counter_long`, `counter_double`) cannot be aggregated meaningfully —
standard aggregation functions like `MAX()`, `SUM()`, and `AVG()` reject counter field types. There is no workaround.
@@ -512,6 +518,10 @@ the `TS` command and `RATE()` are required (9.2+) and the query cannot be expres
For **gauge** fields in time-series indices on pre-9.2 clusters, `FROM` with standard aggregations (`AVG`, `MAX`, `MIN`)
still works — only counter fields are affected.
+**Sliding window restriction (9.2-9.3):** When the user wants a per-time-series aggregation window different from the
+`TBUCKET` interval (`RATE(field, 10m) BY TBUCKET(1m)`), the window must be a multiple of the bucket interval on preview
+clusters. **9.4+** (GA) accepts arbitrary windows.
+
### INLINE STATS (9.2+)
`INLINE STATS` is available in **9.2+** only. It computes an aggregation and appends the result as a new column to every
@@ -522,6 +532,70 @@ ES|QL before 9.2**. There is no fallback.
When the cluster is pre-9.2 and the question requires per-row vs. aggregate comparison, explain that `INLINE STATS` is
needed and suggest the user either upgrade or perform the comparison client-side.
+### Pipe Commands: URI_PARTS, USER_AGENT, REGISTERED_DOMAIN (Serverless)
+
+These are **pipe commands** (like `DISSECT`/`GROK`), not scalar functions. They must appear on their own pipeline stage
+with `target = expression` syntax. A target prefix is mandatory.
+
+```esql
+// WRONG — function-call syntax does not work
+| EVAL parts = URI_PARTS(url.full)
+
+// CORRECT — pipe command syntax with target prefix
+| URI_PARTS parts = url.full
+| KEEP parts.domain, parts.path, parts.scheme
+```
+
+When the user asks to "parse URLs", "extract domains", or "parse user agents", reach for these commands instead of
+`DISSECT`/`GROK`:
+
+| User Request | Command |
+| ------------------------- | ------------------- |
+| Parse/decompose a URL | `URI_PARTS` |
+| Parse a user agent string | `USER_AGENT` |
+| Extract registered domain | `REGISTERED_DOMAIN` |
+
+### Grouped Top-N with LIMIT BY (Serverless)
+
+`LIMIT n BY field` keeps the top N rows per group after sorting. The number comes **before** `BY`.
+
+```esql
+// Top 3 error-producing hosts per service
+FROM logs-*
+| WHERE level == "error"
+| STATS cnt = COUNT(*) BY service.name, host.name
+| SORT cnt DESC
+| LIMIT 3 BY service.name
+```
+
+This replaces the common `INLINE STATS` + rank-and-filter pattern for simple grouped top-N.
+
+### Subqueries in FROM vs FORK
+
+**Subqueries** (Serverless tech preview) combine results from **different** data sources (UNION ALL semantics). **FORK**
+runs **different analyses** on the **same** data source.
+
+| Scenario | Use |
+| ------------------------------------- | ---------- |
+| Combine errors from two index sets | Subqueries |
+| Run multiple aggregations on one set | FORK |
+| Compare time windows of the same data | FORK |
+| Union independent pipelines | Subqueries |
+
+```esql
+// Subqueries — different sources
+FROM
+ (FROM web_logs | WHERE status >= 500 | KEEP @timestamp, message, service.name),
+ (FROM app_logs | WHERE level == "error" | KEEP @timestamp, message, service.name)
+| SORT @timestamp DESC
+
+// FORK — same source, different analyses
+FROM logs-*
+| FORK
+ ( WHERE level == "error" | STATS errors = COUNT(*) BY service.name )
+ ( WHERE level == "warning" | STATS warnings = COUNT(*) BY service.name )
+```
+
### External IPs — CIDR_MATCH with RFC 1918
When the user asks about "external IPs" or "public IPs", exclude private (RFC 1918) ranges with `NOT CIDR_MATCH`:
diff --git a/plugins/elasticsearch/skills/elasticsearch-esql/references/promql-command.md b/plugins/elasticsearch/skills/elasticsearch-esql/references/promql-command.md
new file mode 100644
index 0000000..31b01c9
--- /dev/null
+++ b/plugins/elasticsearch/skills/elasticsearch-esql/references/promql-command.md
@@ -0,0 +1,323 @@
+# ES|QL PROMQL Command
+
+Query time series indices using **Prometheus Query Language (PromQL)** as a source command in ES|QL. The `PROMQL`
+command is the bridge for users who already know PromQL or are migrating Prometheus dashboards and alerts onto an
+Elasticsearch backend, while still letting them post-process results with regular ES|QL pipes.
+
+> **Version:** `PROMQL` is a **preview** feature available since Elastic Stack **9.4** and on Elastic Cloud Serverless.
+> Treat it as preview — syntax, options, and supported PromQL functions may change in future releases. See
+> [esql-version-history.md](esql-version-history.md) for version availability.
+
+## Table of Contents
+
+- [When to Use PROMQL](#when-to-use-promql)
+- [Syntax](#syntax)
+- [Options](#options)
+- [Output Columns](#output-columns)
+- [Implicit Range Selectors](#implicit-range-selectors)
+- [Examples](#examples)
+- [Post-Processing with ES|QL](#post-processing-with-esql)
+- [PROMQL vs TS](#promql-vs-ts)
+- [Limitations](#limitations)
+- [Kibana Time Filtering](#kibana-time-filtering)
+- [Guidelines](#guidelines)
+- [References](#references)
+
+---
+
+## When to Use PROMQL
+
+Prefer `PROMQL` when **any** of the following apply:
+
+- The user explicitly asks for a PromQL query, references Prometheus syntax (`sum by (instance) (...)`, label matchers
+ like `{cluster="prod"}`, etc), or is migrating a Prometheus dashboard or alert. If the user explicitly requests for
+ PromQL but the query is not supported yet (check [Limitations](#limitations) below), state the issue.
+- Compatibility with Prometheus tooling is required (Grafana panels, alerting rules, scripts that already speak PromQL).
+
+Prefer the [`TS` command](time-series-queries.md) when:
+
+- The user wrote ES|QL (or is asking in natural language without PromQL terms) and the query is naturally expressed in
+ the inner/outer aggregation paradigm (`SUM(RATE(...))`, `AVG(AVG_OVER_TIME(...))`).
+- The query mixes time series with non-time-series data sources or uses ES|QL features like `LOOKUP JOIN`,
+ `CHANGE_POINT`, or `INLINE STATS` _before_ the metrics aggregation.
+
+`PROMQL` and `TS` target the same TSDS indices — choose based on the syntax that best matches the user's intent.
+
+---
+
+## Syntax
+
+```esql
+PROMQL [ ... ] [ = ] ( )
+```
+
+- Zero or more space-separated `key=value` options.
+- A PromQL expression, optionally wrapped in parentheses and assigned a ``.
+- The expression follows standard
+ [Prometheus query language](https://prometheus.io/docs/prometheus/latest/querying/basics/) syntax (label matchers,
+ range selectors, aggregations, binary operations) within the [Limitations](#limitations) below.
+
+### Minimal example
+
+```esql
+PROMQL sum by (instance) (rate(http_requests_total))
+```
+
+### Named result
+
+```esql
+PROMQL http_rate = (sum by (instance) (rate(http_requests_total)))
+```
+
+When a `` is provided, the metric column is named `` instead of the raw PromQL expression. In
+the example above, the column would be named `http_rate`.
+
+---
+
+## Options
+
+The options mirror the Prometheus [HTTP API](https://prometheus.io/docs/prometheus/latest/querying/api/#range-queries)
+with ES|QL-specific additions.
+
+| Option | Default | Description |
+| ----------------- | ----------- | --------------------------------------------------------------------------------------------------------------------- |
+| `index` | `metrics-*` | Indices, data streams, or aliases. Supports wildcards and date math. |
+| `step` | inferred | Query resolution step width. Auto-derived from `buckets` and the time range when omitted. |
+| `buckets` | `100` | Target bucket count for auto-step derivation. Mutually exclusive with `step`. Requires a known time range. |
+| `start` | inferred | Inclusive start of the time range. Falls back to Kibana's date picker, or unrestricted if missing. |
+| `end` | inferred | Inclusive end of the time range. Falls back to Kibana's date picker, or unrestricted if missing. |
+| `scrape_interval` | `1m` | Expected metric collection interval. Used as the implicit range selector window: `max(step, scrape_interval)`. |
+| `=` | _none_ | Optional name for the metric output column. Defaults to the PromQL expression text. Wrap the expression in `( ... )`. |
+
+**Time format for `start` / `end`:** ISO-8601 strings (e.g., `"2026-04-01T00:00:00Z"`). The same formats accepted by
+`TRANGE` work here.
+
+**`step` vs `buckets`:** Pass exactly one. `step` fixes the resolution (`step=5m`); `buckets` lets the engine pick a
+step that produces around N buckets across the time range (`buckets=50`).
+
+---
+
+## Output Columns
+
+The result table has these columns:
+
+| Column | Type | Description |
+| ------------------------------------------------------- | --------- | --------------------------------------------------------------- |
+| The PromQL expression (or `` if specified) | `double` | The computed metric value |
+| `step` | `date` | Timestamp for each evaluation step |
+| Grouping labels (when `by (...)` or `without (...)`) | `keyword` | One column per grouping label |
+| `_timeseries` | `keyword` | JSON-encoded labels when there is no `by`/`without` aggregation |
+
+When the PromQL expression includes a cross-series aggregation like `sum by (instance) (...)`, each grouping label
+becomes its own column (`instance:keyword`). Without a cross-series aggregation, all labels collapse into a single
+`_timeseries` column as a JSON string.
+
+---
+
+## Implicit Range Selectors
+
+Standard PromQL requires range vector functions to specify a range selector: `rate(http_requests_total[5m])`. The
+`PROMQL` command **allows omitting the range selector** entirely:
+
+```esql
+PROMQL scrape_interval=15s sum(rate(http_requests_total))
+```
+
+When the range selector is absent, the window is computed automatically as `max(step, scrape_interval)`. This is
+particularly useful for Kibana dashboards where `step` is determined by the date picker and you want the range vector to
+scale with it.
+
+You can still pass an explicit range selector when you need a fixed window: `rate(http_requests_total[5m])`.
+
+---
+
+## Examples
+
+### Fully adaptive query (recommended for Kibana)
+
+Let Kibana's date picker drive the time range, and let `step` and the range selector be inferred:
+
+```esql
+PROMQL index=metrics-* sum by (instance) (rate(http_requests_total))
+```
+
+The query responds to the date picker, adjusts the step size to the selected range, and sizes the implicit range
+selector window accordingly. This is the recommended pattern for dashboard panels.
+
+### Range query with explicit parameters
+
+```esql
+PROMQL index=k8s step=5m start="2024-05-10T00:20:00.000Z" end="2024-05-10T00:25:00.000Z" (
+ sum(avg_over_time(network.cost[5m]))
+)
+```
+
+| sum(avg_over_time(network.cost[5m])):double | step:date |
+| ------------------------------------------- | ------------------------ |
+| 50.25 | 2024-05-10T00:20:00.000Z |
+
+### Cross-series aggregation by label
+
+```esql
+PROMQL index=k8s step=1h result=(sum by (cluster) (network.cost))
+| SORT result
+```
+
+| result:double | step:datetime | cluster:keyword |
+| ------------- | ------------------------ | --------------- |
+| 15.875 | 2024-05-10T00:00:00.000Z | staging |
+| 18.625 | 2024-05-10T00:00:00.000Z | prod |
+| 26.5 | 2024-05-10T00:00:00.000Z | qa |
+
+### Label filtering with named result
+
+```esql
+PROMQL index=k8s step=1h cost=(max by (cluster) (network.total_bytes_in{cluster!="prod"}))
+| SORT cluster
+```
+
+| cost:double | step:datetime | cluster:keyword |
+| ----------- | ------------------------ | --------------- |
+| 10797.0 | 2024-05-10T00:00:00.000Z | qa |
+| 7403.0 | 2024-05-10T00:00:00.000Z | staging |
+
+### Ad-hoc query with inferred step
+
+For queries outside Kibana, set `start` and `end` explicitly. The step and range selector window are still inferred from
+the time range and the default `buckets` value:
+
+```esql
+PROMQL index=metrics-*
+ start="2026-04-01T00:00:00Z"
+ end="2026-04-01T01:00:00Z"
+ sum by (instance) (rate(http_requests_total))
+```
+
+### Bucket count instead of fixed step
+
+```esql
+PROMQL index=metrics-*
+ buckets=50
+ start="2026-04-01T00:00:00Z"
+ end="2026-04-01T01:00:00Z"
+ sum(rate(http_requests_total))
+```
+
+---
+
+## Post-Processing with ES|QL
+
+Because `PROMQL` is a source command, its output flows into the rest of the pipeline. Use ES|QL commands after the
+PROMQL stage for further aggregation, filtering, ordering, and enrichment:
+
+```esql
+PROMQL index=k8s step=1h bytes=(max by (cluster) (network.bytes_in))
+| STATS max_bytes = MAX(bytes) BY cluster
+| SORT cluster
+```
+
+| max_bytes:double | cluster:keyword |
+| ---------------- | --------------- |
+| 931.0 | prod |
+| 972.0 | qa |
+| 238.0 | staging |
+
+### Enrich with LOOKUP JOIN
+
+Join PromQL results with a lookup index using a grouping label as the join key:
+
+```esql
+PROMQL index=metrics-*
+ http_rate=(sum by (instance) (rate(http_requests_total)))
+| LOOKUP JOIN instance_metadata ON instance
+```
+
+This pattern combines PromQL's expressiveness for time series math with ES|QL's strengths for joining external metadata,
+filtering, and shaping output.
+
+---
+
+## PROMQL vs TS
+
+| Aspect | `PROMQL` | `TS` |
+| ------------------- | ------------------------------------------- | -------------------------------------------------- |
+| Syntax | Prometheus Query Language | ES\|QL inner/outer aggregation |
+| Default index | `metrics-*` | None — caller must specify |
+| Time filtering | `start`/`end` options or Kibana date picker | `WHERE TRANGE(...)` or `WHERE @timestamp ...` |
+| Bucketing | `step` / `buckets` options | `BY TBUCKET(interval)` |
+| Range vector window | Implicit (`max(step, scrape_interval)`) | Bucket interval, or sliding window arg (9.3+) |
+| Counter aggregation | `sum(rate(metric))` | `STATS SUM(RATE(metric)) BY TBUCKET(...)` |
+| Gauge aggregation | `avg_over_time(metric[5m])` | `STATS AVG(AVG_OVER_TIME(metric)) BY TBUCKET(...)` |
+| Label filtering | `metric{cluster="prod"}` | `WHERE cluster == "prod"` |
+| Available since | 9.4 (preview) | 9.2 (preview) |
+
+Both commands target TSDS indices and can be followed by the same set of ES|QL processing commands (`WHERE`, `EVAL`,
+`STATS`, `SORT`, `LIMIT`, `LOOKUP JOIN`, etc.).
+
+---
+
+## Limitations
+
+In 9.4 preview, `PROMQL` has the following limitations:
+
+- **Group modifiers are not supported.** Constructs like `on(chip) group_left(chip_name)` will fail. Use `LOOKUP JOIN`
+ in ES|QL after the PROMQL stage to attach extra labels.
+- **Set operators are not supported.** `or`, `and`, and `unless` between PromQL expressions are unavailable. Express set
+ logic in ES|QL after the PROMQL stage instead.
+- **Some PromQL functions are unavailable.** Notably `histogram_quantile`, `predict_linear`, and `label_join` are not
+ supported. Use `TS` with `PERCENTILE_OVER_TIME` for percentile-style metrics, or compute equivalents in ES|QL.
+- **Time bucket alignment differs.** Buckets align to fixed calendar boundaries rather than the query start time. This
+ can cause slight differences from native Prometheus, especially for short ranges or large step sizes.
+- **Index defaults to `metrics-*`.** If your TSDS data lives elsewhere, always set `index` explicitly to avoid scanning
+ unrelated indices.
+- **Preview status.** Behavior, supported PromQL surface, and option names may evolve before GA.
+
+When a question requires a feature in this list, fall back to the [`TS` command](time-series-queries.md) and express the
+equivalent computation in ES|QL.
+
+---
+
+## Kibana Time Filtering
+
+When writing `PROMQL` queries for Kibana (Discover, dashboards, alerts), **do not set `start` and `end` manually**.
+Kibana injects the date picker's range automatically and the engine derives `step` from it. Setting `start`/`end`
+explicitly overrides the date picker.
+
+```esql
+// Kibana — let the date picker drive start/end and step
+PROMQL index=metrics-* sum by (instance) (rate(http_requests_total))
+```
+
+For ad-hoc queries outside Kibana (HTTP API, `node scripts/esql.js raw "..."`), set `start` and `end` explicitly.
+
+---
+
+## Guidelines
+
+- **Prefer `PROMQL` only when the user explicitly thinks in PromQL** or is porting a Prometheus query/dashboard.
+ Otherwise, prefer `TS` — it integrates more naturally with the rest of ES|QL and is GA in 9.4.
+- **Always set `index`** in production queries instead of relying on the `metrics-*` default — narrower patterns reduce
+ scan volume and prevent accidental matches against unrelated indices.
+- **Use named results** (`http_rate=(...)`) when chaining further ES|QL commands. Named columns are easier to reference
+ than the raw PromQL expression text.
+- **Omit range selectors for adaptive dashboards.** Implicit range selectors (`rate(http_requests_total)` without
+ `[5m]`) make the query scale with the date picker.
+- **Pick `step` or `buckets`, not both.** Use `buckets` when you want a target panel resolution; use `step` when you
+ need a fixed grain (e.g., to align with downstream aggregation).
+- **Fall back to `TS` for unsupported features.** Histograms (`histogram_quantile`), set logic (`or`/`and`/`unless`),
+ group modifiers, and `label_join` are not available — express the computation with ES|QL primitives instead.
+- **Do not mix `WHERE @timestamp` filters with `start`/`end`.** Time filtering belongs in the PROMQL options or via
+ Kibana's date picker; standard ES|QL `WHERE` clauses run _after_ the PromQL stage and don't bound the metric scan.
+
+---
+
+## References
+
+- [ES|QL PROMQL command](https://www.elastic.co/docs/reference/query-languages/esql/commands/promql) — official
+ documentation
+- [Prometheus Query Language](https://prometheus.io/docs/prometheus/latest/querying/basics/) — PromQL fundamentals
+- [Prometheus HTTP API](https://prometheus.io/docs/prometheus/latest/querying/api/#range-queries) — origin of the option
+ semantics
+- [Time series data streams (TSDS)](https://www.elastic.co/docs/manage-data/data-store/data-streams/time-series-data-stream-tsds)
+- [time-series-queries.md](time-series-queries.md) — `TS` command and ES|QL native time series functions
+- [esql-version-history.md](esql-version-history.md) — feature availability by Elasticsearch version
diff --git a/plugins/elasticsearch/skills/elasticsearch-esql/references/query-patterns.md b/plugins/elasticsearch/skills/elasticsearch-esql/references/query-patterns.md
index bd47dd8..50ff4bd 100644
--- a/plugins/elasticsearch/skills/elasticsearch-esql/references/query-patterns.md
+++ b/plugins/elasticsearch/skills/elasticsearch-esql/references/query-patterns.md
@@ -437,6 +437,60 @@ FROM flights
| KEEP flight_id, destination, distance, avg_dist
```
+### Grouped Top-N with LIMIT BY (Serverless)
+
+```text
+"top 3 error types per service"
+→
+FROM logs-*
+| WHERE @timestamp > NOW() - 24 hours AND level == "error"
+| STATS cnt = COUNT(*) BY service.name, error.type
+| SORT cnt DESC
+| LIMIT 3 BY service.name
+```
+
+```text
+"most recent event per user"
+→
+// Serverless: use LATEST to get the most recent value per group
+FROM events-*
+| STATS last_action = LATEST(event.action), last_ts = LATEST(@timestamp) BY user.name
+
+// Alternative with LIMIT BY
+FROM events-*
+| SORT @timestamp DESC
+| LIMIT 1 BY user.name
+| KEEP user.name, @timestamp, event.action
+```
+
+### Subquery Composition (Serverless tech preview)
+
+```text
+"combine web server errors and application errors into one view"
+→
+FROM
+ (FROM web_logs
+ | WHERE @timestamp > NOW() - 1 hour AND status_code >= 500
+ | EVAL source = "web"
+ | KEEP @timestamp, message, service.name, source),
+ (FROM app_logs
+ | WHERE @timestamp > NOW() - 1 hour AND level == "error"
+ | EVAL source = "app"
+ | KEEP @timestamp, message, service.name, source)
+| SORT @timestamp DESC
+| LIMIT 100
+```
+
+```text
+"count errors from different log sources by service"
+→
+FROM
+ (FROM web_logs | WHERE status_code >= 500 | KEEP @timestamp, service.name),
+ (FROM app_logs | WHERE level == "error" | KEEP @timestamp, service.name)
+| STATS errors = COUNT(*) BY service.name
+| SORT errors DESC
+```
+
### MATCH_PHRASE (8.19/9.1+)
```text
diff --git a/plugins/elasticsearch/skills/elasticsearch-esql/references/time-series-queries.md b/plugins/elasticsearch/skills/elasticsearch-esql/references/time-series-queries.md
index d4afc01..a4d5f08 100644
--- a/plugins/elasticsearch/skills/elasticsearch-esql/references/time-series-queries.md
+++ b/plugins/elasticsearch/skills/elasticsearch-esql/references/time-series-queries.md
@@ -3,12 +3,25 @@
Query metrics data in Elasticsearch using the `TS` source command and time series aggregation functions. Requires
Elasticsearch 9.2+.
+> **Status:** `TS`, **all** time series aggregation functions (the 9.2-introduced set — `RATE`, `IRATE`, `INCREASE`,
+> `DELTA`, `IDELTA`, `AVG_OVER_TIME`, `SUM_OVER_TIME`, `MIN_OVER_TIME`, `MAX_OVER_TIME`, `FIRST_OVER_TIME`,
+> `LAST_OVER_TIME`, `COUNT_OVER_TIME`, `COUNT_DISTINCT_OVER_TIME`, `PRESENT_OVER_TIME`, `ABSENT_OVER_TIME` — and the
+> 9.3-introduced set — `DERIV`, `PERCENTILE_OVER_TIME`, `STDDEV_OVER_TIME`, `VARIANCE_OVER_TIME`), the `TBUCKET`
+> grouping function, the new `WITHOUT(...)` grouping function, and the new `METRICS_INFO` and `TS_INFO` discovery
+> commands are **GA since 9.4** (preview from 9.2 to 9.3 for the 9.2/9.3 features; new in 9.4 for `WITHOUT`,
+> `METRICS_INFO`, and `TS_INFO`). `TRANGE` remains in preview. **Looking for PromQL?** Elasticsearch 9.4+ also exposes a
+> `PROMQL` source command for running Prometheus Query Language directly against TSDS indices. See
+> [promql-command.md](promql-command.md). Prefer `PROMQL` only when the user explicitly thinks in PromQL or is migrating
+> Prometheus dashboards/alerts; otherwise prefer `TS` and the inner/outer aggregation paradigm described below.
+
## Table of Contents
- [TS Source Command](#ts-source-command)
- [Inner/Outer Aggregation Paradigm](#innerouter-aggregation-paradigm)
- [Time Series Aggregation Functions](#time-series-aggregation-functions)
- [TBUCKET Grouping Function](#tbucket-grouping-function)
+- [WITHOUT Grouping Function](#without-grouping-function)
+- [Metric and Time Series Discovery](#metric-and-time-series-discovery)
- [TRANGE Time Filter](#trange-time-filter)
- [CLAMP Functions](#clamp-functions)
- [Kibana Time Filtering](#kibana-time-filtering)
@@ -24,6 +37,8 @@ Elasticsearch 9.2+.
[time series data streams (TSDS)](https://www.elastic.co/docs/manage-data/data-store/data-streams/time-series-data-stream-tsds).
It enables time series aggregation functions (`RATE`, `AVG_OVER_TIME`, etc.) inside `STATS`.
+**Availability:** Preview from 9.2 to 9.3, **GA since 9.4**. GA on Elastic Cloud Serverless.
+
**Syntax:**
```esql
@@ -36,6 +51,8 @@ TS index_pattern [METADATA fields]
- Enables inner/outer aggregation paradigm in `STATS`
- Cannot combine with `FORK` before `STATS` is applied
- Optimized for processing time series data; `FROM` may produce unexpected results on TSDS indices
+- When there is **no** `STATS` command in the query, `TS` returns rows sorted by `@timestamp` descending by default —
+ useful for listing recent values across many time series.
**Best practices:**
@@ -72,7 +89,9 @@ TS metrics | STATS AVG(LAST_OVER_TIME(memory_usage))
```
Since 9.3 (preview), use a time series function directly without an outer aggregation to get one value per time series
-per bucket:
+per bucket. The result is implicitly grouped by all dimensions of each time series and includes a `_timeseries` column
+with the dimension key/value pairs — see [WITHOUT Grouping Function](#without-grouping-function) for narrowing this
+grouping (`BY WITHOUT(dim, ...)`, GA since 9.4).
```esql
TS metrics
@@ -99,11 +118,11 @@ is used as the window.
For fields with `time_series_metric: counter` (`counter_double`, `counter_integer`, `counter_long`).
-| Function | Description | Since |
-| ---------- | ---------------------------------------------------------------------- | ----- |
-| `RATE` | Per-second average rate of increase; handles counter resets | 9.2 |
-| `IRATE` | Per-second rate between the last two data points; responsive to spikes | 9.2 |
-| `INCREASE` | Absolute increase of the counter in the time window; handles resets | 9.2 |
+| Function | Description | Since | Status |
+| ---------- | ---------------------------------------------------------------------- | ------------- | -------- |
+| `RATE` | Per-second average rate of increase; handles counter resets | 9.2 (preview) | GA (9.4) |
+| `IRATE` | Per-second rate between the last two data points; responsive to spikes | 9.2 (preview) | GA (9.4) |
+| `INCREASE` | Absolute increase of the counter in the time window; handles resets | 9.2 (preview) | GA (9.4) |
```esql
// Average rate per host per hour
@@ -127,22 +146,22 @@ TS k8s
For gauge metrics and general numeric fields (`double`, `integer`, `long`, `aggregate_metric_double`).
-| Function | Description | Since |
-| -------------------------- | -------------------------------------------------- | ----- |
-| `AVG_OVER_TIME` | Average value over the time window | 9.2 |
-| `SUM_OVER_TIME` | Sum of values over the time window | 9.2 |
-| `MIN_OVER_TIME` | Minimum value over the time window | 9.2 |
-| `MAX_OVER_TIME` | Maximum value over the time window | 9.2 |
-| `FIRST_OVER_TIME` | Earliest value by `@timestamp` | 9.2 |
-| `LAST_OVER_TIME` | Latest value by `@timestamp` (implicit default) | 9.2 |
-| `COUNT_OVER_TIME` | Count of values over the time window | 9.2 |
-| `COUNT_DISTINCT_OVER_TIME` | Count of distinct values over the time window | 9.2 |
-| `PERCENTILE_OVER_TIME` | Percentile of values; takes `(field, percentile)` | 9.3 |
-| `STDDEV_OVER_TIME` | Population standard deviation over the time window | 9.3 |
-| `VARIANCE_OVER_TIME` | Population variance over the time window | 9.3 |
-| `DELTA` | Absolute change of a gauge in the time window | 9.2 |
-| `IDELTA` | Change between the last two data points only | 9.2 |
-| `DERIV` | Derivative over time using linear regression | 9.3 |
+| Function | Description | Since | Status |
+| -------------------------- | -------------------------------------------------- | ------------- | -------- |
+| `AVG_OVER_TIME` | Average value over the time window | 9.2 (preview) | GA (9.4) |
+| `SUM_OVER_TIME` | Sum of values over the time window | 9.2 (preview) | GA (9.4) |
+| `MIN_OVER_TIME` | Minimum value over the time window | 9.2 (preview) | GA (9.4) |
+| `MAX_OVER_TIME` | Maximum value over the time window | 9.2 (preview) | GA (9.4) |
+| `FIRST_OVER_TIME` | Earliest value by `@timestamp` | 9.2 (preview) | GA (9.4) |
+| `LAST_OVER_TIME` | Latest value by `@timestamp` (implicit default) | 9.2 (preview) | GA (9.4) |
+| `COUNT_OVER_TIME` | Count of values over the time window | 9.2 (preview) | GA (9.4) |
+| `COUNT_DISTINCT_OVER_TIME` | Count of distinct values over the time window | 9.2 (preview) | GA (9.4) |
+| `PERCENTILE_OVER_TIME` | Percentile of values; takes `(field, percentile)` | 9.3 (preview) | GA (9.4) |
+| `STDDEV_OVER_TIME` | Population standard deviation over the time window | 9.3 (preview) | GA (9.4) |
+| `VARIANCE_OVER_TIME` | Population variance over the time window | 9.3 (preview) | GA (9.4) |
+| `DELTA` | Absolute change of a gauge in the time window | 9.2 (preview) | GA (9.4) |
+| `IDELTA` | Change between the last two data points only | 9.2 (preview) | GA (9.4) |
+| `DERIV` | Derivative over time using linear regression | 9.3 (preview) | GA (9.4) |
```esql
// Average memory per cluster per 5 minutes
@@ -171,10 +190,10 @@ TS k8s
Detect whether a field has data in a given time window. Return `boolean`.
-| Function | Description | Since |
-| ------------------- | ----------------------------------------------- | ----- |
-| `PRESENT_OVER_TIME` | `true` if field has values in the window | 9.2 |
-| `ABSENT_OVER_TIME` | `true` if field has **no** values in the window | 9.2 |
+| Function | Description | Since | Status |
+| ------------------- | ----------------------------------------------- | ------------- | -------- |
+| `PRESENT_OVER_TIME` | `true` if field has values in the window | 9.2 (preview) | GA (9.4) |
+| `ABSENT_OVER_TIME` | `true` if field has **no** values in the window | 9.2 (preview) | GA (9.4) |
```esql
// Detect pods with missing data
@@ -182,10 +201,11 @@ TS k8s
| STATS missing = MAX(ABSENT_OVER_TIME(events_received)) BY pod, TBUCKET(2 minute)
```
-### Sliding Window (9.3+)
+### Sliding Window
-Pass a `time_duration` as the second argument to any time series function to use a sliding window larger than the bucket
-interval. The window must be a multiple of the `TBUCKET` interval.
+Pass a `time_duration` as the second argument to any time series function to use a sliding window for the
+per-time-series aggregation. The window is orthogonal to time bucketing of output results (`TBUCKET`). If the window is
+omitted, the `TBUCKET` interval is used implicitly.
```esql
// Average rate per host over a 10-minute sliding window, bucketed by 1 minute
@@ -195,6 +215,15 @@ TS metrics
| STATS AVG(RATE(requests, 10m)) BY TBUCKET(1m), host
```
+**Version behavior:**
+
+- **9.2-9.3 (preview):** the window must be a **multiple of the `TBUCKET` interval** in the `BY` clause (for example,
+ with `TBUCKET(1m)` you may use `1m`, `2m`, `10m`; `7m` is rejected). If no window is specified, the `TBUCKET` interval
+ is used implicitly.
+- **9.4+ (GA):** all window values are accepted, with performance optimizations when the window is a multiple of the
+ `TBUCKET` interval. **Restriction:** within a single query you cannot mix windows that are **smaller** than the bucket
+ interval for one metric with windows that are **larger** than the bucket interval for another metric.
+
---
## TBUCKET Grouping Function
@@ -214,7 +243,7 @@ The interval is a time duration (`1 hour`, `5 minute`, `30s`) or date period (`1
`TBUCKET` is the preferred bucketing function for `TS` queries. It has a simpler signature than
`DATE_TRUNC(interval, @timestamp)` and is aware of time series semantics.
-**Availability:** Preview since 9.2.
+**Availability:** Preview from 9.2 to 9.3, **GA since 9.4**.
```esql
// 1-hour buckets
@@ -229,6 +258,161 @@ TS metrics
---
+## WITHOUT Grouping Function
+
+When the first `STATS` after `TS` uses a **bare** time series aggregation function (one not wrapped in an outer
+aggregation such as `AVG()` or `SUM()`), rows are implicitly grouped by **all** dimensions of each time series. The
+output includes a `_timeseries` `keyword` column containing a JSON-encoded object with the dimension key/value pairs
+identifying each group. Only the dimensions that actually exist for a given time series appear in `_timeseries` — not
+every dimension declared in the index mappings — so different rows in the result may carry different dimension keys.
+
+`WITHOUT(...)` lets you make this grouping explicit, or narrow it to a subset of dimensions:
+
+- `BY WITHOUT(dim1, dim2, ...)` groups by **all** dimensions **except** those listed.
+- `BY WITHOUT()` (no arguments) explicitly groups by every dimension; it is equivalent to the implicit "group by all"
+ behavior.
+
+When combining a bare time series function with other groupings, **only grouping functions** (`TBUCKET`, `WITHOUT`) are
+allowed in the `BY` clause — bare dimension columns are rejected. For example:
+
+```esql
+// INVALID -- bare time series function with a bare dimension column in BY
+TS k8s | STATS rate(network.total_bytes_in) BY host
+```
+
+Use `BY TBUCKET(...)` and/or `BY WITHOUT(...)`, or wrap the time series function with an outer aggregation.
+
+**Availability:** **GA since 9.4**. Can only be used in the **first** `STATS` command under a `TS` source — using it in
+a `FROM | STATS ... BY WITHOUT(...)` query is rejected.
+
+**Examples:**
+
+```esql
+// Group by every dimension implicitly — _timeseries column carries the dimension labels
+TS k8s
+| STATS avg = AVG_OVER_TIME(network.cost)
+| SORT avg DESC
+
+// Group by every dimension EXCEPT pod
+TS k8s
+| STATS total_cost = SUM(network.cost) BY WITHOUT(pod)
+| SORT total_cost
+
+// Combine WITHOUT with TBUCKET to add a time bucket to the surviving dimensions
+TS k8s
+| STATS total_cost = SUM(network.cost) BY WITHOUT(pod), tbucket = TBUCKET(1 hour)
+| SORT total_cost
+
+// Equivalent to implicit grouping (group by all dimensions)
+TS k8s
+| STATS avg = AVG_OVER_TIME(network.cost) BY WITHOUT()
+```
+
+---
+
+## Metric and Time Series Discovery
+
+Two processing commands introduced in 9.4 expose the metric catalogue of a TSDS so you can discover what to query
+without inspecting index mappings or calling the field capabilities API. Both must follow a `TS` source command and must
+appear before pipeline-breaking commands (`STATS`, `SORT`, `LIMIT`).
+
+| Command | Granularity | Status | Adds beyond `METRICS_INFO` |
+| -------------- | ------------------------------------- | ------------ | ---------------------------------------------------------------- |
+| `METRICS_INFO` | One row per **metric** | GA since 9.4 | — |
+| `TS_INFO` | One row per **(metric, time series)** | GA since 9.4 | `dimensions` JSON column with the labels identifying each series |
+
+### METRICS_INFO
+
+Returns one row per distinct metric in the targeted TSDS, with applicable dimensions and metadata. Useful for "what
+metrics exist in this stream?" and "what dimensions apply to metric X?".
+
+**Output columns** (all `keyword`):
+
+- `metric_name` — single-valued
+- `data_stream` — multi-valued when several streams align on unit/metric_type/field_type
+- `unit` — declared unit; may be `null` or multi-valued
+- `metric_type` — `counter`, `gauge`, etc.
+- `field_type` — `long`, `double`, `integer`, etc.
+- `dimension_fields` — union of dimension field names across the series for the metric
+
+```esql
+// List every metric, alphabetically
+TS k8s
+| METRICS_INFO
+| SORT metric_name
+
+// Restrict to metrics that have data matching a filter
+TS k8s
+| WHERE cluster == "prod"
+| METRICS_INFO
+| SORT metric_name
+
+// Filter by metric type, count by it
+TS k8s
+| METRICS_INFO
+| STATS metric_count = COUNT(*) BY metric_type
+| SORT metric_type
+
+// Find metrics whose name matches a pattern
+TS k8s
+| METRICS_INFO
+| WHERE metric_name LIKE "network.eth0*"
+| SORT metric_name
+```
+
+### TS_INFO
+
+Returns one row per (metric, time series) combination, including the dimension key/value pairs that identify each
+series. Useful for "which time series report this metric?" and "what label combinations exist?".
+
+`TS_INFO` includes **all `METRICS_INFO` columns** plus a `dimensions` column — a JSON-encoded object such as
+`{"job":"elasticsearch","instance":"instance_1"}` (single-valued).
+
+```esql
+// Every (metric, time series) pair in the data stream
+TS k8s
+| TS_INFO
+| SORT metric_name, dimensions
+
+// Filter the underlying series before discovery
+TS k8s
+| WHERE cluster == "prod"
+| TS_INFO
+| KEEP metric_name, dimensions
+| SORT metric_name, dimensions
+
+// Filter by metadata after TS_INFO (gauges only)
+TS k8s
+| TS_INFO
+| WHERE metric_type == "gauge"
+| SORT metric_name, dimensions
+
+// Count distinct time series per metric
+TS k8s
+| TS_INFO
+| STATS series_count = COUNT(*) BY metric_name
+| SORT metric_name
+
+// Count distinct metrics per time series — spot under- or over-reporting series
+TS k8s
+| TS_INFO
+| STATS metric_count = COUNT_DISTINCT(metric_name) BY dimensions
+| SORT dimensions
+```
+
+### Guidelines
+
+- **Use these for TSDS schema discovery** before writing `RATE`/`AVG_OVER_TIME` queries — they replace the older
+ workflow of inspecting `_settings`, `_mapping`, or field capabilities for time series indices.
+- **Reach for `METRICS_INFO` first** to enumerate metrics; reach for `TS_INFO` only when you need the exact dimension
+ combinations (label sets) of individual series.
+- **Filter before discovery** with `WHERE` on dimension fields to scope the catalogue to a relevant subset of series.
+- **The output replaces the original table.** Anything you `STATS` / `SORT` / `LIMIT` afterwards operates on metadata
+ rows, not raw documents — there's no way to re-attach the data points after `METRICS_INFO` or `TS_INFO`.
+- **Both commands are TSDS-only.** `FROM | METRICS_INFO` and `FROM | TS_INFO` are rejected.
+
+---
+
## TRANGE Time Filter
Filter data by time range using `@timestamp`. Prefer `TRANGE` over manual `WHERE @timestamp > NOW() - ...` filters.
@@ -376,7 +560,7 @@ TS k8s
| STATS SUM(IRATE(network.total_bytes_in)) BY cluster, TBUCKET(10 minute)
```
-### Sliding Window Rate (9.3+)
+### Sliding Window Rate
```esql
TS metrics
@@ -384,6 +568,9 @@ TS metrics
| STATS AVG(RATE(requests, 10m)) BY TBUCKET(1m), host
```
+In 9.2-9.3 the window must be a multiple of the `TBUCKET` interval. In 9.4+ any window value is accepted (mixing windows
+smaller and larger than the bucket interval for different metrics in the same query is not supported).
+
---
## Guidelines
@@ -395,9 +582,15 @@ TS metrics
- **Always add a time range filter** with `TRANGE` (or `WHERE @timestamp`) to limit scan volume, except in Kibana where
the date picker handles this automatically. Don't add a range filter if the user explicitly asks not to add it.
- **Version requirements:**
- - `TS`, `TBUCKET`, time series functions: 9.2+ (preview)
- - `TRANGE`, sliding windows, `DERIV`, `STDDEV_OVER_TIME`, `VARIANCE_OVER_TIME`, `PERCENTILE_OVER_TIME`: 9.3+ (preview)
- - `CLAMP`, `CLAMP_MIN`, `CLAMP_MAX`: 9.3+ (preview)
+ - `TS`, `TBUCKET`, `WITHOUT`, `METRICS_INFO`, `TS_INFO`, and **all** time series aggregation functions (the
+ 9.2-introduced set — `RATE`, `IRATE`, `INCREASE`, `DELTA`, `IDELTA`, all `*_OVER_TIME` from 9.2,
+ `PRESENT_OVER_TIME`, `ABSENT_OVER_TIME` — and the 9.3-introduced set — `DERIV`, `PERCENTILE_OVER_TIME`,
+ `STDDEV_OVER_TIME`, `VARIANCE_OVER_TIME`): **GA since 9.4** (preview from 9.2-9.3 for the 9.2/9.3 functions; new in
+ 9.4 for `WITHOUT`, `METRICS_INFO`, `TS_INFO`).
+ - `TRANGE`: 9.3+ (preview).
+ - Sliding window parameter (second argument to time series functions): introduced in 9.2-9.3 (preview, restricted to
+ multiples of the `TBUCKET` interval); **GA in 9.4** with arbitrary durations.
+ - `CLAMP`, `CLAMP_MIN`, `CLAMP_MAX`: 9.3+ (preview).
- **Do not nest time series functions.** `AVG_OVER_TIME(RATE(field))` is invalid. Use a standard aggregation as the
outer function.
- **Avoid mixing metrics with different dimensions** in one query. If `foo` and `bar` have different dimension values,
@@ -406,6 +599,9 @@ TS metrics
## References
- [ES|QL TS Command](https://www.elastic.co/docs/reference/query-languages/esql/commands/ts)
+- [ES|QL PROMQL Command](https://www.elastic.co/docs/reference/query-languages/esql/commands/promql) — alternative
+ source command using PromQL syntax (9.4+ preview)
+- [promql-command.md](promql-command.md) — PROMQL command reference in this skill
- [Time Series Aggregation Functions](https://www.elastic.co/docs/reference/query-languages/esql/functions-operators/time-series-aggregation-functions)
- [TBUCKET Function](https://www.elastic.co/docs/reference/query-languages/esql/functions-operators/grouping-functions/tbucket)
- [TRANGE Function](https://www.elastic.co/docs/reference/query-languages/esql/functions-operators/date-time-functions/trange)
diff --git a/plugins/elasticsearch/skills/elasticsearch-onboarding/SKILL.md b/plugins/elasticsearch/skills/elasticsearch-onboarding/SKILL.md
index 35a39f6..979d3be 100644
--- a/plugins/elasticsearch/skills/elasticsearch-onboarding/SKILL.md
+++ b/plugins/elasticsearch/skills/elasticsearch-onboarding/SKILL.md
@@ -3,8 +3,9 @@ name: elasticsearch-onboarding
description: >
Help developers new to Elasticsearch get from zero to a working search experience.
Guide them through understanding their intent, mapping their data, and building
- a search experience with best practices baked in. Use this when developers are new
- to Elasticsearch and need help getting started with their search use case.
+ a search experience with best practices baked in. Use this when the user shows intent
+ to build search-related functionality, asks about Elasticsearch-related concepts
+ for their use case, or expresses the need for help getting started with Elasticsearch.
compatibility: Elasticsearch 9.x
metadata:
author: elastic
@@ -29,6 +30,16 @@ Example user intents that should trigger this skill:
- "What are the best practices for building a search experience?"
- "Can you help me understand how to model my data for search?"
- "How do I build a vector database?"
+- "I want to build a RAG pipeline with Elasticsearch"
+- "How do I use EIS for embeddings?"
+- "How do I connect an LLM to Elasticsearch?"
+- "How do I do kNN search in Elasticsearch?"
+- "How do I use ELSER for semantic search?"
+- "How do I set up the Elasticsearch MCP?"
+- "How do I combine keyword and vector results with RRF?"
+- "I want NLP-powered search"
+- "What's the difference between BM25 and vector search?"
+- "Can I use ES|QL to query my data?"
## Guidelines
diff --git a/plugins/elasticsearch/skills/elasticsearch-onboarding/references/elasticsearch-onboarding-playbook.md b/plugins/elasticsearch/skills/elasticsearch-onboarding/references/elasticsearch-onboarding-playbook.md
index 80f840e..c5366f5 100644
--- a/plugins/elasticsearch/skills/elasticsearch-onboarding/references/elasticsearch-onboarding-playbook.md
+++ b/plugins/elasticsearch/skills/elasticsearch-onboarding/references/elasticsearch-onboarding-playbook.md
@@ -218,8 +218,8 @@ explanations:
> Does this look right, or would you add/remove anything?
**Surface the hybrid option when it adds value.** If the use case involves descriptive or natural-language queries,
-recommend semantic search alongside keyword. Explain the tradeoff: requires an embedding model (ELSER built-in, or
-OpenAI/Cohere), slightly slower indexing — but catches meaning-based queries keywords miss.
+recommend semantic search alongside keyword. Explain the tradeoff: requires an embedding model served via EIS (managed)
+or a user-provided inference endpoint, slightly slower indexing — but catches meaning-based queries keywords miss.
**For RAG retrieval**, recommend hybrid if documents contain specific terms or codes users will search for exactly
(policy names, product IDs, error codes).
diff --git a/plugins/kibana/.claude-plugin/plugin.json b/plugins/kibana/.claude-plugin/plugin.json
index 48db7e5..af49bcc 100644
--- a/plugins/kibana/.claude-plugin/plugin.json
+++ b/plugins/kibana/.claude-plugin/plugin.json
@@ -1,5 +1,5 @@
{
- "version": "0.2.4",
+ "version": "0.3.0",
"name": "elastic-kibana",
"description": "Kibana skills - dashboards, alerting rules, connectors, Vega visualizations, Agent Builder, streams, and audit logging",
"author": {
diff --git a/plugins/kibana/plugin.json b/plugins/kibana/plugin.json
index 0226ce5..2abc260 100644
--- a/plugins/kibana/plugin.json
+++ b/plugins/kibana/plugin.json
@@ -1,7 +1,7 @@
{
"name": "elastic-kibana",
"description": "Kibana skills - dashboards, alerting rules, connectors, Vega visualizations, Agent Builder, streams, and audit logging",
- "version": "0.2.4",
+ "version": "0.3.0",
"author": {
"name": "Elastic",
"url": "https://www.elastic.co"
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/SKILL.md b/plugins/kibana/skills/kibana-anomaly-detection/SKILL.md
new file mode 100644
index 0000000..5f03663
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/SKILL.md
@@ -0,0 +1,416 @@
+---
+name: kibana-anomaly-detection
+description: Elastic ML anomaly detection skill — investigation/RCA, score explanation,
+ job operations (create, datafeed, start/stop, results), and troubleshooting (missing
+ docs, memory limits, datafeed health, lifecycle). Operates against Kibana Agent
+ Builder MCP tools (`ad_*`) on `.ml-anomalies-*`, `.ml-config`, `.ml-notifications-*`,
+ `.ml-annotations-*`. Use when answering "what broke?"/"which entity?"/RCA, "why
+ is score high/low?"/renormalization, "datafeed stopped"/"memory limit", or any request
+ to set up or configure an ML anomaly detection job.
+metadata:
+ author: elastic
+ version: 0.2.0
+compatibility: Kibana 8.x–9.x with Agent Builder and Workflows; Elasticsearch 8.x–9.x
+ with machine learning
+---
+
+# Elastic ML Anomaly Detection
+
+Single skill covering all anomaly detection work against **Kibana Agent Builder** MCP at
+`{KIBANA_URL}/api/agent_builder/mcp`. Use the **Mode Selector** below to pick the right approach for the user's question
+— modes share the same tool surface and concepts.
+
+## Platform
+
+- Read path: ES|QL against `.ml-anomalies-*`, `.ml-config`, `.ml-notifications-*`, `.ml-annotations-*`
+- Always-available: `platform.core.execute_esql` (plus additional platform tools for search, index mapping, and
+ documentation — see `scripts/agent_builder_constants.json`)
+- ML API spec (if available): `.kibana_ai_openapi_spec_elasticsearch` — see
+ [references/anomaly-detection-openapi-spec-discover.md](references/anomaly-detection-openapi-spec-discover.md) for
+ discovery pattern.
+- **Run `ad_validate_ml_tool_permissions` first** when tools return empty/misleading results — missing privileges are
+ the most common cause of false negatives. Full permissions matrix:
+ [references/permissions-matrix.md](references/permissions-matrix.md).
+
+## Mode Selector
+
+| User intent | Mode |
+| ----------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------ |
+| "What broke?" / RCA / cross-job / blast radius / influencers / log categories | **Investigate** |
+| "Why score high/low?" / renormalization / model bounds / forecasts | **Explain** |
+| Missing docs / memory limit / datafeed stopped / CCS / lifecycle / calendars | **Troubleshoot** |
+| Create a job / configure a datafeed / start analysis / retrieve results | **Manage** |
+| Security framing (attack chains, MITRE, exfil) | Investigate + [references/security-anomaly-expert.md](references/security-anomaly-expert.md) |
+| Observability/SRE framing (degradation, capacity, deployment regression) | Investigate + [references/observability-anomaly-expert.md](references/observability-anomaly-expert.md) |
+
+When a question spans modes: **Investigate → Explain → Troubleshoot**. Don't blend mode logic — finish one before moving
+on.
+
+---
+
+## Score Quick Reference
+
+- `record_score` bands: **>75** critical · **50–75** warning · **25–50** minor · **<25** informational
+- `multi_bucket_impact ≥ 3` → sustained shift (not a transient spike)
+- `initial_record_score >> record_score` → renormalization (model saw worse anomalies later)
+- `actual << typical` with `count`/`low_count`/`low_mean` → absence/outage, not just low value
+- Low scores across many jobs > one high score — composite cross-job signal often beats single-detector severity
+
+> Full score definitions, renormalization mechanics, and `anomaly_score_explanation` components:
+> [references/score-reference.md](references/score-reference.md).
+
+## Core concepts
+
+Treat `.ml-anomalies-*` as three layers, accessed via `result_type`:
+
+- **`bucket`** — bucket-level unusualness per `bucket_span`. `anomaly_score` is the aggregate across all detectors.
+- **`record`** — finest-grained rows with `actual` vs `typical`, `probability`, `record_score`,
+ `anomaly_score_explanation`.
+- **`influencer`** — entity contributions ranked within a bucket (`influencer_score`).
+
+Read scores this way:
+
+- `anomaly_score` / `record_score` = **current normalized** values (move as the model sees new extremes).
+- `initial_anomaly_score` / `initial_record_score` = **immutable snapshots** from detection time.
+- Compare `actual` to `typical`; use `probability` for raw likelihood.
+- Map entities via `partition_field_value` / `by_field_value` / `over_field_value`.
+- Read `multi_bucket_impact` (-5 to +5) to separate single-bucket spikes from sustained trends.
+
+---
+
+## Mode: Investigate — RCA
+
+**When:** "what broke?", "which entity caused this?", cross-job correlation, blast radius, attack/cascade chains.
+
+### Tool chain
+
+| Phase | Tools |
+| --------------------- | -------------------------------------------------------------------------------------------------------------- |
+| Discovery | `ad_get_available_metadata`, `ad_get_jobs`, `ad_discover_related_jobs`, `ad_discover_jobs_by_datafeed_index` |
+| Timeline / scope | `ad_query_anomaly_timeline` |
+| Cross-job / entities | `ad_rca_cross_job_entity_match`, `ad_rca_multi_job_entities`, `ad_rca_entity_profile` |
+| Records / influencers | `ad_query_anomaly_records`, `ad_query_influencers` |
+| RCA depth | `ad_rca_detector_fingerprint`, `ad_rca_correlation`, `ad_rca_blast_radius`, `ad_rca_score_reassessment` |
+| Evidence / categories | `ad_get_job_datafeed_config`, `ad_rca_source_evidence`, `ad_get_categories`, `ad_search_log_category_examples` |
+
+### Protocol
+
+Follow the 14-step sequence in [references/protocols/investigation.md](references/protocols/investigation.md). High
+level: `ad_get_available_metadata` → pair `ad_discover_jobs_by_datafeed_index` with `ad_discover_related_jobs` →
+`ad_query_anomaly_timeline` → rank with `ad_rca_multi_job_entities` (`min_job_count=2`) → `ad_rca_detector_fingerprint`
+→ drill with `ad_query_anomaly_records` + `ad_query_influencers` (low `min_score=25`) → profile with
+`ad_rca_entity_profile` → order with `ad_rca_correlation` → confirm with `ad_rca_source_evidence`. When
+`by_field_name == "mlcategory"`, compare with `ad_get_categories` + paired `ad_search_log_category_examples` (baseline
+vs. anomaly window).
+
+Finish with a written RCA: **root cause entity · affected jobs · temporal progression · fault class
+(resource/network/application) · severity · recommended actions**. Worked example:
+[references/worked-example.md](references/worked-example.md). Full ES|QL templates and parameters:
+[references/investigate-anomaly-esql-tools.md](references/investigate-anomaly-esql-tools.md).
+
+### Rules
+
+1. **Multi-job entities are prime suspects; single-job entities are usually victims.** Use `min_job_count=2`.
+2. **Earliest anomaly timestamp wins** — sort `ad_rca_correlation` by timestamp; first-appearing entity = origin.
+3. **`multi_bucket_impact ≥ 3` = sustained behavioral shift**, weight higher than transient spikes.
+4. **Never close an RCA without `ad_rca_source_evidence`** — raw source documents are ground truth.
+5. **Use low `min_score` (25 or lower) for influencer queries** — high thresholds miss correlated entities.
+
+---
+
+## Mode: Explain — Score / model behavior
+
+**When:** "why is my score 30/90?", "score dropped overnight", "what is renormalization?", "why wasn't this detected?".
+
+### Score types
+
+| Field | Scope | Meaning |
+| ---------------------- | --------------- | ----------------------------------------------------------------------- |
+| `record_score` | Single record | Normalized severity after renormalization. |
+| `initial_record_score` | Single record | Score at detection time. Gap vs `record_score` = renormalization drift. |
+| `anomaly_score` | Bucket | Aggregate severity across all detectors in a bucket. |
+| `influencer_score` | Entity × bucket | How anomalous a specific entity is in that bucket. |
+
+### `anomaly_score_explanation` components
+
+| Component | Effect | What it means |
+| -------------------------------- | ------- | ------------------------------------------------------------ |
+| `anomaly_length` | ↑ score | More consecutive anomalous buckets |
+| `single_bucket_impact` | ↑ score | Lower probability → higher impact |
+| `multi_bucket_impact` | ↑ score | Sustained pattern contribution |
+| `anomaly_characteristics_impact` | ↑ score | Mean shift vs. variance change |
+| `high_variance_penalty` | ↓ score | Noisy data → wide bounds → anomaly less surprising |
+| `incomplete_bucket_penalty` | ↓ score | Bucket has less data than expected (ingest lag, sparse data) |
+
+### Why a score looks wrong
+
+- **Unexpectedly low:** `high_variance_penalty`, renormalization, <3 weeks training for weekly seasonality,
+ `bucket_span` too large, wrong detector function (`mean` vs `high_mean`), `incomplete_bucket_penalty`, suppression by
+ `custom_rules`.
+- **Unexpectedly high:** insufficient history (early training over-flags), high-cardinality split (too few points per
+ entity), `use_null: true` on a sparse field.
+
+### Tool chain
+
+| Purpose | Tools |
+| ---------------------- | ---------------------------------------------------------------------------------- |
+| Records + explanation | `ad_query_anomaly_records` (exact `job_id_pattern`) |
+| Renormalization drift | `ad_rca_score_reassessment` (`score_drift = initial_record_score - record_score`) |
+| Model bounds (visual) | `ad_get_model_plot` — actual outside `model_lower`/`model_upper` = anomaly |
+| Forecast overlap | `ad_get_forecast_results` |
+| Influencer attribution | `ad_query_influencers` |
+| Config & detector | `ad_get_job_datafeed_config` — `bucket_span`, function, `custom_rules`, `use_null` |
+| Categorization | `ad_get_categories` |
+| Model snapshots | `ad_get_model_snapshots` |
+| Structured diagnostic | **`ad_wf_troubleshoot_anomaly_score`** (full decision tree) |
+
+### Decision tree (`ad_wf_troubleshoot_anomaly_score`)
+
+1. `ad_get_jobs` — ≥3 weeks data for weekly seasonality?
+2. `ad_ts_model_memory_health` — `memory_status` healthy?
+3. `ad_ts_delayed_data_annotations` — no incomplete buckets?
+4. `ad_query_anomaly_records` — compare `record_score` vs `initial_record_score`.
+5. `ad_get_job_datafeed_config` — `bucket_span`, detector function, `custom_rules`, `use_null`.
+6. `ad_get_model_plot` — wide bounds → `high_variance_penalty`.
+7. `ad_rca_score_reassessment` — renormalization drift across history.
+8. Explain `anomaly_score_explanation` factors.
+
+### Rules
+
+1. **Always show both `initial_record_score` and `record_score`** — the gap is the renormalization story.
+2. **Explain renormalization before diagnosing config** — score drift is the most common "score dropped" cause and needs
+ no config change.
+3. **`actual << typical` with `count`/`low_count` is an absence anomaly** — distinguish outages from value spikes.
+4. **`high_variance_penalty` and `incomplete_bucket_penalty` explain most "low score" surprises** without remediation.
+5. **Weekly seasonality needs ≥3 weeks of training data** — flag young jobs as the cause.
+
+For detector function selection details, see
+[references/anomaly-detection-functions.md](references/anomaly-detection-functions.md).
+
+---
+
+## Mode: Troubleshoot — Job ops
+
+**When:** "missing documents", "datafeed stopped", "hard_limit", "results look wrong", lifecycle changes, calendars,
+CCS.
+
+### Common issues → fast paths
+
+| Issue | Fast path | Full decision tree |
+| ------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------- |
+| Missing docs / `query_delay` warning | `ad_ts_delayed_data_annotations` → `ad_ts_bucket_event_gaps` → `ad_ts_ingest_latency_estimate` → `ad_update_datafeed_query_delay` | `ad_wf_troubleshoot_query_delay` |
+| Memory `soft_limit` / `hard_limit` | `ad_ts_model_memory_health` → `ad_wf_ts_field_cardinality` → `ad_estimate_memory_requirement` → `ad_update_model_memory_limit` | `ad_wf_troubleshoot_memory_limit` |
+| Datafeed not running / job state | `ad_get_jobs` (state) → `ad_get_job_messages` → `ad_manage_datafeed` | — |
+| CCS / `remote_cluster:` indices | `ad_ts_ccs_diagnostics` | — |
+| Score sanity check | — | `ad_wf_troubleshoot_anomaly_score` |
+
+> `hard_limit` corrupts model state and causes downstream missing-doc false alarms (categorizer silently skips events
+> for unknown categories). **Fix memory before fixing `query_delay`.**
+
+### Memory concepts
+
+| Field | Meaning |
+| ----------------------------------- | ------------------------------------------------------- |
+| `model_bytes` | Current memory used |
+| `peak_model_bytes` | High-water mark since job opened |
+| `model_bytes_memory_limit` | Configured `model_memory_limit` |
+| `memory_status` | `ok` / `soft_limit` (pruning) / `hard_limit` (critical) |
+| `total_by_field_count > 100k` | `by_field` cardinality too high — dominant driver |
+| `total_partition_field_count > 10k` | Partition explosion |
+| `total_category_count > 10k` | Too many distinct log patterns |
+
+Prefer **`ad_estimate_memory_requirement`** (samples cardinality from source, calls Estimate Model Memory API) over
+heuristics like `peak_model_bytes * 1.3` — the heuristic ignores pure influencer and categorization memory.
+
+### Datafeed & timing concepts
+
+- **`query_delay`** — how far behind real time the datafeed queries. Too small → missing docs; too large → slower
+ alerts. Set to **P95 ingest latency + buffer** (default `60s`–`120s`).
+- **`delayed_data_check_config`** — how aggressively the datafeed checks for late data.
+- **`bucket_span`** — analysis interval. Align with data granularity and detection window.
+- **`frequency`** — defaults to `min(query_delay, bucket_span / 2)`.
+
+### Lifecycle for config changes (memory limit, query_delay)
+
+1. Stop datafeed: `ad_manage_datafeed` (`action=_stop`)
+2. Close job
+3. Update config: `ad_update_model_memory_limit`, `ad_update_datafeed_query_delay`,
+ `ad_update_delayed_data_check_config`
+4. Open job: `ad_open_job`
+5. Start datafeed: `ad_manage_datafeed` (`action=_start`)
+
+Recover a corrupted period without resetting the whole model: `ad_revert_model_snapshot`.
+
+### Tool surface
+
+| Category | Tools |
+| ---------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| Permissions / metadata | `ad_validate_ml_tool_permissions`, `ad_get_available_metadata`, `ad_get_jobs` |
+| Job + datafeed state | `ad_get_job_datafeed_config`, `ad_get_job_messages`, `ad_manage_datafeed`, `ad_preview_datafeed_with_latency` |
+| Timing / missing docs | `ad_ts_delayed_data_annotations`, `ad_ts_bucket_event_gaps`, `ad_ts_ingest_latency_estimate`, `ad_update_datafeed_query_delay`, `ad_update_delayed_data_check_config`, `ad_wf_troubleshoot_query_delay` |
+| Memory | `ad_ts_model_memory_health`, `ad_wf_ts_field_cardinality`, `ad_estimate_memory_requirement`, `ad_update_model_memory_limit`, `ad_wf_troubleshoot_memory_limit` |
+| Model / lifecycle | `ad_get_model_snapshots`, `ad_revert_model_snapshot`, `ad_open_job`, `ad_create_job` |
+| CCS | `ad_ts_ccs_diagnostics` |
+| Calendars | `ad_get_calendar_events`, `ad_create_calendar_event` |
+
+Full parameter tables, ES|QL templates, and REST step lists:
+[references/troubleshoot-anomaly-tool-reference.md](references/troubleshoot-anomaly-tool-reference.md).
+
+### Rules
+
+1. **`ad_validate_ml_tool_permissions` first** — missing privileges produce misleading empty results.
+2. **Fix memory before `query_delay`** — `hard_limit` corrupts state; `query_delay` fixes on a memory-limited job are
+ wasted.
+3. **Stop the datafeed before updating it.** Updating a running datafeed is rejected.
+4. **Close the job before updating memory limit.** Sequence above.
+5. **Prefer workflow tools (`ad_wf_*`) over manually chaining diagnostics** for complex decisions.
+6. **`ad_preview_datafeed_with_latency` before starting** — confirm the datafeed returns data after config changes.
+
+---
+
+## Mode: Manage — Create / configure jobs
+
+**When:** "set up a job", "create an ML detector", "monitor X over time", "detect rare/unusual/anomalous values".
+
+### 4-step workflow
+
+```text
+PUT _ml/anomaly_detectors/ # 1. Define job (ad_create_job)
+PUT _ml/datafeeds/datafeed- # 2. Define datafeed (ad_create_datafeed)
+POST _ml/anomaly_detectors//_open # 3a. Open job (ad_open_job)
+POST _ml/datafeeds/datafeed-/_start # 3b. Start datafeed (ad_manage_datafeed action=_start)
+GET _ml/anomaly_detectors//results/records # 4. Read results
+```
+
+### Process
+
+1. **Build configs.** Parse the user request into job + datafeed JSON with no null fields.
+2. **Apply smart defaults:**
+
+ | Field | Default | Override when |
+ | ---------------- | --------------------------------------- | ------------------------------------------------- |
+ | `bucket_span` | `"15m"` | User specifies a different span |
+ | `time_field` | `"@timestamp"` | User names a different timestamp field |
+ | `index` | `"logs-*"` | User specifies an index or pattern |
+ | `datafeed_query` | `{"match_all": {}}` | User mentions filters, processes, or time windows |
+ | `influencers` | by/over/partition fields from detectors | User adds extra influencer fields |
+ | `job_id` | Generated from user description | User provides an explicit ID |
+ | `query_delay` | `"60s"` | P95 ingest latency is higher |
+
+3. **Choose detector function** from user intent — full table in
+ [references/anomaly-detection-functions.md](references/anomaly-detection-functions.md):
+ - "high CPU" / "unusually large" → `high_mean` or `high_sum`
+ - "rare logins" / "unusual values" → `rare` (variants below)
+ - "too many requests" / "spike in count" → `high_count`
+
+ `rare` variants:
+ - Infrequent globally → `rare by_field_name: X`
+ - Infrequent vs peers → `rare by_field_name: X over_field_name: Y`
+ - Infrequent per segment → `rare by_field_name: X partition_field_name: Y`
+ - Infrequent per segment vs peers → `rare by_field_name: X over_field_name: Y partition_field_name: Z`
+
+4. **Validate.** `platform.core.get_index_mapping` on the target index to verify field existence/types →
+ `ad_validate_job_spec`. If errors, fix and re-validate (max 3 attempts).
+
+5. **Present and confirm.** Show the **complete** job + datafeed bodies formatted as the exact API calls. Ask for
+ approval **once**. If feedback, incorporate and re-present (up to 3 rounds).
+
+6. **Deploy.** After confirmation: `ad_create_job` → `ad_create_datafeed` → `ad_open_job` → `ad_manage_datafeed`
+ (`action=_start`). Report final `job_id` and `datafeed_id`.
+
+For **batch analysis on historical data**, pass `start` and `end` to the datafeed start call.
+
+> Worked examples (rare-username, DNS exfil, large-downloads) with full JSON bodies and datafeed filters:
+> [references/job-creation-recipes.md](references/job-creation-recipes.md).
+
+### Rules
+
+1. **Create job before datafeed.** Datafeed references job by ID.
+2. **Open job before starting datafeed.** Start on a closed job is rejected.
+3. **`query_delay` = P95 ingest latency + buffer** (60s–120s safe default).
+4. **Forecasts require non-population jobs** — `over_field_name` jobs cannot be forecasted; warn before attempting.
+5. **`by_field_name` vs `over_field_name`:** `by` compares entity to its own history; `over` compares to peer group in
+ the same bucket. `partition_field_name` = fully independent sub-model with its own normalization.
+6. **`bucket_span` matches detection granularity** — 15m for high-frequency, 1h for operational metrics, 1d for daily
+ patterns. Larger smooths short spikes; smaller increases noise.
+
+---
+
+## Registration (Kibana Agent Builder)
+
+Requires Node.js 18+. Defaults to `elastic`/`changeme` when no credentials are supplied.
+
+```bash
+cd skills/kibana/kibana-anomaly-detection
+
+# tools → workflows → skills
+node scripts/kibana-agent-builder.mjs all register --kibana-url http://localhost:5601
+
+# HTTPS with self-signed cert
+node scripts/kibana-agent-builder.mjs all register --kibana-url https://localhost:5601 --insecure
+```
+
+`all register` runs `tools register`, then `workflows register`, then `skills register`. Kibana allows **at most five**
+`tool_ids` per skill; the script fills them by scanning `SKILL.md` for tool mentions (in document order), then appends
+ids from `references/kibana/tools/esql/*.json` until the cap (workflow-only tools omitted by default). If you run
+`skills register` alone, run `tools register` first so those ids exist.
+
+Workflow tool exclusions and prefixes live in `scripts/agent_builder_constants.json`.
+
+**MCP API key permissions:**
+
+- Kibana: `read_onechat`, `space_read`
+- Index: `read`, `view_index_metadata` on `.ml-anomalies-*`, `.ml-annotations-*`, `.ml-notifications-*`, `.ml-config`
+- For source evidence: `read` on source data indices
+
+---
+
+## Tool inventory
+
+ES|QL tool specs live under `references/kibana/tools/esql/*.json`; workflow definitions under
+`references/kibana/workflows/*.yaml`. Each Mode section above lists the tools it uses. Full surface:
+[references/tools.md](references/tools.md) (ES|QL) and [references/workflow-tools.md](references/workflow-tools.md)
+(workflows).
+
+### Key system indices
+
+| Index | Relevant content |
+| --------------------- | ----------------------------------------------------------------------------------------------------------------------------- |
+| `.ml-anomalies-*` | `record`, `bucket`, `influencer`, `model_plot`, `model_forecast`, `model_snapshot`, `category_definition`, `model_size_stats` |
+| `.ml-config` | job/datafeed documents (visible even for never-run jobs) |
+| `.ml-annotations-*` | delayed data (`event == "delayed_data"`) |
+| `.ml-notifications-*` | job messages (`level`: info/warning/error) |
+
+---
+
+## Examples
+
+**RCA:** "Something caused a spike in our error rate at 2pm — what broke?" → Investigate → `ad_get_available_metadata` →
+`ad_query_anomaly_timeline` → `ad_rca_cross_job_entity_match` → `ad_rca_multi_job_entities` → RCA report.
+
+**Score drop:** "My anomaly score went from 90 to 55 — did the model change?" → Explain → `ad_rca_score_reassessment`
+for drift → explain renormalization if `score_drift` is large.
+
+**Memory limit:** "Job status shows `hard_limit` and results look wrong." → Troubleshoot → `ad_ts_model_memory_health` →
+`ad_wf_ts_field_cardinality` → `ad_estimate_memory_requirement` → `ad_update_model_memory_limit` (lifecycle: stop
+datafeed → close → update → open → start).
+
+**New job:** "Detect unusual error rates per host on nginx access logs." → Manage → `high_count` detector with
+`by_field_name: "host.keyword"` → validate → present → deploy.
+
+**Multi-mode:** "We had an incident last night, scores were high but now low — is the job healthy?" → Investigate the
+incident → Explain the score drift → Troubleshoot if `hard_limit` or delayed data is suspected.
+
+---
+
+## Guidelines
+
+1. **Pick a mode first.** Don't blend RCA logic with score-explanation logic in one response.
+2. **`ad_validate_ml_tool_permissions` first** on empty results — privileges are the most common false-negative cause.
+3. **Score bands are absolute thresholds**: `>75` critical, `50–75` warning, `25–50` minor, `<25` informational.
+4. **Multi-job entities are prime suspects.** Use `min_job_count=2` in `ad_rca_multi_job_entities`.
+5. **Show `initial_record_score` alongside `record_score`** — the gap tells the renormalization story.
+6. **Fix memory before `query_delay`.** `hard_limit` invalidates downstream diagnostics.
+7. **Stop datafeed → close job → update config → open job → start datafeed** for any config change to memory or query
+ delay.
+8. **Confirm RCAs with `ad_rca_source_evidence`.** Raw source documents are ground truth.
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/package-lock.json b/plugins/kibana/skills/kibana-anomaly-detection/package-lock.json
new file mode 100644
index 0000000..31dd026
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/package-lock.json
@@ -0,0 +1,12 @@
+{
+ "name": "kibana-anomaly-detection-skills",
+ "version": "0.1.0",
+ "lockfileVersion": 3,
+ "requires": true,
+ "packages": {
+ "": {
+ "name": "kibana-anomaly-detection-skills",
+ "version": "0.1.0"
+ }
+ }
+}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/package.json b/plugins/kibana/skills/kibana-anomaly-detection/package.json
new file mode 100644
index 0000000..afab6e9
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/package.json
@@ -0,0 +1,13 @@
+{
+ "name": "kibana-anomaly-detection-skills",
+ "version": "0.1.0",
+ "private": true,
+ "type": "module",
+ "scripts": {
+ "tools:register": "node scripts/kibana-agent-builder.mjs tools register",
+ "workflows:register": "node scripts/kibana-agent-builder.mjs workflows register",
+ "skills:register": "node scripts/kibana-agent-builder.mjs skills register",
+ "all:register": "node scripts/kibana-agent-builder.mjs all register"
+ },
+ "dependencies": {}
+}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/README.md b/plugins/kibana/skills/kibana-anomaly-detection/references/README.md
new file mode 100644
index 0000000..6fb5277
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/README.md
@@ -0,0 +1,114 @@
+# kibana-anomaly-detection
+
+Claude Code plugin for Elastic ML anomaly detection — investigation, score explanation, and job operations via Kibana
+Agent Builder MCP tools.
+
+## What's included
+
+A single Agent Skill (`SKILL.md` at the package root, frontmatter `name: kibana-anomaly-detection`) covering four modes:
+**Investigate**, **Explain**, **Troubleshoot**, **Manage**. Domain-specific framing for security and observability lives
+under `references/`.
+
+**Kibana version:** Registration scripts are written for **Kibana 9.4+** (Agent Builder + Workflows). They use PUT for
+idempotent updates when the API supports it and fall back to DELETE + POST on stacks without tool/skill PUT so repeated
+`all register` runs stay safe.
+
+**Kibana Agent Builder assets** (under `references/kibana/`):
+
+- **24** ES|QL tool specs (`references/kibana/tools/esql/*.json`)
+- **23** workflow YAML files (`references/kibana/workflows/*.yaml`)
+- `scripts/kibana-agent-builder.mjs` — registers tools, workflows, and skills (Node.js 18+)
+
+## Reference pages
+
+| File | Purpose |
+| ---------------------------------------------------------------------------------------- | ------------------------------------------------------------- |
+| [score-reference.md](score-reference.md) | Score field definitions, bands, renormalization, explanation |
+| [anomaly-detection-functions.md](anomaly-detection-functions.md) | Detector function selection guide |
+| [investigate-anomaly-esql-tools.md](investigate-anomaly-esql-tools.md) | Full ES\|QL templates for the Investigate mode |
+| [troubleshoot-anomaly-tool-reference.md](troubleshoot-anomaly-tool-reference.md) | ES\|QL + workflow tool detail for the Troubleshoot mode |
+| [tools.md](tools.md) / [workflow-tools.md](workflow-tools.md) | Complete tool surface (ES\|QL and workflow) |
+| [protocols/investigation.md](protocols/investigation.md) | 14-step RCA protocol |
+| [worked-example.md](worked-example.md) | End-to-end investigation walkthrough |
+| [job-creation-recipes.md](job-creation-recipes.md) | `rare`, `high_mean`, `high_sum` job + datafeed recipes |
+| [permissions-matrix.md](permissions-matrix.md) | Privileges by tool category |
+| [anomaly-detection-openapi-spec-discover.md](anomaly-detection-openapi-spec-discover.md) | Discover ML REST endpoints via `.kibana_ai_openapi_spec_*` |
+| [security-anomaly-expert.md](security-anomaly-expert.md) | Threat-first framing (MITRE mappings, attack-chain protocol) |
+| [observability-anomaly-expert.md](observability-anomaly-expert.md) | SRE / reliability framing (degradation, capacity, regression) |
+
+## Using this package from agent-skills-sandbox
+
+This directory is part of the **agent-skills-sandbox** monorepo. There is **no** `install.sh` here — consume the skill
+by pointing your Agent Skills configuration at `skills/kibana/kibana-anomaly-detection/` (single `SKILL.md` at the
+package root). Restart your agent runtime after adding paths.
+
+## Kibana Agent Builder setup
+
+Requires Node.js 18+. Defaults to `elastic`/`changeme` when no credentials are supplied.
+
+```bash
+cd skills/kibana/kibana-anomaly-detection
+
+# Local Kibana — credentials default to elastic/changeme (tools → workflows → skills)
+node scripts/kibana-agent-builder.mjs all register --kibana-url http://localhost:5601
+
+# HTTPS with self-signed cert (common for local deployments)
+node scripts/kibana-agent-builder.mjs all register --kibana-url https://localhost:5601 --insecure
+```
+
+Environment-only flow (recommended):
+
+```bash
+cd skills/kibana/kibana-anomaly-detection
+export KIBANA_URL=https://localhost:5601
+export KIBANA_INSECURE=true
+node scripts/kibana-agent-builder.mjs all register
+```
+
+Elastic Cloud (API key):
+
+```bash
+cd skills/kibana/kibana-anomaly-detection
+export KIBANA_CLOUD_ID=""
+export KIBANA_API_KEY=""
+node scripts/kibana-agent-builder.mjs all register
+```
+
+`all register` runs `tools register`, `workflows register`, and `skills register`. The skill is posted with at most
+**five** `tool_ids` (Kibana limit): tools detected from `SKILL.md` text first, then supplemental ids from the
+registration script until the cap. Workflow-backed tools require Elastic Workflows (preview) where applicable;
+exclusions live in **`scripts/agent_builder_constants.json`**.
+
+**MCP API key permissions required:**
+
+- Kibana: `read_onechat`, `space_read`
+- Index: `read`, `view_index_metadata` on `.ml-anomalies-*`, `.ml-annotations-*`, `.ml-notifications-*`, `.ml-config`
+- For source evidence: `read` on source data indices
+
+## Layout (this repo)
+
+```text
+kibana-anomaly-detection/
+├── SKILL.md # Single skill: Investigate / Explain / Troubleshoot / Manage modes
+├── package.json # Dev deps for the registration script
+├── scripts/
+│ ├── kibana-agent-builder.mjs
+│ └── agent_builder_constants.json
+└── references/
+ ├── README.md
+ ├── score-reference.md
+ ├── anomaly-detection-functions.md
+ ├── investigate-anomaly-esql-tools.md
+ ├── troubleshoot-anomaly-tool-reference.md
+ ├── job-creation-recipes.md
+ ├── worked-example.md
+ ├── permissions-matrix.md
+ ├── tools.md
+ ├── workflow-tools.md
+ ├── security-anomaly-expert.md # threat-first framing
+ ├── observability-anomaly-expert.md # SRE / reliability framing
+ ├── protocols/investigation.md
+ └── kibana/
+ ├── tools/esql/
+ └── workflows/
+```
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/anomaly-detection-functions.md b/plugins/kibana/skills/kibana-anomaly-detection/references/anomaly-detection-functions.md
new file mode 100644
index 0000000..c564ddb
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/anomaly-detection-functions.md
@@ -0,0 +1,194 @@
+# Elastic ML Anomaly Detection Functions
+
+Functions marked with `*` support **high/low one-sided variants** (e.g., `high_count`, `low_mean`) to detect anomalies
+in only one direction.
+
+---
+
+## Count Functions
+
+Analyze the **occurrence rate** of events or documents over time.
+
+| Function | Description |
+| ------------------- | --------------------------------------------------------------------- |
+| `count` \* | Number of documents in a bucket |
+| `non_zero_count` \* | Like `count` but ignores zero-count buckets — use for **sparse data** |
+| `distinct_count` \* | Cardinality (uniqueness) of values for a specific field |
+
+### `count` vs `non_zero_count`
+
+| Scenario | Use |
+| ------------------------------------------------------------- | ----------------------------------------------------- |
+| Events arrive in every bucket (e.g., web traffic, heartbeats) | `count` — zero buckets are meaningful (outage signal) |
+| Events arrive intermittently (e.g., batch jobs, error logs) | `non_zero_count` — zeros are expected, not anomalous |
+| Detecting a **drop to zero** as an outage | `count` with `low_count` variant |
+| Detecting **bursts** in normally sparse traffic | `non_zero_count` with `high_non_zero_count` variant |
+
+**Example:** An intrusion detection log index gets events only when triggered. Using `count` means the model learns that
+zero-event buckets are normal overnight — making it impossible to distinguish a genuine quiet night from a monitoring
+gap. Use `non_zero_count` so the model only learns from buckets that had events, and flags when event volumes spike
+unexpectedly.
+
+### `distinct_count`
+
+Counts the number of unique values for a field within each bucket. Useful for detecting credential stuffing (unusually
+high distinct usernames), data exfiltration (high distinct destination IPs), or DGA activity (high distinct DNS query
+names).
+
+---
+
+## Metric Functions
+
+Operate on **numerical fields** within the data.
+
+| Function | Description |
+| ------------------------- | -------------------------------------------------------------- |
+| `min` / `max` | Minimum or maximum value in a bucket |
+| `mean` \* | Average value |
+| `median` \* | Median value |
+| `sum` / `non_null_sum` \* | Total sum of a field; `non_null_sum` is for **sparse data** |
+| `varp` \* | Variance / volatility of a metric |
+| `metric` | Shorthand that applies `min`, `max`, and `mean` simultaneously |
+
+### Choosing the right metric function
+
+| Goal | Function |
+| ---------------------------------------------- | ---------------------------------- |
+| Detect **average** latency spike | `mean` or `high_mean` |
+| Detect **worst-case** latency (tail) | `max` |
+| Detect **sustained volume** drop | `low_sum` |
+| Detect **unusual volatility** (erratic metric) | `high_varp` |
+| Detect both high and low deviations | `mean` (bidirectional, default) |
+| Detect only spikes, not drops | `high_mean` |
+| Monitor noisy metrics with outliers | `median` (more robust than `mean`) |
+
+### `sum` vs `non_null_sum`
+
+Use `non_null_sum` when the field is frequently absent from documents (sparse). Like `non_zero_count`, it skips empty
+buckets so the model learns only from active periods.
+
+### `metric` shorthand
+
+Creates three detectors in one: `min`, `max`, and `mean` on the same field. Convenient for a quick initial setup, but
+produces three anomaly records per detection event. Prefer explicit functions once you know which direction matters.
+
+---
+
+## Advanced & Specialized Functions
+
+Handle complex analysis types such as rarity or geographic data.
+
+| Function | Description |
+| ----------------- | -------------------------------------------------------------------------------- |
+| `rare` | Identifies values that occur at **low frequency** compared to the dataset |
+| `freq_rare` | Finds population members that **cause rare values to occur frequently** |
+| `info_content` \* | Entropy of text strings — useful for detecting **encrypted/obfuscated commands** |
+| `lat_long` | Detects unusual **geographic locations** from latitude/longitude coordinates |
+| `time_of_day` | Detects behavioral changes relative to **time of day** |
+| `time_of_week` | Detects behavioral changes relative to **day of week** |
+
+---
+
+### `rare` vs `freq_rare`
+
+These two are often confused but answer different questions:
+
+| Function | Question answered | Requires `over_field` |
+| ----------- | ----------------------------------------------------------- | --------------------- |
+| `rare` | "Which values of the `by_field` are unusual globally?" | No |
+| `freq_rare` | "Which entities (over_field) frequently cause rare values?" | Yes |
+
+**`rare` example:** Detect rare `process.name` values across all hosts. If `svchost.exe` with unusual arguments appears
+only once in 30 days, `rare` flags it. The focus is the rarity of the _value_.
+
+**`freq_rare` example:** Detect which _users_ (`over_field`) frequently trigger rare process executions (`by_field`).
+Most users run rare processes occasionally, but a user running rare processes consistently is a lateral movement signal.
+The focus is the entity's behavior pattern.
+
+**Practical guidance:**
+
+- Use `rare` for hunting unknown unknowns — values that shouldn't exist at all.
+- Use `freq_rare` for insider threat and lateral movement scenarios — who is repeatedly doing unusual things.
+- `rare` generates high false-positive rates in noisy environments; use custom rules to suppress known-good rare values.
+- `freq_rare` requires an `over_field` (population) — without it, use `rare`.
+
+### Types of rare analysis
+
+Translate business goals to the correct `rare` detector configuration:
+
+| Goal | Example | Detector config |
+| -------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------- |
+| Find infrequent values for a field | Detect hosts occurring infrequently | `rare` `by_field_name: host` |
+| Find infrequent values compared to peers | Detect hosts visited by few users vs. other hosts; detect users visiting rare hosts | `rare` `by_field_name: host` `over_field_name: user` |
+| Find infrequent values segmented by another field | Per location, detect infrequently seen hosts | `rare` `by_field_name: host` `partition_field_name: location` |
+| Find infrequent values by another field compared to peers, segmented | Per location, detect hosts visited by few users vs. peers; detect users visiting rare hosts in a location | `rare` `by_field_name: host` `over_field_name: user` `partition_field_name: location` |
+
+---
+
+### `info_content`
+
+Measures the **Shannon entropy** of text strings. High entropy = high randomness = potentially encoded, encrypted, or
+machine-generated content.
+
+| Entropy range | Interpretation | Example |
+| ----------------- | ------------------------------------ | -------------------------------------------------- |
+| Low (predictable) | Normal human-readable text | `GET /api/users HTTP/1.1` |
+| Medium | Mixed structured/variable content | Log messages with variable IDs |
+| High (random) | Encoded, encrypted, or DGA-generated | `aGVsbG8gd29ybGQ=` (base64), `zxq7v9abc.com` (DGA) |
+
+**Use cases:**
+
+- **DNS query names** with `by_field_name: "dns.question.name"` — DGA malware generates high-entropy domain names (e.g.,
+ `xk9p2mnqabcdef.ru`).
+- **User-agent strings** — malware C2 frameworks often use randomized or encoded user agents.
+- **URL paths / query strings** — webshell commands embedded in request parameters.
+- **Command-line arguments** — base64-encoded PowerShell payloads.
+
+**`high_info_content`** (one-sided): only alerts on unusually high entropy, which is almost always the right choice for
+security use cases.
+
+---
+
+### `time_of_day` vs `time_of_week`
+
+Both detect _when_ something happens relative to established patterns, not _how much_.
+
+| Function | Granularity | Best for |
+| -------------- | ------------------ | -------------------------------------------------------- |
+| `time_of_day` | Hour/minute of day | Detecting off-hours access (3am login for a 9-5 user) |
+| `time_of_week` | Day of week | Detecting weekend/holiday activity (batch job on Sunday) |
+
+**`time_of_day` example:** A database admin who always connects between 08:00–18:00 on weekdays. `time_of_day` learns
+this pattern. A connection at 02:30 scores anomalous — even if the volume of activity is normal.
+
+**`time_of_week` example:** A data pipeline that runs Monday–Friday. `time_of_week` flags it running on Saturday.
+`time_of_day` would not catch this (the time of day, e.g., 08:00, may be normal).
+
+**Choosing between them:**
+
+- Use `time_of_day` when the anomaly is "wrong hour of the day."
+- Use `time_of_week` when the anomaly is "wrong day of the week."
+- Use both together (two detectors) for comprehensive temporal coverage.
+- Neither function cares about _count_ or _metric values_ — use `count`/`mean` detectors in the same job for
+ volume-based detection.
+
+---
+
+## One-Sided Variants
+
+Functions marked with `*` support `high_` and `low_` prefixes:
+
+| Variant | Detects |
+| ------------------------ | ------------------------------------------------------ |
+| `high_` | Only values significantly **above** the expected range |
+| `low_` | Only values significantly **below** the expected range |
+| `` (no prefix) | Both directions (bidirectional) |
+
+**When to use one-sided:**
+
+- `high_mean(response_time)` — only alert on latency spikes, not drops (a faster response is never a problem).
+- `low_count(login_events)` — only alert on unusually low login volume (could indicate authentication system failure).
+- `high_distinct_count(destination.ip)` — only alert on abnormally high unique destination IPs (exfiltration signal).
+
+Bidirectional variants (`mean`, `count`) generate alerts for both directions, which can produce noise when only one
+direction is operationally relevant.
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/anomaly-detection-openapi-spec-discover.md b/plugins/kibana/skills/kibana-anomaly-detection/references/anomaly-detection-openapi-spec-discover.md
new file mode 100644
index 0000000..4e89588
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/anomaly-detection-openapi-spec-discover.md
@@ -0,0 +1,74 @@
+## ML API Spec Discovery
+
+The `.kibana_ai_openapi_spec_elasticsearch` index, if present, contains one document per API endpoint. Document shape:
+
+```json
+{
+ "description": "Forecasts are not supported for jobs that perform population analysis; an\nerror occurs if you try to create a forecast for a job that has an\n`over_field_name` in its configuration. Forecasts predict future behavior\nbased on historical data.\n\n## Required authorization\n\n* Cluster privileges: `manage_ml`\n",
+ "endpoint": "POST /_ml/anomaly_detectors/{job_id}/_forecast",
+ "method": "post",
+ "operationId": "ml-forecast",
+ "path": "/_ml/anomaly_detectors/{job_id}/_forecast",
+ "path.keyword": "/_ml/anomaly_detectors/{job_id}/_forecast",
+ "summary": "Predict future behavior of a time series",
+ "tags": "ml anomaly"
+}
+```
+
+Use this to make informed API requests.
+
+`summary` and `description` are **semantic text fields** — use `MATCH()` for natural-language lookup.
+
+### Step 1 — Confirm index exists
+
+```text
+platform.core.list_indices → check for .kibana_ai_openapi_spec_elasticsearch
+```
+
+or use ES|QL query
+
+```esql
+FROM .kibana_ai_openapi_spec_elasticsearch
+```
+
+If index missing or error, fall back to researching the official documentation:
+[Elastic ML API docs](https://www.elastic.co/docs/api/doc/elasticsearch/group/endpoint-ml) and
+[Elastic ML Anomaly API docs](https://www.elastic.co/docs/api/doc/elasticsearch/group/endpoint-ml-anomaly).
+
+### Step 2 — List all ML endpoints
+
+```esql
+FROM .kibana_ai_openapi_spec_elasticsearch
+| WHERE tags == "ml"
+| SORT endpoint ASC
+| LIMIT 100
+```
+
+### Step 3 — Look up a specific endpoint by path
+
+```esql
+FROM .kibana_ai_openapi_spec_elasticsearch
+| WHERE path LIKE "/_ml/calendars*"
+| SORT endpoint ASC
+```
+
+### Step 4 — Semantic search when you know the intent, not the path
+
+```esql
+FROM .kibana_ai_openapi_spec_elasticsearch
+| WHERE tags == "ml" AND MATCH(summary, "model memory limit")
+| LIMIT 10
+```
+
+```esql
+FROM .kibana_ai_openapi_spec_elasticsearch
+| WHERE tags == "ml" AND MATCH(description, "revert snapshot")
+| KEEP method, endpoint, summary, description
+| LIMIT 5
+```
+
+**When to use this:**
+
+- A workflow tool returns 400/404 and you suspect a wrong path or missing required field
+- You want to discover optional parameters not covered by the current tool definition
+- You're adding a new `ad_*` workflow tool and need the exact endpoint and request body schema
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/investigate-anomaly-esql-tools.md b/plugins/kibana/skills/kibana-anomaly-detection/references/investigate-anomaly-esql-tools.md
new file mode 100644
index 0000000..aa93499
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/investigate-anomaly-esql-tools.md
@@ -0,0 +1,371 @@
+# Investigate mode — ES|QL tool reference
+
+Supporting detail for the **Investigate** mode of the parent [SKILL.md](../SKILL.md). Use these templates with Kibana
+Agent Builder ES|QL tools.
+
+## ES|QL Tools (15)
+
+### Discovery & Metadata
+
+### `ad_get_available_metadata`
+
+Discover all jobs and their configured metadata. **Call first** when jobs are unknown.
+
+```esql
+FROM .ml-config
+| WHERE job_type == "anomaly_detector"
+| STATS job_count = COUNT(*),
+ job_ids = VALUES(job_id),
+ functions = VALUES(`analysis_config.detectors.function`),
+ fields = VALUES(`analysis_config.detectors.field_name`),
+ by_fields = VALUES(`analysis_config.detectors.by_field_name`),
+ over_fields = VALUES(`analysis_config.detectors.over_field_name`),
+ partition_fields = VALUES(`analysis_config.detectors.partition_field_name`),
+ influencers = VALUES(`analysis_config.influencers`),
+ bucket_spans = VALUES(`analysis_config.bucket_span`)
+```
+
+_No parameters._
+
+---
+
+### `ad_get_jobs`
+
+List all jobs with full config: bucket_span, detector functions, field names, memory limit.
+
+```esql
+FROM .ml-config
+| WHERE job_type == "anomaly_detector"
+| KEEP job_id, `analysis_config.bucket_span`, `analysis_config.detectors.function`,
+ `analysis_config.detectors.field_name`, `analysis_config.detectors.partition_field_name`,
+ `analysis_config.detectors.by_field_name`, `analysis_config.detectors.over_field_name`,
+ `analysis_config.influencers`, `analysis_limits.model_memory_limit`, groups, description
+| SORT job_id ASC | LIMIT 100
+```
+
+_No parameters._
+
+---
+
+### `ad_discover_related_jobs`
+
+Find jobs sharing the same entity field name (partition/by/over). Call early even when job names differ completely.
+
+| Parameter | Type | Description |
+| --------- | ---- | ------------------- |
+| `job_id` | text | The job of interest |
+
+```esql
+FROM .ml-config
+| WHERE job_type == "anomaly_detector"
+| EVAL entity_field = COALESCE(`analysis_config.detectors.partition_field_name`,
+ `analysis_config.detectors.by_field_name`,
+ `analysis_config.detectors.over_field_name`)
+| MV_EXPAND entity_field
+| WHERE entity_field IS NOT NULL
+| STATS job_count = COUNT_DISTINCT(job_id), jobs = VALUES(job_id),
+ influencers = VALUES(`analysis_config.influencers`) BY entity_field
+| WHERE MV_CONTAINS(jobs, ?job_id)
+| SORT job_count DESC | LIMIT 50
+```
+
+---
+
+### Anomaly Records & Influencers
+
+### `ad_query_anomaly_records`
+
+Primary anomaly search. Cross-job with `*` or single-job with exact ID.
+
+| Parameter | Type | Description |
+| ---------------- | ------ | ------------------------------------------------------------------- |
+| `job_id_pattern` | text | LIKE wildcards: `*` all, `rcaeval-*` group, exact ID for drill-down |
+| `min_score` | double | 50 significant · 25 broad · 75 critical |
+| `start_time` | text | ISO 8601 |
+| `end_time` | text | ISO 8601 |
+
+```esql
+FROM .ml-anomalies-*
+| WHERE result_type == "record"
+ AND job_id LIKE ?job_id_pattern
+ AND record_score >= ?min_score
+ AND timestamp >= ?start_time AND timestamp <= ?end_time
+| SORT record_score DESC | LIMIT 50
+| KEEP job_id, timestamp, record_score, function, field_name, actual, typical,
+ by_field_name, by_field_value, over_field_name, over_field_value,
+ partition_field_name, partition_field_value,
+ multi_bucket_impact, initial_record_score, detector_index, probability
+```
+
+---
+
+### `ad_query_anomaly_timeline`
+
+Cross-job bucket timeline. Composite scores reveal coordinated events (5 jobs × 30 = composite 150).
+
+| Parameter | Type | Description |
+| ---------------- | ------ | ----------------------------------- |
+| `job_id_pattern` | text | LIKE wildcards |
+| `min_score` | double | 25 for signal boosting, 50 standard |
+| `start_time` | text | ISO 8601 |
+| `end_time` | text | ISO 8601 |
+
+```esql
+FROM .ml-anomalies-*
+| WHERE result_type == "bucket"
+ AND job_id LIKE ?job_id_pattern
+ AND anomaly_score >= ?min_score
+ AND timestamp >= ?start_time AND timestamp <= ?end_time
+| STATS max_score = MAX(anomaly_score), job_count = COUNT_DISTINCT(job_id),
+ jobs = VALUES(job_id), composite_score = SUM(anomaly_score),
+ avg_score = AVG(anomaly_score) BY timestamp
+| SORT timestamp
+```
+
+---
+
+### `ad_query_influencers`
+
+Most anomalous entities. `job_count > 1` filter = cross-job shared influencers = strongest RCA signal.
+
+| Parameter | Type | Description |
+| ---------------- | ------ | ---------------------------------- |
+| `job_id_pattern` | text | LIKE wildcards |
+| `min_score` | double | 25 for shared influencer discovery |
+| `start_time` | text | ISO 8601 |
+| `end_time` | text | ISO 8601 |
+
+```esql
+FROM .ml-anomalies-*
+| WHERE result_type == "influencer"
+ AND job_id LIKE ?job_id_pattern
+ AND influencer_score >= ?min_score
+ AND timestamp >= ?start_time AND timestamp <= ?end_time
+| STATS total_score = SUM(influencer_score), job_count = COUNT_DISTINCT(job_id),
+ jobs = VALUES(job_id), max_score = MAX(influencer_score)
+ BY influencer_field_name, influencer_field_value
+| SORT total_score DESC | LIMIT 30
+```
+
+---
+
+### RCA Tools
+
+### `ad_rca_multi_job_entities`
+
+**Strongest root cause signal.** Entities anomalous in 2+ jobs simultaneously. Resource faults → multi-job; network
+faults → single-job.
+
+| Parameter | Type | Description |
+| --------------- | ------ | ------------------------- |
+| `min_score` | double | 25 broad · 50 significant |
+| `min_job_count` | double | Use 2 for cross-job RCA |
+| `start_time` | text | ISO 8601 |
+| `end_time` | text | ISO 8601 |
+
+```esql
+FROM .ml-anomalies-*
+| WHERE result_type == "record"
+ AND record_score >= ?min_score
+ AND timestamp >= ?start_time AND timestamp <= ?end_time
+| STATS job_count = COUNT_DISTINCT(job_id), jobs = VALUES(job_id),
+ max_score = MAX(record_score), total_records = COUNT(*),
+ functions = VALUES(function), fields = VALUES(field_name)
+ BY partition_field_value
+| WHERE job_count >= ?min_job_count
+| SORT job_count DESC, max_score DESC | LIMIT 20
+```
+
+---
+
+### `ad_rca_cross_job_entity_match`
+
+From an alert's entity value, find ALL jobs where it's anomalous. Returns `first_anomaly` per job for chronology
+reconstruction.
+
+| Parameter | Type | Description |
+| -------------- | ------ | ------------------------------------------ |
+| `entity_value` | text | From alert's partition/by/over field value |
+| `min_score` | double | 10–25 for comprehensive search |
+| `start_time` | text | Wider than alert window (alert minus 6h) |
+| `end_time` | text | Alert plus 1–2h to catch delayed effects |
+
+```esql
+FROM .ml-anomalies-*
+| WHERE result_type == "record"
+ AND record_score >= ?min_score
+ AND timestamp >= ?start_time AND timestamp <= ?end_time
+ AND (partition_field_value == ?entity_value
+ OR by_field_value == ?entity_value
+ OR over_field_value == ?entity_value)
+| STATS max_score = MAX(record_score), anomaly_count = COUNT(*),
+ functions = VALUES(function), fields = VALUES(field_name),
+ first_anomaly = MIN(timestamp), last_anomaly = MAX(timestamp)
+ BY job_id
+| SORT max_score DESC
+```
+
+---
+
+### `ad_rca_detector_fingerprint`
+
+Incident fingerprint — which system aspects are anomalous (CPU? Latency? Error rate?).
+
+| Parameter | Type | Description |
+| ---------------- | ------ | -------------------- |
+| `job_id_pattern` | text | LIKE wildcards |
+| `min_score` | double | Minimum record_score |
+| `start_time` | text | ISO 8601 |
+| `end_time` | text | ISO 8601 |
+
+```esql
+FROM .ml-anomalies-*
+| WHERE result_type == "record"
+ AND job_id LIKE ?job_id_pattern
+ AND record_score >= ?min_score
+ AND timestamp >= ?start_time AND timestamp <= ?end_time
+| STATS count = COUNT(*), max_score = MAX(record_score), avg_score = AVG(record_score)
+ BY job_id, function, field_name, detector_index
+| SORT max_score DESC
+```
+
+---
+
+### `ad_rca_correlation`
+
+Temporally ordered anomalies for cascade analysis. Earliest anomaly for an entity → root cause direction.
+
+| Parameter | Type | Description |
+| ---------------- | ------ | --------------------------------------- |
+| `job_id_pattern` | text | LIKE wildcards (scope to related group) |
+| `min_score` | double | Minimum record_score |
+| `start_time` | text | ISO 8601 |
+| `end_time` | text | ISO 8601 |
+
+```esql
+FROM .ml-anomalies-*
+| WHERE result_type == "record"
+ AND job_id LIKE ?job_id_pattern
+ AND record_score >= ?min_score
+ AND timestamp >= ?start_time AND timestamp <= ?end_time
+| SORT timestamp ASC
+| KEEP job_id, timestamp, record_score, function, field_name,
+ by_field_name, by_field_value, partition_field_name, partition_field_value,
+ over_field_name, over_field_value, multi_bucket_impact
+| LIMIT 100
+```
+
+---
+
+### `ad_rca_blast_radius`
+
+Scope of impact — how many partitions/jobs are affected by a specific anomalous value.
+
+| Parameter | Type | Description |
+| ----------------- | ------ | --------------------------------------- |
+| `anomalous_value` | text | e.g. `node-backdoor`, `payment-service` |
+| `min_score` | double | 10–25 for weak-signal aggregation |
+| `start_time` | text | ISO 8601 |
+| `end_time` | text | ISO 8601 |
+
+```esql
+FROM .ml-anomalies-*
+| WHERE result_type == "record"
+ AND record_score >= ?min_score
+ AND timestamp >= ?start_time AND timestamp <= ?end_time
+ AND (by_field_value == ?anomalous_value
+ OR over_field_value == ?anomalous_value
+ OR partition_field_value == ?anomalous_value)
+| STATS affected_partitions = COUNT_DISTINCT(partition_field_value),
+ affected_jobs = COUNT_DISTINCT(job_id), total_anomalies = COUNT(*),
+ max_score = MAX(record_score),
+ time_span_hours = DATE_DIFF("hour", MIN(timestamp), MAX(timestamp))
+ BY by_field_value
+| SORT affected_partitions DESC
+```
+
+---
+
+### `ad_rca_entity_profile`
+
+Complete anomaly dossier for a suspect entity across ALL jobs and field types.
+
+| Parameter | Type | Description |
+| -------------- | ---- | ----------------------------------- |
+| `entity_value` | text | e.g. `server-01`, `payment-service` |
+| `start_time` | text | ISO 8601 |
+| `end_time` | text | ISO 8601 |
+
+```esql
+FROM .ml-anomalies-*
+| WHERE result_type == "record"
+ AND timestamp >= ?start_time AND timestamp <= ?end_time
+ AND (by_field_value == ?entity_value
+ OR over_field_value == ?entity_value
+ OR partition_field_value == ?entity_value)
+| SORT timestamp ASC | LIMIT 100
+| KEEP job_id, timestamp, record_score, function, field_name,
+ by_field_name, by_field_value, over_field_name, over_field_value,
+ partition_field_name, partition_field_value, actual, typical, multi_bucket_impact
+```
+
+---
+
+### `ad_rca_source_evidence`
+
+Raw source documents from the original data index. Get source index from `ad_get_job_datafeed_config` first.
+
+| Parameter | Type | Description |
+| -------------- | ---- | ------------------------------------------------------------------------ |
+| `source_index` | text | LIKE pattern from datafeed config (e.g. `rcaeval-re1-ob`, `otel-flat-*`) |
+| `start_time` | text | ISO 8601 |
+| `end_time` | text | ISO 8601 |
+
+```esql
+FROM * METADATA _index
+| WHERE _index LIKE ?source_index
+ AND @timestamp >= ?start_time AND @timestamp <= ?end_time
+| SORT @timestamp DESC | LIMIT 50
+```
+
+---
+
+### Log Categorization (when `by_field_name == "mlcategory"`)
+
+### `ad_get_categories`
+
+Category definitions (terms, regex, examples) for jobs with `categorization_field_name`.
+
+| Parameter | Type | Description |
+| --------- | ---- | ---------------------------- |
+| `job_id` | text | The anomaly detection job ID |
+
+```esql
+FROM .ml-anomalies-*
+| WHERE result_type == "category_definition" AND job_id == ?job_id
+| SORT category_id ASC
+| KEEP job_id, category_id, terms, regex, max_matching_length, examples
+| LIMIT 100
+```
+
+---
+
+### `ad_search_log_category_examples`
+
+Raw log samples for two-window comparison (baseline vs anomaly window). Compare variable parts — IPs, hostnames, error
+codes — to find what changed.
+
+| Parameter | Type | Description |
+| -------------- | ---- | --------------------------------------- |
+| `source_index` | text | LIKE pattern from datafeed config |
+| `start_time` | text | ISO 8601 (baseline: 24h before anomaly) |
+| `end_time` | text | ISO 8601 |
+
+```esql
+FROM * METADATA _index
+| WHERE _index LIKE ?source_index
+ AND @timestamp >= ?start_time AND @timestamp <= ?end_time
+| SORT @timestamp DESC | LIMIT 50
+```
+
+---
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/job-creation-recipes.md b/plugins/kibana/skills/kibana-anomaly-detection/references/job-creation-recipes.md
new file mode 100644
index 0000000..573d46e
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/job-creation-recipes.md
@@ -0,0 +1,186 @@
+# Job creation recipes
+
+Worked examples for creating Elastic ML anomaly detection jobs and datafeeds via Agent Builder workflow tools.
+
+> For the high-level process, see the **Manage** section of the parent `SKILL.md`. For detector function selection, see
+> [anomaly-detection-functions.md](anomaly-detection-functions.md).
+
+---
+
+## API call sequence
+
+```text
+PUT _ml/anomaly_detectors/ # 1. Define job + detectors (ad_create_job)
+PUT _ml/datafeeds/datafeed- # 2. Define datafeed (source + query) (ad_create_datafeed)
+POST _ml/anomaly_detectors//_open # 3a. Open job (ad_open_job)
+POST _ml/datafeeds/datafeed-/_start # 3b. Start datafeed (ad_manage_datafeed action=_start)
+GET _ml/anomaly_detectors//results/records # 4. Read results
+```
+
+To stop: `ad_manage_datafeed` (`action=_stop`) → `POST _ml/anomaly_detectors//_close`.
+
+For **batch analysis on historical data**, pass `start` and `end` to the datafeed start call:
+
+```json
+POST _ml/datafeeds/datafeed-revenue_over_users_api/_start
+{ "start": "2024-01-01T00:00:00Z", "end": "2024-03-01T00:00:00Z" }
+```
+
+---
+
+## Key `analysis_config` fields
+
+| Field | Description |
+| ---------------------------------- | -------------------------------------------------------------------------------------------------------------------------- |
+| `bucket_span` | Analysis interval (e.g. `15m`, `1h`). Align with data granularity and detection window. |
+| `detectors[].function` | Analysis function (`high_sum`, `rare`, `mean`, etc). See [anomaly-detection-functions.md](anomaly-detection-functions.md). |
+| `detectors[].field_name` | Numeric field to analyze. |
+| `detectors[].over_field_name` | Population analysis — each entity compared to its peers in the same bucket. |
+| `detectors[].by_field_name` | Per-entity modeling — each entity compared to its own history. |
+| `detectors[].partition_field_name` | Fully independent sub-model per entity with its own score normalization. |
+| `influencers` | Fields to track as anomaly contributors (shown as `influencer_score`). |
+| `data_description.time_field` | Timestamp field for time series ordering (e.g. `@timestamp`, `order_date`). |
+
+## Key datafeed fields
+
+| Field | Description |
+| ------------- | --------------------------------------------------------------------------------------------------------------------------------- |
+| `indices` | Array of index patterns containing source data. |
+| `query` | Elasticsearch DSL to filter source documents. Defaults to `match_all`. |
+| `query_delay` | How far behind real time the datafeed queries. Set to P95 ingest latency + buffer (default `60s`–`120s`). Too low → missing docs. |
+| `scroll_size` | Documents fetched per scroll request. Default `1000`. |
+| `frequency` | How often the datafeed polls. Defaults to `min(query_delay, bucket_span / 2)`. |
+
+---
+
+## Recipe 1 — `rare` detector: rare usernames
+
+**User query:** "Create an ML job to detect rare usernames in login events across logs-\*"
+
+Verify `user.name` (keyword) and `@timestamp` (date) exist via `platform.core.get_index_mapping`. Validate with
+`ad_validate_job_spec`. Then `ad_create_job` → `ad_create_datafeed` → `ad_open_job` → `ad_manage_datafeed`
+(`action=_start`).
+
+**Job body:**
+
+```json
+{
+ "description": "Detects rare values of user.name during login activity",
+ "analysis_config": {
+ "bucket_span": "15m",
+ "detectors": [
+ { "function": "rare", "by_field_name": "user.name", "detector_description": "Rare user.name values" }
+ ],
+ "influencers": ["user.name", "source.ip"]
+ },
+ "data_description": { "time_field": "@timestamp" }
+}
+```
+
+**Datafeed body:**
+
+```json
+{ "job_id": "rare-login-usernames", "indices": ["logs-*"], "query": { "match_all": {} } }
+```
+
+---
+
+## Recipe 2 — `high_mean` detector: DNS exfiltration with datafeed filter
+
+**User query:** "detect hosts with unusually high DNS query volume per domain, only for external DNS traffic on port 53"
+
+Verify `dns.question.count` (numeric), `host.name` (keyword), `dns.question.name` (keyword), `@timestamp` (date).
+
+**Job body:**
+
+```json
+{
+ "description": "Detects unusually high DNS query volume per host, partitioned by queried domain",
+ "analysis_config": {
+ "bucket_span": "15m",
+ "detectors": [
+ {
+ "function": "high_mean",
+ "field_name": "dns.question.count",
+ "over_field_name": "host.name",
+ "partition_field_name": "dns.question.name",
+ "detector_description": "High DNS query volume per host per domain"
+ }
+ ],
+ "influencers": ["host.name", "dns.question.name", "source.ip"]
+ },
+ "data_description": { "time_field": "@timestamp" }
+}
+```
+
+**Datafeed body:**
+
+```json
+{
+ "job_id": "high-dns-query-volume-per-host",
+ "indices": ["logs-*"],
+ "query": {
+ "bool": { "filter": [{ "term": { "network.transport": "udp" } }, { "term": { "destination.port": 53 } }] }
+ }
+}
+```
+
+---
+
+## Recipe 3 — `high_sum` detector: large downloads with time range
+
+**User query:** "detect users downloading unusually large amounts of data for sshd and sftp processes in the last 30
+days"
+
+Verify `destination.bytes` (numeric), `user.name` (keyword), `process.name` (keyword), `@timestamp` (date).
+
+**Job body:**
+
+```json
+{
+ "description": "Detects unusually high total bytes downloaded per user for specific processes",
+ "analysis_config": {
+ "bucket_span": "1h",
+ "detectors": [
+ {
+ "function": "high_sum",
+ "field_name": "destination.bytes",
+ "by_field_name": "user.name",
+ "detector_description": "High total bytes downloaded per user"
+ }
+ ],
+ "influencers": ["user.name", "process.name", "source.ip"]
+ },
+ "data_description": { "time_field": "@timestamp" }
+}
+```
+
+**Datafeed body:**
+
+```json
+{
+ "job_id": "high-download-volume-per-user",
+ "indices": ["logs-*"],
+ "query": {
+ "bool": {
+ "filter": [
+ { "terms": { "process.name": ["sshd", "sftp"] } },
+ { "range": { "@timestamp": { "gte": "now-30d", "lte": "now" } } }
+ ]
+ }
+ }
+}
+```
+
+---
+
+## Retrieving results
+
+| Result type | Endpoint | Description |
+| ----------- | -------------------------------------------------------- | ------------------------------------------------------------------- |
+| Buckets | `GET _ml/anomaly_detectors//results/buckets` | Aggregate anomaly score per time bucket |
+| Records | `GET _ml/anomaly_detectors//results/records` | Individual anomaly records with `actual`, `typical`, `record_score` |
+| Influencers | `GET _ml/anomaly_detectors//results/influencers` | Entity contribution scores |
+| Forecast | `POST _ml/anomaly_detectors//_forecast` | Predict future values; specify `duration` (e.g. `"10d"`) |
+
+> Forecasts are **not supported** for population analysis jobs (`over_field_name` set).
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/agent/anomaly_detective.json b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/agent/anomaly_detective.json
new file mode 100644
index 0000000..28878e3
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/agent/anomaly_detective.json
@@ -0,0 +1,21 @@
+{
+ "tools": [
+ "ad_get_available_metadata",
+ "ad_get_jobs",
+ "ad_discover_related_jobs",
+ "ad_rca_cross_job_entity_match",
+ "ad_query_anomaly_records",
+ "ad_query_anomaly_timeline",
+ "ad_query_influencers",
+ "ad_rca_entity_profile",
+ "ad_rca_detector_fingerprint",
+ "ad_rca_correlation",
+ "ad_rca_blast_radius",
+ "ad_rca_multi_job_entities",
+ "ad_rca_source_evidence",
+ "ad_discover_jobs_by_datafeed_index",
+ "ad_get_job_datafeed_config",
+ "ad_get_categories",
+ "ad_search_log_category_examples"
+ ]
+}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/agent/anomaly_explainer.json b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/agent/anomaly_explainer.json
new file mode 100644
index 0000000..d96fc9f
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/agent/anomaly_explainer.json
@@ -0,0 +1,14 @@
+{
+ "tools": [
+ "ad_get_available_metadata",
+ "ad_get_jobs",
+ "ad_query_anomaly_records",
+ "ad_query_influencers",
+ "ad_rca_score_reassessment",
+ "ad_get_model_plot",
+ "ad_get_categories",
+ "ad_get_forecast_results",
+ "ad_wf_troubleshoot_anomaly_score",
+ "ad_get_job_datafeed_config"
+ ]
+}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/agent/anomaly_maintainer.json b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/agent/anomaly_maintainer.json
new file mode 100644
index 0000000..4b0ea19
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/agent/anomaly_maintainer.json
@@ -0,0 +1,28 @@
+{
+ "tools": [
+ "ad_get_available_metadata",
+ "ad_get_jobs",
+ "ad_get_job_messages",
+ "ad_get_model_snapshots",
+ "ad_ts_bucket_event_gaps",
+ "ad_ts_delayed_data_annotations",
+ "ad_ts_ingest_latency_estimate",
+ "ad_ts_model_memory_health",
+ "ad_wf_ts_field_cardinality",
+ "ad_get_job_datafeed_config",
+ "ad_manage_datafeed",
+ "ad_preview_datafeed_with_latency",
+ "ad_create_job",
+ "ad_revert_model_snapshot",
+ "ad_create_calendar_event",
+ "ad_get_calendar_events",
+ "ad_update_datafeed_query_delay",
+ "ad_update_delayed_data_check_config",
+ "ad_update_model_memory_limit",
+ "ad_estimate_memory_requirement",
+ "ad_validate_ml_tool_permissions",
+ "ad_ts_ccs_diagnostics",
+ "ad_wf_troubleshoot_query_delay",
+ "ad_wf_troubleshoot_memory_limit"
+ ]
+}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_discover_related_jobs.json b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_discover_related_jobs.json
new file mode 100644
index 0000000..698c9b6
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_discover_related_jobs.json
@@ -0,0 +1,14 @@
+{
+ "name": "ad_discover_related_jobs",
+ "description": "Given a job of interest, find all other jobs that share the same entity field name (partition, by, or over field) by reading job configs from .ml-config. Returns the shared field name, the full list of sibling jobs in that group, their configured influencer fields, and the total sibling count. A job appears in its primary entity field group (partition > by > over). Call this early in investigation to find sibling jobs even when job names and prefixes differ completely. Complement with ad_discover_jobs_by_datafeed_index for source index overlap.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-config | WHERE job_type == \"anomaly_detector\" | EVAL entity_field = COALESCE(`analysis_config.detectors.partition_field_name`, `analysis_config.detectors.by_field_name`, `analysis_config.detectors.over_field_name`) | MV_EXPAND entity_field | WHERE entity_field IS NOT NULL | STATS job_count = COUNT_DISTINCT(job_id), jobs = VALUES(job_id), influencers = VALUES(`analysis_config.influencers`) BY entity_field | WHERE MV_CONTAINS(jobs, ?job_id) | SORT job_count DESC | LIMIT 50"
+ },
+ "parameters": {
+ "job_id": {
+ "type": "string",
+ "description": "The job ID of interest. Returns all other jobs that share the same entity split field (partition, by, or over field) as this job."
+ }
+ }
+}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_available_metadata.json b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_available_metadata.json
new file mode 100644
index 0000000..20362e8
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_available_metadata.json
@@ -0,0 +1,9 @@
+{
+ "name": "ad_get_available_metadata",
+ "description": "Discover all anomaly detection jobs and their configured metadata: job IDs, detector functions, monitored field names, entity split fields (by/over/partition), influencer field names, and bucket spans. Reads from .ml-config so all jobs are visible even if they have never produced an anomaly. Returns a single summary row. Call this FIRST before other tools to learn valid parameter values for job_id, function, field_name, and entity field filters.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-config | WHERE job_type == \"anomaly_detector\" | STATS job_count = COUNT(*), job_ids = VALUES(job_id), functions = VALUES(`analysis_config.detectors.function`), fields = VALUES(`analysis_config.detectors.field_name`), by_fields = VALUES(`analysis_config.detectors.by_field_name`), over_fields = VALUES(`analysis_config.detectors.over_field_name`), partition_fields = VALUES(`analysis_config.detectors.partition_field_name`), influencers = VALUES(`analysis_config.influencers`), bucket_spans = VALUES(`analysis_config.bucket_span`)"
+ },
+ "parameters": {}
+}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_categories.json b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_categories.json
new file mode 100644
index 0000000..8b2fb01
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_categories.json
@@ -0,0 +1,14 @@
+{
+ "name": "ad_get_categories",
+ "description": "Get log categories for categorization jobs: category definitions, regex patterns, and examples. Useful for understanding what types of log messages the model has learned and which categories are generating anomalies.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-anomalies-* | WHERE result_type == \"category_definition\" AND job_id == ?job_id | SORT category_id ASC | KEEP job_id, category_id, terms, regex, max_matching_length, examples | LIMIT 100"
+ },
+ "parameters": {
+ "job_id": {
+ "type": "string",
+ "description": "The anomaly detection job ID"
+ }
+ }
+}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_forecast_results.json b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_forecast_results.json
new file mode 100644
index 0000000..8738a67
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_forecast_results.json
@@ -0,0 +1,14 @@
+{
+ "name": "ad_get_forecast_results",
+ "description": "Retrieve forecast predictions with upper/lower bounds for capacity planning. Queries model_forecast results from .ml-anomalies-*.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-anomalies-* | WHERE result_type == \"model_forecast\" AND job_id == ?job_id | SORT timestamp ASC | KEEP job_id, timestamp, forecast_prediction, forecast_upper, forecast_lower, bucket_span | LIMIT 200"
+ },
+ "parameters": {
+ "job_id": {
+ "type": "string",
+ "description": "The anomaly detection job ID"
+ }
+ }
+}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_job_messages.json b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_job_messages.json
new file mode 100644
index 0000000..51975dc
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_job_messages.json
@@ -0,0 +1,14 @@
+{
+ "name": "ad_get_job_messages",
+ "description": "Retrieve all notifications for a job from the ML notifications index, including all severity levels (info, warning, error). Covers datafeed warnings, delayed data alerts, missing document warnings, query errors, timeout messages, CCS connectivity issues, memory limit warnings, and job lifecycle events. Returns all levels so the agent can filter by level in its reasoning. Use this for both general notification browsing and targeted datafeed warning investigation.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-notifications-* | WHERE job_id == ?job_id | SORT timestamp DESC | KEEP timestamp, level, message, node_name, job_id | LIMIT 50"
+ },
+ "parameters": {
+ "job_id": {
+ "type": "string",
+ "description": "The anomaly detection job ID"
+ }
+ }
+}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_jobs.json b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_jobs.json
new file mode 100644
index 0000000..fc5f769
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_jobs.json
@@ -0,0 +1,9 @@
+{
+ "name": "ad_get_jobs",
+ "description": "List all anomaly detection jobs with their full configuration: bucket_span, detector functions, monitored field names, entity split fields (partition/by/over), influencers, memory limit, groups, and description. Reads from .ml-config so ALL jobs appear regardless of whether they have run or produced any anomaly results. Use this early in any investigation to understand what jobs exist and how they are configured.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-config | WHERE job_type == \"anomaly_detector\" | KEEP job_id, `analysis_config.bucket_span`, `analysis_config.detectors.function`, `analysis_config.detectors.field_name`, `analysis_config.detectors.partition_field_name`, `analysis_config.detectors.by_field_name`, `analysis_config.detectors.over_field_name`, `analysis_config.influencers`, `analysis_limits.model_memory_limit`, groups, description | SORT job_id ASC | LIMIT 100"
+ },
+ "parameters": {}
+}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_model_plot.json b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_model_plot.json
new file mode 100644
index 0000000..2ac387d
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_model_plot.json
@@ -0,0 +1,22 @@
+{
+ "name": "ad_get_model_plot",
+ "description": "Get model bounds (upper/lower/median) to explain why something was or was not flagged as anomalous. Shows the model's confidence interval at each time point. If actual value is within bounds, no anomaly; if outside, anomaly score depends on distance from bounds.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-anomalies-* | WHERE result_type == \"model_plot\" AND job_id == ?job_id AND timestamp >= ?start_time AND timestamp <= ?end_time | SORT timestamp ASC | KEEP job_id, timestamp, model_lower, model_upper, model_median, actual, partition_field_value, by_field_value | LIMIT 500"
+ },
+ "parameters": {
+ "job_id": {
+ "type": "string",
+ "description": "The anomaly detection job ID"
+ },
+ "start_time": {
+ "type": "string",
+ "description": "Start of time range in ISO 8601 format"
+ },
+ "end_time": {
+ "type": "string",
+ "description": "End of time range in ISO 8601 format"
+ }
+ }
+}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_model_snapshots.json b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_model_snapshots.json
new file mode 100644
index 0000000..0f914ed
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_model_snapshots.json
@@ -0,0 +1,14 @@
+{
+ "name": "ad_get_model_snapshots",
+ "description": "List available model snapshots for a job, including timestamp, description, and size. Used for model revert operations when data quality issues have corrupted the model.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-anomalies-* | WHERE result_type == \"model_snapshot\" AND job_id == ?job_id | SORT timestamp DESC | KEEP job_id, timestamp, description, snapshot_doc_count | LIMIT 20"
+ },
+ "parameters": {
+ "job_id": {
+ "type": "string",
+ "description": "The anomaly detection job ID"
+ }
+ }
+}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_query_anomaly_records.json b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_query_anomaly_records.json
new file mode 100644
index 0000000..aeff156
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_query_anomaly_records.json
@@ -0,0 +1,26 @@
+{
+ "name": "ad_query_anomaly_records",
+ "description": "Search anomaly records by score threshold and time range. Use job_id_pattern='*' for cross-job search (default) or an exact job ID for single-job drill-down. Supports global cross-job search, job-scoped deep dive, absence detection (actual << typical), and value trend analysis. This is the primary tool for questions like 'find anomalies related to entity X' as well as per-job investigation.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-anomalies-* | WHERE result_type == \"record\" AND job_id LIKE ?job_id_pattern AND record_score >= ?min_score AND timestamp >= ?start_time AND timestamp <= ?end_time | SORT record_score DESC | LIMIT 50 | KEEP job_id, timestamp, record_score, function, field_name, actual, typical, by_field_name, by_field_value, over_field_name, over_field_value, partition_field_name, partition_field_value, multi_bucket_impact, initial_record_score, detector_index, probability"
+ },
+ "parameters": {
+ "job_id_pattern": {
+ "type": "string",
+ "description": "Job ID pattern using LIKE wildcards (* = multi-char). Use '*' for all jobs (cross-job search), or an exact job ID to scope to a single job (e.g. 'rcaeval-ob-cpu'). Use 'rcaeval-*' to scope to a prefix group."
+ },
+ "min_score": {
+ "type": "string",
+ "description": "Minimum record_score threshold (0-100). Use 50 for significant anomalies, 25 for broader search, 75 for critical only."
+ },
+ "start_time": {
+ "type": "string",
+ "description": "Start of time range in ISO 8601 format (e.g. 2024-01-01T00:00:00Z)"
+ },
+ "end_time": {
+ "type": "string",
+ "description": "End of time range in ISO 8601 format"
+ }
+ }
+}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_query_anomaly_timeline.json b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_query_anomaly_timeline.json
new file mode 100644
index 0000000..68bfd1e
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_query_anomaly_timeline.json
@@ -0,0 +1,26 @@
+{
+ "name": "ad_query_anomaly_timeline",
+ "description": "Get a time-series summary of anomaly severity across all or selected jobs, bucketed by time. Use for building cross-job swimlanes, detecting anomaly storms (multiple jobs firing together), and computing composite signal scores. When multiple jobs have anomaly_score > threshold in the same bucket, it indicates a common external trigger. Scope to a job group with job_id_pattern (e.g. 'rcaeval-*') or use '*' for all jobs.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-anomalies-* | WHERE result_type == \"bucket\" AND job_id LIKE ?job_id_pattern AND anomaly_score >= ?min_score AND timestamp >= ?start_time AND timestamp <= ?end_time | STATS max_score = MAX(anomaly_score), job_count = COUNT_DISTINCT(job_id), jobs = VALUES(job_id), composite_score = SUM(anomaly_score), avg_score = AVG(anomaly_score) BY timestamp | SORT timestamp"
+ },
+ "parameters": {
+ "job_id_pattern": {
+ "type": "string",
+ "description": "Job ID pattern using LIKE wildcards. Use '*' for all jobs, or a prefix pattern to scope to a related group (e.g. 'rcaeval-*')."
+ },
+ "min_score": {
+ "type": "string",
+ "description": "Minimum bucket anomaly_score. Use 25 for signal boosting (catch coordinated low-severity events), 50 for standard, 75 for critical."
+ },
+ "start_time": {
+ "type": "string",
+ "description": "Start of time range in ISO 8601 format"
+ },
+ "end_time": {
+ "type": "string",
+ "description": "End of time range in ISO 8601 format"
+ }
+ }
+}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_query_influencers.json b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_query_influencers.json
new file mode 100644
index 0000000..ba73a9a
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_query_influencers.json
@@ -0,0 +1,26 @@
+{
+ "name": "ad_query_influencers",
+ "description": "Find the most unusual entities across all or selected jobs for a time range. Answers 'What entities are most anomalous right now?' and 'Which entities appear as influencers in MULTIPLE jobs simultaneously?' When require_multi_job semantics are needed, filter results where job_count > 1 to find shared influencers for cross-job RCA. Scope to a job group with job_id_pattern (e.g. 'rcaeval-*') or use '*' for all jobs.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-anomalies-* | WHERE result_type == \"influencer\" AND job_id LIKE ?job_id_pattern AND influencer_score >= ?min_score AND timestamp >= ?start_time AND timestamp <= ?end_time | STATS total_score = SUM(influencer_score), job_count = COUNT_DISTINCT(job_id), jobs = VALUES(job_id), max_score = MAX(influencer_score) BY influencer_field_name, influencer_field_value | SORT total_score DESC | LIMIT 30"
+ },
+ "parameters": {
+ "job_id_pattern": {
+ "type": "string",
+ "description": "Job ID pattern using LIKE wildcards. Use '*' for all jobs, or a prefix pattern to scope to a related group (e.g. 'rcaeval-*')."
+ },
+ "min_score": {
+ "type": "string",
+ "description": "Minimum influencer_score threshold. Use 25 for broad search (shared influencer discovery), 50 for significant entities."
+ },
+ "start_time": {
+ "type": "string",
+ "description": "Start of time range in ISO 8601 format"
+ },
+ "end_time": {
+ "type": "string",
+ "description": "End of time range in ISO 8601 format"
+ }
+ }
+}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_blast_radius.json b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_blast_radius.json
new file mode 100644
index 0000000..ad63bd0
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_blast_radius.json
@@ -0,0 +1,26 @@
+{
+ "name": "ad_rca_blast_radius",
+ "description": "Measure how widespread a threat or issue is by counting how many hosts, partitions, or entities are affected by a specific anomalous value. Given a value (e.g. process.name: node-backdoor), counts distinct affected partitions across all jobs. Also aggregates weak signals: entities with multiple low-score anomalies across different jobs that individually are below alerting thresholds but collectively indicate a real problem.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-anomalies-* | WHERE result_type == \"record\" AND record_score >= ?min_score AND timestamp >= ?start_time AND timestamp <= ?end_time AND (by_field_value == ?anomalous_value OR over_field_value == ?anomalous_value OR partition_field_value == ?anomalous_value) | STATS affected_partitions = COUNT_DISTINCT(partition_field_value), affected_jobs = COUNT_DISTINCT(job_id), total_anomalies = COUNT(*), max_score = MAX(record_score), time_span_hours = DATE_DIFF(\"hour\", MIN(timestamp), MAX(timestamp)) BY by_field_value | SORT affected_partitions DESC"
+ },
+ "parameters": {
+ "anomalous_value": {
+ "type": "string",
+ "description": "The specific anomalous value to measure blast radius for (e.g. 'node-backdoor', 'suspicious-script.ps1')"
+ },
+ "min_score": {
+ "type": "string",
+ "description": "Minimum record_score. Use low values (10-25) for weak signal aggregation."
+ },
+ "start_time": {
+ "type": "string",
+ "description": "Start of time range in ISO 8601 format"
+ },
+ "end_time": {
+ "type": "string",
+ "description": "End of time range in ISO 8601 format"
+ }
+ }
+}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_correlation.json b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_correlation.json
new file mode 100644
index 0000000..1379832
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_correlation.json
@@ -0,0 +1,26 @@
+{
+ "name": "ad_rca_correlation",
+ "description": "Find temporally correlated anomalies across different jobs. Supports two modes: (1) co_occurrence — anomalies from different jobs in overlapping time windows regardless of shared influencers, essential for cross-domain correlation (K8s + APM + logs); (2) ordered_sequence — anomalies sorted by time to detect cascading failures (network -> app -> DB propagation). The agent examines temporal ordering and job types to infer causality. Use job_id_pattern to scope to a subset of jobs (e.g. 'rcaeval-ob-*') when many jobs exist.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-anomalies-* | WHERE result_type == \"record\" AND job_id LIKE ?job_id_pattern AND record_score >= ?min_score AND timestamp >= ?start_time AND timestamp <= ?end_time | SORT timestamp ASC | KEEP job_id, timestamp, record_score, function, field_name, by_field_name, by_field_value, partition_field_name, partition_field_value, over_field_name, over_field_value, multi_bucket_impact | LIMIT 100"
+ },
+ "parameters": {
+ "job_id_pattern": {
+ "type": "string",
+ "description": "Job ID pattern using LIKE wildcards (* = multi-char). Use '*' for all jobs, 'rcaeval-*' for all RCAEval jobs, 'rcaeval-ob-*' for Online Boutique only, 'nab-*' for NAB jobs, etc."
+ },
+ "min_score": {
+ "type": "string",
+ "description": "Minimum record_score threshold"
+ },
+ "start_time": {
+ "type": "string",
+ "description": "Start of time range in ISO 8601 format"
+ },
+ "end_time": {
+ "type": "string",
+ "description": "End of time range in ISO 8601 format"
+ }
+ }
+}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_cross_job_entity_match.json b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_cross_job_entity_match.json
new file mode 100644
index 0000000..9cef60c
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_cross_job_entity_match.json
@@ -0,0 +1,26 @@
+{
+ "name": "ad_rca_cross_job_entity_match",
+ "description": "Starting from a specific entity value (e.g. a service name extracted from an alert), find ALL other jobs where that entity appears as anomalous — across partition, by, and over fields. Returns per-job summary with max score, anomaly count, detector functions, field names, and the first/last anomaly timestamps. Use first_anomaly to reconstruct chronology: the job that detected the entity earliest is closest to the root cause. Unlike ad_rca_multi_job_entities (which groups by partition_field_value only), this matches the entity value across ALL split field types.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-anomalies-* | WHERE result_type == \"record\" AND record_score >= ?min_score AND timestamp >= ?start_time AND timestamp <= ?end_time AND (partition_field_value == ?entity_value OR by_field_value == ?entity_value OR over_field_value == ?entity_value) | STATS max_score = MAX(record_score), anomaly_count = COUNT(*), functions = VALUES(function), fields = VALUES(field_name), first_anomaly = MIN(timestamp), last_anomaly = MAX(timestamp) BY job_id | SORT max_score DESC"
+ },
+ "parameters": {
+ "entity_value": {
+ "type": "string",
+ "description": "The entity value to search for across all jobs (e.g. 'frontend', 'server-01', 'payment-service'). Extract this from the alerting anomaly's partition_field_value, by_field_value, or over_field_value."
+ },
+ "min_score": {
+ "type": "string",
+ "description": "Minimum record_score threshold. Use 10-25 for comprehensive search (catches weak cascade signals), 50 for significant anomalies only."
+ },
+ "start_time": {
+ "type": "string",
+ "description": "Start of time range in ISO 8601 format. Use a wider window than the alert (e.g. alert_time minus 6 hours) to catch the full cascade."
+ },
+ "end_time": {
+ "type": "string",
+ "description": "End of time range in ISO 8601 format. Use alert_time plus 1-2 hours to catch delayed effects."
+ }
+ }
+}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_detector_fingerprint.json b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_detector_fingerprint.json
new file mode 100644
index 0000000..726356b
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_detector_fingerprint.json
@@ -0,0 +1,26 @@
+{
+ "name": "ad_rca_detector_fingerprint",
+ "description": "For a specific incident time window, produce a fingerprint showing exactly which detectors fired across all or selected jobs and what they monitor. Shows which aspects of the system are anomalous (CPU? Network? Error rate? Latency?). Group by job_id + function + field_name to understand the incident signature. Scope to a job group with job_id_pattern (e.g. 'rcaeval-*') or use '*' for all jobs.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-anomalies-* | WHERE result_type == \"record\" AND job_id LIKE ?job_id_pattern AND record_score >= ?min_score AND timestamp >= ?start_time AND timestamp <= ?end_time | STATS count = COUNT(*), max_score = MAX(record_score), avg_score = AVG(record_score) BY job_id, function, field_name, detector_index | SORT max_score DESC"
+ },
+ "parameters": {
+ "job_id_pattern": {
+ "type": "string",
+ "description": "Job ID pattern using LIKE wildcards. Use '*' for all jobs, or a prefix pattern to scope to a related group (e.g. 'rcaeval-*')."
+ },
+ "min_score": {
+ "type": "string",
+ "description": "Minimum record_score threshold"
+ },
+ "start_time": {
+ "type": "string",
+ "description": "Start of time range in ISO 8601 format"
+ },
+ "end_time": {
+ "type": "string",
+ "description": "End of time range in ISO 8601 format"
+ }
+ }
+}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_entity_profile.json b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_entity_profile.json
new file mode 100644
index 0000000..98d0696
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_entity_profile.json
@@ -0,0 +1,22 @@
+{
+ "name": "ad_rca_entity_profile",
+ "description": "Build a complete anomaly dossier for a suspect entity across ALL jobs. Given an entity value (e.g. host.name: server-01), shows every anomaly where it appeared as an influencer, by_field, partition_field, or over_field value. Use after identifying a suspect via ad_query_influencers to build full context.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-anomalies-* | WHERE result_type == \"record\" AND timestamp >= ?start_time AND timestamp <= ?end_time AND (by_field_value == ?entity_value OR over_field_value == ?entity_value OR partition_field_value == ?entity_value) | SORT timestamp ASC | LIMIT 100 | KEEP job_id, timestamp, record_score, function, field_name, by_field_name, by_field_value, over_field_name, over_field_value, partition_field_name, partition_field_value, actual, typical, multi_bucket_impact"
+ },
+ "parameters": {
+ "entity_value": {
+ "type": "string",
+ "description": "The entity value to profile (e.g. 'server-01', 'payment-service', 'user@example.com')"
+ },
+ "start_time": {
+ "type": "string",
+ "description": "Start of time range in ISO 8601 format"
+ },
+ "end_time": {
+ "type": "string",
+ "description": "End of time range in ISO 8601 format"
+ }
+ }
+}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_multi_job_entities.json b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_multi_job_entities.json
new file mode 100644
index 0000000..1dc7283
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_multi_job_entities.json
@@ -0,0 +1,26 @@
+{
+ "name": "ad_rca_multi_job_entities",
+ "description": "Find entities that are anomalous in MULTIPLE jobs simultaneously — the strongest root cause signal. Returns entities ranked by the number of distinct jobs they appear in, with per-job max scores and detector functions. Entities in 2+ jobs are prime root cause candidates (e.g., a service with CPU AND latency anomalies); entities in only 1 job are likely secondary effects or victims. This is the key discriminator for RCA: resource faults produce multi-job anomalies, network faults produce single-job anomalies.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-anomalies-* | WHERE result_type == \"record\" AND record_score >= ?min_score AND timestamp >= ?start_time AND timestamp <= ?end_time | STATS job_count = COUNT_DISTINCT(job_id), jobs = VALUES(job_id), max_score = MAX(record_score), total_records = COUNT(*), functions = VALUES(function), fields = VALUES(field_name) BY partition_field_value | WHERE job_count >= ?min_job_count | SORT job_count DESC, max_score DESC | LIMIT 20"
+ },
+ "parameters": {
+ "min_score": {
+ "type": "string",
+ "description": "Minimum record_score threshold. Use 25 for broader search (catches weak cascade signals), 50 for significant anomalies."
+ },
+ "min_job_count": {
+ "type": "string",
+ "description": "Minimum number of distinct jobs the entity must appear in. Use 2 to find cross-job root cause candidates, 1 to include all entities."
+ },
+ "start_time": {
+ "type": "string",
+ "description": "Start of time range in ISO 8601 format"
+ },
+ "end_time": {
+ "type": "string",
+ "description": "End of time range in ISO 8601 format"
+ }
+ }
+}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_score_reassessment.json b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_score_reassessment.json
new file mode 100644
index 0000000..32fffd2
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_score_reassessment.json
@@ -0,0 +1,26 @@
+{
+ "name": "ad_rca_score_reassessment",
+ "description": "Find anomalies where the model has significantly changed its assessment over time due to renormalization. This tool defines score_drift = initial_record_score - record_score. When renormalization lowers the current score, initial_record_score stays higher → large positive drift (initial >> current). Large negative drift means the current score rose versus the initial snapshot (upward reconsideration). Records where scores stayed high indicate persistent anomalies the model never explained away. Scope to a specific job with job_id_pattern or use '*' for all jobs.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-anomalies-* | WHERE result_type == \"record\" AND job_id LIKE ?job_id_pattern AND timestamp >= ?start_time AND timestamp <= ?end_time | EVAL score_drift = initial_record_score - record_score | WHERE ABS(score_drift) >= ?min_drift | SORT ABS(score_drift) DESC | LIMIT 30 | KEEP job_id, timestamp, initial_record_score, record_score, score_drift, function, field_name, by_field_value, partition_field_value"
+ },
+ "parameters": {
+ "job_id_pattern": {
+ "type": "string",
+ "description": "Job ID pattern using LIKE wildcards. Use '*' for all jobs, or an exact job ID to scope to a single job (e.g. 'rcaeval-ob-cpu')."
+ },
+ "min_drift": {
+ "type": "string",
+ "description": "Minimum score point difference between initial and current score (default 20)"
+ },
+ "start_time": {
+ "type": "string",
+ "description": "Start of time range in ISO 8601 format"
+ },
+ "end_time": {
+ "type": "string",
+ "description": "End of time range in ISO 8601 format"
+ }
+ }
+}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_source_evidence.json b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_source_evidence.json
new file mode 100644
index 0000000..e4cab76
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_source_evidence.json
@@ -0,0 +1,22 @@
+{
+ "name": "ad_rca_source_evidence",
+ "description": "After identifying an anomaly, retrieve raw source documents from the ORIGINAL data index for the anomaly's time window. This is the evidence drilldown — see the actual log lines, traces, or metrics that caused the statistical deviation. Without this, the agent can say 'something unusual happened' but cannot say 'here is what actually happened'.\n\nUsage: First find the source index from the job's datafeed configuration (via ad_get_job_datafeed_config). Pass the index name as source_index using LIKE wildcards (* for multi-char, ? for single-char). The tool returns raw documents with all original fields from that index.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM * METADATA _index | WHERE _index LIKE ?source_index AND @timestamp >= ?start_time AND @timestamp <= ?end_time | SORT @timestamp DESC | LIMIT 50"
+ },
+ "parameters": {
+ "source_index": {
+ "type": "string",
+ "description": "Source data index name or LIKE pattern (* = multi-char wildcard). Get this from the job's datafeed config. Examples: 'rcaeval-re1-ob', 'nab', 'otel-flat-*', 'smd'"
+ },
+ "start_time": {
+ "type": "string",
+ "description": "Start of time range in ISO 8601 format"
+ },
+ "end_time": {
+ "type": "string",
+ "description": "End of time range in ISO 8601 format"
+ }
+ }
+}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_search_log_category_examples.json b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_search_log_category_examples.json
new file mode 100644
index 0000000..4523ae9
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_search_log_category_examples.json
@@ -0,0 +1,22 @@
+{
+ "name": "ad_search_log_category_examples",
+ "description": "Search for log message examples in the source data for a specific time window. Used for two-window comparison in log categorization RCA: run once for a baseline window (e.g., 24h before anomaly) and once for the anomaly window. Compare the samples to identify what changed in the variable parts (IPs, hostnames, error codes, paths) that caused the category count anomaly. Returns raw log documents so you can inspect the categorization_field_name field specified in the job config.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM * METADATA _index | WHERE _index LIKE ?source_index AND @timestamp >= ?start_time AND @timestamp <= ?end_time | SORT @timestamp DESC | LIMIT 50"
+ },
+ "parameters": {
+ "source_index": {
+ "type": "string",
+ "description": "Source data index name or LIKE pattern from the job's datafeed config. Get this from ad_get_job_datafeed_config. Examples: 'logs-*', 'filebeat-*', 'otel-logs-*'"
+ },
+ "start_time": {
+ "type": "string",
+ "description": "Start of time range in ISO 8601 format. For baseline window, use a period before the anomaly (e.g., 24h before)."
+ },
+ "end_time": {
+ "type": "string",
+ "description": "End of time range in ISO 8601 format"
+ }
+ }
+}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_ts_bucket_event_gaps.json b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_ts_bucket_event_gaps.json
new file mode 100644
index 0000000..b437cf5
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_ts_bucket_event_gaps.json
@@ -0,0 +1,22 @@
+{
+ "name": "ad_ts_bucket_event_gaps",
+ "description": "Find buckets with zero or suspiciously low event counts for a specific job. These are the buckets actually affected by missing data. Compare event_count across buckets to identify time ranges where data was lost. Correlate with delayed data annotations and anomaly scores to confirm false positives from missing data.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-anomalies-* | WHERE result_type == \"bucket\" AND job_id == ?job_id AND timestamp >= ?start_time AND timestamp <= ?end_time | SORT timestamp ASC | KEEP timestamp, event_count, anomaly_score, bucket_span, is_interim | LIMIT 500"
+ },
+ "parameters": {
+ "job_id": {
+ "type": "string",
+ "description": "The anomaly detection job ID"
+ },
+ "start_time": {
+ "type": "string",
+ "description": "Start of time range in ISO 8601 format"
+ },
+ "end_time": {
+ "type": "string",
+ "description": "End of time range in ISO 8601 format"
+ }
+ }
+}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_ts_delayed_data_annotations.json b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_ts_delayed_data_annotations.json
new file mode 100644
index 0000000..142f55f
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_ts_delayed_data_annotations.json
@@ -0,0 +1,14 @@
+{
+ "name": "ad_ts_delayed_data_annotations",
+ "description": "Retrieve all delayed data annotations for a job, showing exactly when and how many documents were missed. The annotation field contains text like 'Datafeed has missed 30 documents due to ingest latency...' \u2014 frequent annotations indicate chronic ingest latency. This is the starting point for any 'missing documents' investigation.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-annotations-* | WHERE job_id == ?job_id AND event == \"delayed_data\" | SORT timestamp DESC | KEEP job_id, timestamp, end_timestamp, annotation | LIMIT 100"
+ },
+ "parameters": {
+ "job_id": {
+ "type": "string",
+ "description": "The anomaly detection job ID"
+ }
+ }
+}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_ts_ingest_latency_estimate.json b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_ts_ingest_latency_estimate.json
new file mode 100644
index 0000000..9895d1e
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_ts_ingest_latency_estimate.json
@@ -0,0 +1,22 @@
+{
+ "name": "ad_ts_ingest_latency_estimate",
+ "description": "Measure actual ingest latency by comparing event timestamps with ingestion timestamps in the SOURCE data. Determines whether the current query_delay is sufficient. If P95(event.ingested - @timestamp) > query_delay, data will be lost. Requires the source index to have an event.ingested or _ingest.timestamp field.\n\nUsage: Get the source index from the job's datafeed config (via ad_get_job_datafeed_config). Pass it as source_index using LIKE wildcards if needed.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM * METADATA _index | WHERE _index LIKE ?source_index AND @timestamp >= ?start_time AND @timestamp <= ?end_time AND event.ingested IS NOT NULL | EVAL latency_seconds = DATE_DIFF(\"second\", @timestamp, event.ingested) | STATS p50_latency = PERCENTILE(latency_seconds, 50), p95_latency = PERCENTILE(latency_seconds, 95), p99_latency = PERCENTILE(latency_seconds, 99), max_latency = MAX(latency_seconds), doc_count = COUNT(*) | LIMIT 1"
+ },
+ "parameters": {
+ "source_index": {
+ "type": "string",
+ "description": "Source data index name or LIKE pattern (* = multi-char wildcard). Get this from the job's datafeed config. Examples: 'rcaeval-re1-ob', 'nab', 'otel-flat-*', 'smd'"
+ },
+ "start_time": {
+ "type": "string",
+ "description": "Start of time range in ISO 8601 format"
+ },
+ "end_time": {
+ "type": "string",
+ "description": "End of time range in ISO 8601 format"
+ }
+ }
+}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_ts_model_memory_health.json b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_ts_model_memory_health.json
new file mode 100644
index 0000000..01b5459
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/tools/esql/ad_ts_model_memory_health.json
@@ -0,0 +1,18 @@
+{
+ "name": "ad_ts_model_memory_health",
+ "description": "Get memory health and growth trend for a job. Returns model_size_stats records in reverse-chronological order. Use limit=1 for a fast current-state snapshot (hard_limit/soft_limit check); use limit=500 for full memory growth trend analysis (stable plateau / linear / exponential). Interpretation: hard_limit = CRITICAL (job blind to new entities), soft_limit = WARNING (aggressive pruning), model_bytes/model_bytes_memory_limit > 0.8 = APPROACHING LIMIT.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-anomalies-* | WHERE result_type == \"model_size_stats\" AND job_id == ?job_id | SORT timestamp DESC | KEEP job_id, timestamp, model_bytes, peak_model_bytes, model_bytes_memory_limit, model_bytes_exceeded, memory_status, total_by_field_count, total_over_field_count, total_partition_field_count, bucket_allocation_failures_count | LIMIT ?limit"
+ },
+ "parameters": {
+ "job_id": {
+ "type": "string",
+ "description": "The anomaly detection job ID"
+ },
+ "limit": {
+ "type": "string",
+ "description": "Number of historical records to return. Use 1 for current memory status (fast snapshot), 500 for full memory growth trend analysis."
+ }
+ }
+}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_create_calendar_event.yaml b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_create_calendar_event.yaml
new file mode 100644
index 0000000..d2c4398
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_create_calendar_event.yaml
@@ -0,0 +1,42 @@
+name: ad_create_calendar_event
+description: >
+ Add a scheduled event to a calendar to suppress false positives during known downtime, maintenance windows, or
+ holidays.
+enabled: true
+tags: ["anomaly-detection", "remediation"]
+
+inputs:
+ - name: calendar_id
+ type: string
+ description: The calendar ID
+ - name: event_description
+ type: string
+ description: "Event description (e.g., 'Planned maintenance window')"
+ - name: start_time
+ type: string
+ description: Event start time in ISO 8601 format
+ - name: end_time
+ type: string
+ description: Event end time in ISO 8601 format
+
+triggers:
+ - type: manual
+
+steps:
+ - name: create_event
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/calendars/{{ inputs.calendar_id }}/events
+ body:
+ events:
+ - description: "{{ inputs.event_description }}"
+ start_time: "{{ inputs.start_time }}"
+ end_time: "{{ inputs.end_time }}"
+
+ - name: result
+ type: console
+ with:
+ message: |
+ Created event '{{ inputs.event_description }}' on calendar {{ inputs.calendar_id }}
+ from {{ inputs.start_time }} to {{ inputs.end_time }}.
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_create_datafeed.yaml b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_create_datafeed.yaml
new file mode 100644
index 0000000..5633649
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_create_datafeed.yaml
@@ -0,0 +1,34 @@
+name: ad_create_datafeed
+description: >
+ Create or replace a datafeed for an anomaly detection job via PUT _ml/datafeeds/{datafeed_id}. Use after job creation
+ and before opening the job / starting the datafeed.
+enabled: true
+tags: ["anomaly-detection", "management"]
+
+inputs:
+ - name: datafeed_id
+ type: string
+ description: Datafeed ID (typically datafeed-{job_id})
+ - name: datafeed_body
+ type: string
+ description: >
+ Full datafeed configuration as JSON text. Parsed with json_parse so the PUT body is a structured object for the ML
+ API, not a JSON-encoded string.
+
+triggers:
+ - type: manual
+
+steps:
+ - name: put_datafeed
+ type: elasticsearch.request
+ with:
+ method: PUT
+ path: /_ml/datafeeds/{{ inputs.datafeed_id }}
+ body: "${{ inputs.datafeed_body | json_parse }}"
+
+ - name: result
+ type: console
+ with:
+ message: |
+ Datafeed {{ inputs.datafeed_id }}:
+ {{ steps.put_datafeed.output | json:2 }}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_create_job.yaml b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_create_job.yaml
new file mode 100644
index 0000000..61ee9b5
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_create_job.yaml
@@ -0,0 +1,36 @@
+name: ad_create_job
+description: >
+ Create a new anomaly detection job from a configuration. Stretch goal — for advanced agent use cases where the agent
+ helps design and create jobs based on data exploration. job_body is JSON text; the step uses Liquid json_parse and typed
+ interpolation (${{ }}) so the PUT body is a structured JSON object for the ML API, not a JSON-encoded string.
+enabled: true
+tags: ["anomaly-detection", "management"]
+
+inputs:
+ - name: job_id
+ type: string
+ description: The new job ID to create
+ - name: job_body
+ type: string
+ description: >
+ Full job configuration as JSON text (same shape as PUT /_ml/anomaly_detectors/{job_id}). Parsed with json_parse
+ before the request so the HTTP body is an object. Runners that support a native object input may still pass JSON
+ text here.
+
+triggers:
+ - type: manual
+
+steps:
+ - name: create_job
+ type: elasticsearch.request
+ with:
+ method: PUT
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}
+ body: "${{ inputs.job_body | json_parse }}"
+
+ - name: result
+ type: console
+ with:
+ message: |
+ Created job {{ inputs.job_id }}:
+ {{ steps.create_job.output | json:2 }}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_discover_jobs_by_datafeed_index.yaml b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_discover_jobs_by_datafeed_index.yaml
new file mode 100644
index 0000000..538c00e
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_discover_jobs_by_datafeed_index.yaml
@@ -0,0 +1,63 @@
+name: ad_discover_jobs_by_datafeed_index
+description: >
+ Given a job of interest, find all other jobs whose datafeed reads from overlapping source indices. Step 1 retrieves
+ the target job config (including datafeed_config.indices). Step 2 logs the target job summary. Step 3 iterates over
+ each index in that list and queries .ml-config for other datafeed documents containing the same index, logging matches
+ per index pattern. Jobs reading from the same indices monitor the same system and are strong candidates for cross-job
+ correlation.
+enabled: true
+tags: ["anomaly-detection", "rca", "discovery"]
+
+inputs:
+ - name: job_id
+ type: string
+ description: >
+ The job ID of interest whose source indices you want to match against all other jobs.
+
+triggers:
+ - type: manual
+
+steps:
+ - name: get_target_job
+ type: elasticsearch.request
+ with:
+ method: GET
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}
+
+ - name: log_target
+ type: console
+ with:
+ message: |
+ TARGET JOB: {{ inputs.job_id }}
+ Source indices: {{ steps.get_target_job.output.jobs[0].datafeed_config.indices }}
+
+ Searching for jobs with overlapping source indices...
+
+ - name: find_related_jobs
+ type: foreach
+ foreach: "${{ steps.get_target_job.output.jobs[0].datafeed_config.indices }}"
+ steps:
+ - name: search_matching_datafeeds
+ type: elasticsearch.search
+ with:
+ index: .ml-config
+ size: 200
+ _source: ["job_id", "datafeed_id", "indices"]
+ query:
+ bool:
+ filter:
+ - term:
+ config_type: datafeed
+ - term:
+ indices: "{{ foreach.item }}"
+ must_not:
+ - term:
+ job_id: "{{ inputs.job_id }}"
+
+ - name: log_matches
+ type: console
+ with:
+ message: |
+ [{{ foreach.index | plus: 1 }}/{{ foreach.total }}] Index: {{ foreach.item }}
+ Matching jobs ({{ steps.search_matching_datafeeds.output.hits.total.value }}):
+ {{ steps.search_matching_datafeeds.output.hits.hits | json:2 }}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_estimate_memory_requirement.yaml b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_estimate_memory_requirement.yaml
new file mode 100644
index 0000000..5230bfb
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_estimate_memory_requirement.yaml
@@ -0,0 +1,61 @@
+name: ad_estimate_memory_requirement
+description: >
+ Compute a principled model_memory_limit estimate by automatically sampling cardinality from source data and calling
+ the Estimate Model Memory API. Dramatically better than guessing or peak_model_bytes * 1.3 because it uses the exact
+ same estimation algorithm Elasticsearch uses internally. Steps: (1) retrieve job+datafeed config, (2) identify fields
+ requiring cardinality estimates, (3) compute overall_cardinality via cardinality aggregations, (4) compute
+ max_bucket_cardinality via date_histogram + cardinality + max_bucket pipeline, (5) call Estimate Model Memory API, (6)
+ compare with current state, (7) produce recommendation.
+enabled: true
+tags: ["anomaly-detection", "troubleshooting", "memory"]
+
+inputs:
+ - name: job_id
+ type: string
+ description: The anomaly detection job ID
+
+triggers:
+ - type: manual
+
+steps:
+ # Step 1: Retrieve job configuration
+ - name: get_job_config
+ type: elasticsearch.request
+ with:
+ method: GET
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}
+
+ - name: get_job_stats
+ type: elasticsearch.request
+ with:
+ method: GET
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}/_stats
+
+ # Step 2: Log what we found
+ # The agent (or manual reviewer) inspects the config to identify:
+ # - overall_fields: all partition/by/over field names from detectors
+ # - pure_influencers: influencer fields NOT used in any detector split
+ # - datafeed source indices and query
+ - name: log_config
+ type: console
+ with:
+ message: |
+ Job config retrieved for {{ inputs.job_id }}.
+ Analysis config: {{ steps.get_job_config.output | json:2 }}
+ Current stats: {{ steps.get_job_stats.output | json:2 }}
+
+ MANUAL STEP REQUIRED:
+ 1. From analysis_config.detectors, extract all unique by_field_name,
+ over_field_name, partition_field_name values → these are "overall_fields"
+ 2. From analysis_config.influencers, find fields NOT in any detector →
+ these are "pure_influencers"
+ 3. From datafeed_config, get indices[] and query{}
+ 4. Run cardinality aggregations on those indices for each field
+ 5. Run date_histogram(bucket_span) + cardinality sub-agg + max_bucket
+ pipeline for each pure influencer
+ 6. Call POST _ml/anomaly_detectors/_estimate_model_memory with the
+ analysis_config and computed cardinalities
+ 7. Compare the estimate with current model_memory_limit and model_bytes
+
+ See tools/workflow/ad_estimate_memory_requirement.json for the full
+ step-by-step algorithm.
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_get_calendar_events.yaml b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_get_calendar_events.yaml
new file mode 100644
index 0000000..81496d5
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_get_calendar_events.yaml
@@ -0,0 +1,28 @@
+name: ad_get_calendar_events
+description: >
+ Get scheduled events from calendars (maintenance windows, holidays). These suppress anomaly detection during known
+ downtime periods.
+enabled: true
+tags: ["anomaly-detection", "config"]
+
+inputs:
+ - name: calendar_id
+ type: string
+ description: The calendar ID
+
+triggers:
+ - type: manual
+
+steps:
+ - name: get_events
+ type: elasticsearch.request
+ with:
+ method: GET
+ path: /_ml/calendars/{{ inputs.calendar_id }}/events
+
+ - name: result
+ type: console
+ with:
+ message: |
+ Calendar events for {{ inputs.calendar_id }}:
+ {{ steps.get_events.output | json:2 }}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_get_job_datafeed_config.yaml b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_get_job_datafeed_config.yaml
new file mode 100644
index 0000000..fdb1ddc
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_get_job_datafeed_config.yaml
@@ -0,0 +1,36 @@
+name: ad_get_job_datafeed_config
+description: >
+ Fetch complete job and datafeed configuration in one call: detectors, by/over/partition fields, bucket_span,
+ frequency, query_delay, delayed_data_check_config, source indices, and datafeed query. Essential for troubleshooting
+ and for ad_estimate_memory_requirement.
+enabled: true
+tags: ["anomaly-detection", "config"]
+
+inputs:
+ - name: job_id
+ type: string
+ description: The anomaly detection job ID
+
+triggers:
+ - type: manual
+
+steps:
+ - name: get_job
+ type: elasticsearch.request
+ with:
+ method: GET
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}
+
+ - name: get_job_stats
+ type: elasticsearch.request
+ with:
+ method: GET
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}/_stats
+
+ - name: summary
+ type: console
+ with:
+ message: |
+ Job: {{ inputs.job_id }}
+ Config: {{ steps.get_job.output | json:2 }}
+ Stats: {{ steps.get_job_stats.output | json:2 }}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_get_log_categories.yaml b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_get_log_categories.yaml
new file mode 100644
index 0000000..4bbac2b
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_get_log_categories.yaml
@@ -0,0 +1,32 @@
+name: ad_get_log_categories
+description: >
+ Retrieve ML log category details — terms, regex pattern, and example messages — for a specific category from a log
+ categorization job. Use when investigating anomalies where by_field_name == "mlcategory" to understand what type of
+ log message the category represents before comparing samples across time windows.
+enabled: true
+tags: ["anomaly-detection", "log-categorization"]
+
+inputs:
+ - name: job_id
+ type: string
+ description: The categorization job ID
+ - name: category_id
+ type: string
+ description: "The category ID from the anomaly record's by_field_value"
+
+triggers:
+ - type: manual
+
+steps:
+ - name: get_categories
+ type: elasticsearch.request
+ with:
+ method: GET
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}/results/categories/{{ inputs.category_id }}
+
+ - name: result
+ type: console
+ with:
+ message: |
+ Log category {{ inputs.category_id }} for job {{ inputs.job_id }}:
+ {{ steps.get_categories.output | json:2 }}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_manage_datafeed.yaml b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_manage_datafeed.yaml
new file mode 100644
index 0000000..8c79c63
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_manage_datafeed.yaml
@@ -0,0 +1,31 @@
+name: ad_manage_datafeed
+description: >
+ Start or stop a datafeed via POST _ml/datafeeds/{id}/{_start|_stop}. Used in remediation sequences (for example stop
+ before updating query_delay, then restart). For payload preview use ad_preview_datafeed_with_latency (GET
+ _ml/datafeeds/{id}/_preview); preview is not supported here because it requires GET, not POST.
+enabled: true
+tags: ["anomaly-detection", "remediation"]
+
+inputs:
+ - name: datafeed_id
+ type: string
+ description: "The datafeed ID (typically 'datafeed-{job_id}')"
+ - name: action
+ type: string
+ description: "Action to perform: _start or _stop only"
+
+triggers:
+ - type: manual
+
+steps:
+ - name: manage_datafeed
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/datafeeds/{{ inputs.datafeed_id }}/{{ inputs.action }}
+
+ - name: result
+ type: console
+ with:
+ message: |
+ Datafeed {{ inputs.datafeed_id }} action {{ inputs.action }}: {{ steps.manage_datafeed.output | json:2 }}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_open_job.yaml b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_open_job.yaml
new file mode 100644
index 0000000..9e9182a
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_open_job.yaml
@@ -0,0 +1,27 @@
+name: ad_open_job
+description: >
+ Open an anomaly detection job so it can receive data and run analysis (POST _ml/anomaly_detectors/{job_id}/_open).
+enabled: true
+tags: ["anomaly-detection", "management"]
+
+inputs:
+ - name: job_id
+ type: string
+ description: The anomaly detection job ID
+
+triggers:
+ - type: manual
+
+steps:
+ - name: open_job
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}/_open
+
+ - name: result
+ type: console
+ with:
+ message: |
+ Opened job {{ inputs.job_id }}:
+ {{ steps.open_job.output | json:2 }}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_preview_datafeed_with_latency.yaml b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_preview_datafeed_with_latency.yaml
new file mode 100644
index 0000000..151ac43
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_preview_datafeed_with_latency.yaml
@@ -0,0 +1,28 @@
+name: ad_preview_datafeed_with_latency
+description: >
+ Preview a datafeed's source payload and measure effective latency before tuning query_delay. Shows what data the
+ datafeed would see at query time, helping identify fields available for latency measurement.
+enabled: true
+tags: ["anomaly-detection", "troubleshooting"]
+
+inputs:
+ - name: datafeed_id
+ type: string
+ description: The datafeed ID to preview
+
+triggers:
+ - type: manual
+
+steps:
+ - name: preview_datafeed
+ type: elasticsearch.request
+ with:
+ method: GET
+ path: /_ml/datafeeds/{{ inputs.datafeed_id }}/_preview
+
+ - name: result
+ type: console
+ with:
+ message: |
+ Datafeed preview for {{ inputs.datafeed_id }}:
+ {{ steps.preview_datafeed.output | json:2 }}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_revert_model_snapshot.yaml b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_revert_model_snapshot.yaml
new file mode 100644
index 0000000..e17ab55
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_revert_model_snapshot.yaml
@@ -0,0 +1,55 @@
+name: ad_revert_model_snapshot
+description: >
+ Revert a job's model to a previous snapshot to 'unlearn' bad data. Stops the datafeed, closes the job, reverts to the
+ snapshot, reopens, and restarts the datafeed from the snapshot timestamp.
+enabled: true
+tags: ["anomaly-detection", "remediation"]
+
+inputs:
+ - name: job_id
+ type: string
+ description: The anomaly detection job ID
+ - name: snapshot_id
+ type: string
+ description: The snapshot ID to revert to
+
+triggers:
+ - type: manual
+
+steps:
+ - name: stop_datafeed
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/datafeeds/datafeed-{{ inputs.job_id }}/_stop
+
+ - name: close_job
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}/_close
+
+ - name: revert_snapshot
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}/model_snapshots/{{ inputs.snapshot_id }}/_revert
+
+ - name: open_job
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}/_open
+
+ - name: start_datafeed
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/datafeeds/datafeed-{{ inputs.job_id }}/_start
+
+ - name: result
+ type: console
+ with:
+ message: |
+ Reverted {{ inputs.job_id }} to snapshot {{ inputs.snapshot_id }}.
+ Job reopened and datafeed restarted from snapshot timestamp.
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_search_log_category_examples.yaml b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_search_log_category_examples.yaml
new file mode 100644
index 0000000..e3f7b96
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_search_log_category_examples.yaml
@@ -0,0 +1,54 @@
+name: ad_search_log_category_examples
+description: >
+ Search a source log index for messages matching specific ML category terms within a time window. Returns concrete log
+ examples belonging to a category during a specific period. Call twice — once for the anomaly window, once for a
+ baseline period (e.g. 24h prior) — to compare log content and identify what changed in the variable parts (IPs,
+ hostnames, error codes) that may reveal the root cause.
+enabled: true
+tags: ["anomaly-detection", "log-categorization", "evidence"]
+
+inputs:
+ - name: source_index
+ type: string
+ description: "The source log index (from ad_get_job_datafeed_config)"
+ default: "it_ops_logs"
+ - name: search_terms
+ type: string
+ description: "Category terms copied from ad_get_log_categories output"
+ - name: start_time
+ type: string
+ description: "Start of the time window (ISO 8601)"
+ - name: end_time
+ type: string
+ description: "End of the time window (ISO 8601)"
+
+triggers:
+ - type: manual
+
+steps:
+ - name: search_logs
+ type: elasticsearch.search
+ with:
+ index: "{{ inputs.source_index }}"
+ size: 20
+ sort: "@timestamp:desc"
+ query:
+ bool:
+ must:
+ - match:
+ message:
+ query: "{{ inputs.search_terms }}"
+ minimum_should_match: "70%"
+ filter:
+ - range:
+ "@timestamp":
+ gte: "{{ inputs.start_time }}"
+ lte: "{{ inputs.end_time }}"
+
+ - name: result
+ type: console
+ with:
+ message: |
+ Log examples matching category terms in {{ inputs.source_index }}
+ Time window: {{ inputs.start_time }} to {{ inputs.end_time }}
+ Results: {{ steps.search_logs.output.hits.hits | json:2 }}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_ts_ccs_diagnostics.yaml b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_ts_ccs_diagnostics.yaml
new file mode 100644
index 0000000..ed4b397
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_ts_ccs_diagnostics.yaml
@@ -0,0 +1,29 @@
+name: ad_ts_ccs_diagnostics
+description: >
+ Diagnose cross-cluster search (CCS) issues for datafeeds that query remote clusters. Checks remote cluster
+ connectivity, latency, and error rates. Helps identify per-cluster skew contributing to missing data.
+enabled: true
+tags: ["anomaly-detection", "troubleshooting", "ccs"]
+
+triggers:
+ - type: manual
+
+steps:
+ - name: check_remote_clusters
+ type: elasticsearch.request
+ with:
+ method: GET
+ path: /_remote/info
+
+ - name: check_cluster_health
+ type: elasticsearch.request
+ with:
+ method: GET
+ path: /_cluster/health
+
+ - name: result
+ type: console
+ with:
+ message: |
+ Remote clusters: {{ steps.check_remote_clusters.output | json:2 }}
+ Cluster health: {{ steps.check_cluster_health.output | json:2 }}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_update_datafeed_query_delay.yaml b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_update_datafeed_query_delay.yaml
new file mode 100644
index 0000000..fe0e3e8
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_update_datafeed_query_delay.yaml
@@ -0,0 +1,45 @@
+name: ad_update_datafeed_query_delay
+description: >
+ Update the query_delay setting on a datafeed. The datafeed must be stopped first. Larger query_delay captures more
+ late-arriving data but delays anomaly alerts. Recommended: set to P95 ingest latency + buffer.
+enabled: true
+tags: ["anomaly-detection", "remediation"]
+
+inputs:
+ - name: datafeed_id
+ type: string
+ description: The datafeed ID
+ - name: new_query_delay
+ type: string
+ description: "New query_delay value (e.g., '3m', '120s', '5m')"
+
+triggers:
+ - type: manual
+
+steps:
+ - name: stop_datafeed
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/datafeeds/{{ inputs.datafeed_id }}/_stop
+
+ - name: update_query_delay
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/datafeeds/{{ inputs.datafeed_id }}/_update
+ body:
+ query_delay: "{{ inputs.new_query_delay }}"
+
+ - name: start_datafeed
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/datafeeds/{{ inputs.datafeed_id }}/_start
+
+ - name: result
+ type: console
+ with:
+ message: |
+ Updated query_delay on {{ inputs.datafeed_id }} to {{ inputs.new_query_delay }}.
+ Datafeed restarted.
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_update_delayed_data_check_config.yaml b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_update_delayed_data_check_config.yaml
new file mode 100644
index 0000000..78e475b
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_update_delayed_data_check_config.yaml
@@ -0,0 +1,47 @@
+name: ad_update_delayed_data_check_config
+description: >
+ Update the delayed_data_check_config on a job to control how aggressively delayed data is detected. POST nests under
+ analysis_config.delayed_data_check_config. A data.parseJson step builds JSON so enabled is a boolean (not a quoted
+ string) and check_window is omitted when the input is empty or whitespace-only after trim.
+enabled: true
+tags: ["anomaly-detection", "remediation"]
+
+inputs:
+ - name: job_id
+ type: string
+ description: The anomaly detection job ID
+ - name: enabled
+ type: boolean
+ description: Whether to enable delayed data checks
+ - name: check_window
+ type: string
+ description: "Time window to check for delayed data (e.g., '2h'). Leave empty, omit, or whitespace-only to exclude check_window from the update body."
+
+triggers:
+ - type: manual
+
+steps:
+ - name: compose_payload
+ type: data.parseJson
+ source: |
+ {%- assign cw = inputs.check_window | default: "" | strip -%}
+ {%- if cw != "" -%}
+ {"analysis_config":{"delayed_data_check_config":{"enabled":{{ inputs.enabled }},"check_window":{{ cw | json }}}}
+ {%- else -%}
+ {"analysis_config":{"delayed_data_check_config":{"enabled":{{ inputs.enabled }}}}
+ {%- endif -%}
+ with: {}
+
+ - name: update_delayed_data
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}/_update
+ body: "${{ steps.compose_payload.output }}"
+
+ - name: result
+ type: console
+ with:
+ message: |
+ Updated delayed_data_check_config on {{ inputs.job_id }}:
+ {{ steps.update_delayed_data.output | json:2 }}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_update_model_memory_limit.yaml b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_update_model_memory_limit.yaml
new file mode 100644
index 0000000..0198d95
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_update_model_memory_limit.yaml
@@ -0,0 +1,60 @@
+name: ad_update_model_memory_limit
+description: >
+ Remediation workflow to change analysis_limits.model_memory_limit on an anomaly detector (for example after
+ ad_estimate_memory_requirement). Stops the datafeed, closes the job, POSTs /_ml/anomaly_detectors/{job_id}/_update with
+ the new limit, opens the job, and starts the datafeed again. Expects the default datafeed id datafeed-{job_id}. You
+ cannot decrease model_memory_limit below current model_bytes — clone the job to shrink.
+enabled: true
+tags: ["anomaly-detection", "remediation"]
+
+inputs:
+ - name: job_id
+ type: string
+ description: The anomaly detection job ID
+ - name: model_memory_limit
+ type: string
+ description: "New model_memory_limit value (e.g., '512mb', '1gb')"
+
+triggers:
+ - type: manual
+
+steps:
+ - name: stop_datafeed
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/datafeeds/datafeed-{{ inputs.job_id }}/_stop
+
+ - name: close_job
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}/_close
+
+ - name: update_model_memory_limit
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}/_update
+ body:
+ analysis_limits:
+ model_memory_limit: "{{ inputs.model_memory_limit }}"
+
+ - name: open_job
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}/_open
+
+ - name: start_datafeed
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/datafeeds/datafeed-{{ inputs.job_id }}/_start
+
+ - name: result
+ type: console
+ with:
+ message: |
+ Updated model_memory_limit on {{ inputs.job_id }} to {{ inputs.model_memory_limit }}.
+ Job reopened and datafeed restarted.
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_validate_job_spec.yaml b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_validate_job_spec.yaml
new file mode 100644
index 0000000..ad36027
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_validate_job_spec.yaml
@@ -0,0 +1,31 @@
+name: ad_validate_job_spec
+description: >
+ Validate an anomaly detection job configuration before creation. POSTs to the ML validate endpoint with the same JSON
+ document shape as PUT job creation. Requires manage_ml (cluster).
+enabled: true
+tags: ["anomaly-detection", "management"]
+
+inputs:
+ - name: job_body
+ type: string
+ description: >
+ Full job configuration as JSON text (same shape as job creation). Parsed with json_parse so the POST body is a
+ structured object for _validate, not a JSON string value.
+
+triggers:
+ - type: manual
+
+steps:
+ - name: validate_job
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/anomaly_detectors/_validate
+ body: "${{ inputs.job_body | json_parse }}"
+
+ - name: result
+ type: console
+ with:
+ message: |
+ Validation result:
+ {{ steps.validate_job.output | json:2 }}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_validate_ml_tool_permissions.yaml b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_validate_ml_tool_permissions.yaml
new file mode 100644
index 0000000..a31104b
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_validate_ml_tool_permissions.yaml
@@ -0,0 +1,41 @@
+name: ad_validate_ml_tool_permissions
+description: >
+ Preflight check for core ML result and config indices via _has_privileges. Verifies read + view_index_metadata on
+ .ml-anomalies-*, .ml-config, .ml-annotations-*, and .ml-notifications-*. Does not check privileges on job-specific
+ source data indices — validate those separately before ad_rca_source_evidence, datafeed preview, or ad_wf_ts_field_cardinality.
+enabled: true
+tags: ["anomaly-detection", "diagnostics"]
+
+triggers:
+ - type: manual
+
+steps:
+ - name: check_security
+ type: elasticsearch.request
+ with:
+ method: GET
+ path: /_security/_authenticate
+
+ - name: check_ml_privileges
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_security/user/_has_privileges
+ body:
+ index:
+ - names: [".ml-anomalies-*"]
+ privileges: ["read", "view_index_metadata"]
+ - names: [".ml-config"]
+ privileges: ["read", "view_index_metadata"]
+ - names: [".ml-annotations-*"]
+ privileges: ["read", "view_index_metadata"]
+ - names: [".ml-notifications-*"]
+ privileges: ["read", "view_index_metadata"]
+
+ - name: result
+ type: console
+ with:
+ message: |
+ User: {{ steps.check_security.output.username }}
+ Roles: {{ steps.check_security.output.roles | json }}
+ ML index privileges: {{ steps.check_ml_privileges.output | json:2 }}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_wf_troubleshoot_anomaly_score.yaml b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_wf_troubleshoot_anomaly_score.yaml
new file mode 100644
index 0000000..5915174
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_wf_troubleshoot_anomaly_score.yaml
@@ -0,0 +1,128 @@
+name: ad_wf_troubleshoot_anomaly_score
+description: >
+ Stored workflow for troubleshooting unexpectedly high or low anomaly scores. Implements a branching decision tree: (0)
+ gate checks (sufficient data, memory status, delayed data, UI aggregation), (1) UI display vs real score
+ (renormalization), (2) job configuration analysis (bucket_span, detector function, partition/influencer
+ fields, custom rules),
+ (3) model learning and data characteristics (insufficient history, high variance
+ penalty, model adaptation),
+ (4) score factor education (anomaly_score_explanation breakdown). Trigger: 'Why is my score low?', 'Expected anomaly
+ not detected', 'Score too high/low'.
+enabled: true
+tags: ["anomaly-detection", "troubleshooting", "scores"]
+
+inputs:
+ - name: job_id
+ type: string
+ description: The anomaly detection job ID
+ - name: record_timestamp
+ type: string
+ description: "Optional: ISO 8601 timestamp of the specific anomaly record to investigate"
+
+triggers:
+ - type: manual
+
+steps:
+ # Gate check 0a: Has the job processed enough data?
+ - name: get_job_stats
+ type: elasticsearch.request
+ with:
+ method: GET
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}/_stats
+
+ - name: gate_check
+ type: console
+ with:
+ message: |
+ GATE CHECKS for {{ inputs.job_id }}:
+ Job stats: {{ steps.get_job_stats.output | json:2 }}
+
+ CHECK 1 - Sufficient data:
+ Model needs >= 3 weeks for weekly seasonality, >= 2 full cycles.
+ Check data_counts.processed_record_count and earliest_record_timestamp.
+
+ CHECK 2 - Memory status:
+ If memory_status is soft_limit or hard_limit, the model is degraded.
+ Fix memory first before investigating scores.
+
+ CHECK 3 - Delayed data:
+ If the job has delayed data warnings, bucket scores may be based on
+ incomplete data. Check .ml-annotations-* for delayed data annotations.
+
+ # Step 1: Compare record_score vs initial_record_score
+ - name: get_anomaly_records
+ type: elasticsearch.search
+ with:
+ index: .ml-anomalies-*
+ size: 10
+ sort: "record_score:desc"
+ query:
+ bool:
+ filter:
+ - term:
+ job_id: "{{ inputs.job_id }}"
+ - term:
+ result_type: record
+
+ - name: score_comparison
+ type: console
+ with:
+ message: |
+ SCORE ANALYSIS for {{ inputs.job_id }}:
+ Top records: {{ steps.get_anomaly_records.output.hits.hits | json:2 }}
+
+ Compare initial_record_score vs record_score:
+ - If initial >> current: renormalization lowered the score after more
+ extreme anomalies appeared later. This is EXPECTED behavior.
+ - initial_record_score is the score at detection time.
+ - record_score is the current (renormalized) score.
+
+ # Step 2: Get job configuration for analysis
+ - name: get_job_config
+ type: elasticsearch.request
+ with:
+ method: GET
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}
+
+ - name: config_analysis
+ type: console
+ with:
+ message: |
+ JOB CONFIGURATION ANALYSIS:
+ Config: {{ steps.get_job_config.output | json:2 }}
+
+ Check these factors:
+ 1. bucket_span: Too large → dilutes anomalies. Too small → noisy.
+ 2. Detector function: mean vs high_mean vs low_mean affects directionality.
+ 3. Partition fields: High cardinality partitions split the model thin.
+ 4. custom_rules: May be suppressing valid anomalies.
+ 5. use_null: If false (default), missing entities produce no anomalies.
+
+ # Step 3-4: Score factor explanation
+ - name: score_education
+ type: console
+ with:
+ message: |
+ ANOMALY SCORE FACTORS (from anomaly_score_explanation):
+
+ 1. anomaly_length: How many consecutive buckets are anomalous.
+ Longer sequences → higher scores.
+
+ 2. single_bucket_impact: How extreme this single bucket is.
+ Driven by probability (lower p → higher impact).
+
+ 3. multi_bucket_impact: Positive (0-5) when anomaly spans multiple
+ buckets. Values >= 3 suggest genuine behavioral shift.
+
+ 4. anomaly_characteristics_impact: Nature of the anomaly (mean shift
+ vs variance change).
+
+ 5. high_variance_penalty: REDUCES score when the model's confidence
+ bounds are wide. Common early in model training or with noisy data.
+ Wide bounds → model is uncertain → anomalies appear less surprising.
+
+ 6. incomplete_bucket_penalty: REDUCES score when the bucket doesn't
+ have the expected amount of data (e.g., due to delayed data).
+
+ To see these factors, query the specific record from .ml-anomalies-*
+ and inspect the anomaly_score_explanation field.
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_wf_troubleshoot_memory_limit.yaml b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_wf_troubleshoot_memory_limit.yaml
new file mode 100644
index 0000000..f2bfa58
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_wf_troubleshoot_memory_limit.yaml
@@ -0,0 +1,165 @@
+name: ad_wf_troubleshoot_memory_limit
+description: >
+ Stored workflow for troubleshooting model_memory_limit issues (hard_limit/soft_limit). Implements a 7-step branching
+ decision tree: (1) identify memory status, (2) check downstream false alarms (hard_limit causing missing-doc
+ warnings), (3) analyze growth trend, (4) inspect model_size_stats thresholds, (5) compute principled estimate via
+ Estimate Model Memory API, (6) CCS-specific checks, (7) recommend action (increase limit / reduce data / restructure).
+ Trigger: 'My job hit memory limit', 'hard_limit', 'soft_limit'.
+enabled: true
+tags: ["anomaly-detection", "troubleshooting", "memory"]
+
+inputs:
+ - name: job_id
+ type: string
+ description: The anomaly detection job ID to troubleshoot
+
+triggers:
+ - type: manual
+
+steps:
+ # Step 1: Get current memory state
+ - name: get_stats
+ type: elasticsearch.request
+ with:
+ method: GET
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}/_stats
+
+ - name: get_config
+ type: elasticsearch.request
+ with:
+ method: GET
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}
+
+ - name: memory_status
+ type: console
+ with:
+ message: |
+ MEMORY STATUS for {{ inputs.job_id }}:
+ Stats: {{ steps.get_stats.output | json:2 }}
+
+ # Step 2: Check for downstream false alarms
+ - name: check_false_alarms
+ type: elasticsearch.search
+ with:
+ index: .ml-notifications-*
+ size: 10
+ sort: "timestamp:desc"
+ query:
+ bool:
+ filter:
+ - term:
+ job_id: "{{ inputs.job_id }}"
+ - terms:
+ level: ["warning", "error"]
+
+ - name: false_alarm_analysis
+ type: console
+ with:
+ message: |
+ DOWNSTREAM FALSE ALARM CHECK:
+ Recent warnings/errors: {{ steps.check_false_alarms.output.hits.hits | json:2 }}
+
+ If hard_limit AND you see "missing documents" warnings:
+ → These are SYMPTOMS of memory exhaustion, NOT ingest lag.
+ The model cannot track new entities, so events for unknown entities
+ are skipped, appearing as "missing documents".
+ Fix: Increase model_memory_limit first.
+
+ # Step 3: Analyze memory growth trend
+ - name: memory_trend
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /.ml-anomalies-*/_search
+ body:
+ size: 0
+ query:
+ bool:
+ filter:
+ - term:
+ job_id: "{{ inputs.job_id }}"
+ - term:
+ result_type: model_size_stats
+ aggs:
+ memory_over_time:
+ date_histogram:
+ field: timestamp
+ fixed_interval: 1d
+ aggs:
+ model_bytes:
+ max:
+ field: model_bytes
+ total_entities:
+ max:
+ field: total_by_field_count
+
+ - name: trend_analysis
+ type: console
+ with:
+ message: |
+ MEMORY GROWTH TREND:
+ {{ steps.memory_trend.output.aggregations | json:2 }}
+
+ Classify the trend:
+ - Stable plateau: Memory stabilized. Current limit may be appropriate
+ if status is ok, or barely insufficient if soft_limit.
+ - Linear growth: Entity cardinality is growing steadily (new hosts,
+ users, services). Will eventually hit limit.
+ - Exponential growth: Rapid cardinality explosion. Likely data or
+ config issue (unbounded by_field, logging explosion).
+
+ # Step 4-5: Inspect thresholds and estimate
+ - name: thresholds_and_estimate
+ type: console
+ with:
+ message: |
+ THRESHOLD INSPECTION:
+ From model_size_stats, check:
+ - total_by_field_count > 100K? → by_field cardinality too high
+ - total_partition_field_count > 10K? → partition explosion
+ - total_category_count > 10K? → categorization unbounded
+ Identify which field drives memory consumption.
+
+ PRINCIPLED ESTIMATION:
+ Run ad_estimate_memory_requirement workflow for this job.
+ It will:
+ 1. Sample cardinality from source data
+ 2. Call POST _ml/anomaly_detectors/_estimate_model_memory
+ 3. Compare estimate vs current limit vs actual usage
+
+ # Step 6-7: CCS checks and recommendations
+ - name: recommendations
+ type: console
+ with:
+ message: |
+ CCS CHECK:
+ If datafeed uses cross-cluster indices (remote_cluster:index pattern):
+ - Check remote cluster connectivity: GET _remote/info
+ - Note cardinality aggregation covers all clusters
+ - Re-run estimation if remote cluster recently added
+
+ RECOMMENDATIONS:
+
+ Branch A — Increase limit:
+ 1. Use the estimate from ad_estimate_memory_requirement, rounded up
+ 2. Fallback: max(estimate, peak_model_bytes * 1.3)
+ 3. Remediation sequence:
+ a) Stop datafeed: POST _ml/datafeeds/datafeed-{{ inputs.job_id }}/_stop
+ b) Close job: POST _ml/anomaly_detectors/{{ inputs.job_id }}/_close
+ c) Update: POST _ml/anomaly_detectors/{{ inputs.job_id }}/_update
+ { "analysis_limits": { "model_memory_limit": "NEW_VALUE" } }
+ d) Open job: POST _ml/anomaly_detectors/{{ inputs.job_id }}/_open
+ e) Start datafeed: POST _ml/datafeeds/datafeed-{{ inputs.job_id }}/_start
+ Use workflow: ad_update_model_memory_limit
+
+ Branch B — Reduce data:
+ - Filter datafeed query to exclude noisy/irrelevant data
+ - Reduce influencer count (each adds memory overhead)
+ - Partition into multiple focused jobs instead of one big job
+ - Exclude high-cardinality fields from analysis
+
+ Branch C — Extreme cases (multi-GB jobs):
+ - Architectural restructuring required
+ - Consider splitting by partition_field into separate jobs
+ - Use population analysis (over_field) instead of per-entity (by_field)
+ when possible — population uses less memory per entity
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_wf_troubleshoot_query_delay.yaml b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_wf_troubleshoot_query_delay.yaml
new file mode 100644
index 0000000..9ea1eda
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_wf_troubleshoot_query_delay.yaml
@@ -0,0 +1,131 @@
+name: ad_wf_troubleshoot_query_delay
+description: >
+ Stored workflow for troubleshooting missing documents and query_delay warnings. Implements a branching decision tree:
+ (1) checks for hard_limit categorization false alarms, (2) retrieves delayed data annotations, (3) checks bucket event
+ gaps, (4) measures ingest latency (two methods depending on available fields), (5) recommends query_delay value, (6)
+ suggests additional remediation. Trigger: 'My job reports missing documents', 'query_delay'.
+enabled: true
+tags: ["anomaly-detection", "troubleshooting", "query-delay"]
+
+inputs:
+ - name: job_id
+ type: string
+ description: The anomaly detection job ID to troubleshoot
+
+triggers:
+ - type: manual
+
+steps:
+ # Step 0: Get job/datafeed configuration
+ - name: get_config
+ type: elasticsearch.request
+ with:
+ method: GET
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}
+
+ - name: get_stats
+ type: elasticsearch.request
+ with:
+ method: GET
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}/_stats
+
+ # Step 1: Check for hard_limit categorization false alarm
+ # If job has categorization AND memory_status == hard_limit, the missing
+ # doc warning is a SYMPTOM of hard_limit, not ingest lag.
+ - name: log_hard_limit_check
+ type: console
+ with:
+ message: |
+ GATE CHECK: hard_limit categorization false alarm
+ Memory status: {{ steps.get_stats.output }}
+ If categorization_field_name is set AND memory_status == "hard_limit":
+ → STOP: Missing doc warning is caused by hard_limit, not ingest lag.
+ The per-partition categorizer cannot create new categories; events
+ are skipped, causing event_count mismatch.
+ Fix: Increase model_memory_limit first (use ad_update_model_memory_limit).
+
+ # Step 2: Check delayed data annotations
+ - name: check_delayed_annotations
+ type: elasticsearch.search
+ with:
+ index: .ml-annotations-*
+ size: 20
+ sort: "timestamp:desc"
+ query:
+ bool:
+ filter:
+ - term:
+ job_id: "{{ inputs.job_id }}"
+ - term:
+ type: annotation
+ - match:
+ annotation: "delayed data"
+
+ - name: log_delayed_annotations
+ type: console
+ with:
+ message: |
+ Delayed data annotations for {{ inputs.job_id }}:
+ Found {{ steps.check_delayed_annotations.output.hits.total.value }} delayed data annotations.
+ Recent annotations: {{ steps.check_delayed_annotations.output.hits.hits | json:2 }}
+
+ # Step 3: Check bucket event gaps
+ - name: check_bucket_gaps
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /.ml-anomalies-*/_search
+ body:
+ size: 0
+ query:
+ bool:
+ filter:
+ - term:
+ job_id: "{{ inputs.job_id }}"
+ - term:
+ result_type: bucket
+ aggs:
+ zero_count_buckets:
+ filter:
+ term:
+ event_count: 0
+ low_count_buckets:
+ filter:
+ range:
+ event_count:
+ gt: 0
+ lte: 5
+
+ - name: log_bucket_gaps
+ type: console
+ with:
+ message: |
+ Bucket event gap analysis for {{ inputs.job_id }}:
+ {{ steps.check_bucket_gaps.output.aggregations | json:2 }}
+
+ # Step 4-6: Recommendations
+ - name: recommendations
+ type: console
+ with:
+ message: |
+ RECOMMENDATIONS for {{ inputs.job_id }}:
+
+ 1. MEASURE INGEST LATENCY:
+ - If event.ingested field exists: compare event.ingested vs @timestamp
+ to get exact latency distribution (P50, P95, P99)
+ - If not: use date_histogram comparison (recent bucket doc count vs
+ same bucket queried later) to estimate stabilization window
+
+ 2. SET QUERY_DELAY:
+ - Recommended: P95 ingest latency + 30s buffer
+ - Trade-off: larger query_delay = slower anomaly alerts
+ - To update: stop datafeed → update query_delay → restart datafeed
+ - Use workflow: ad_update_datafeed_query_delay
+
+ 3. ADDITIONAL REMEDIATION:
+ a) Add ingest timestamp via ingest pipeline for future diagnosis:
+ PUT _ingest/pipeline/add-ingest-ts
+ with a "set" processor that copies the _ingest.timestamp into "event.ingested"
+ b) Consider using ingest timestamp as the job's time_field
+ c) If catastrophically late data: revert model snapshot + backfill
+ d) Note: missed documents warning may persist 24h after fix
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_wf_ts_field_cardinality.yaml b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_wf_ts_field_cardinality.yaml
new file mode 100644
index 0000000..5f93f12
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/kibana/workflows/ad_wf_ts_field_cardinality.yaml
@@ -0,0 +1,50 @@
+name: ad_wf_ts_field_cardinality
+description: >
+ Estimate cardinality of a split field (by_field, over_field, partition_field) in SOURCE data via ES|QL
+ POST /_query. The column for COUNT_DISTINCT is interpolated into the query text (not a ? parameter - those bind only
+ literals). Pass split_field_esql as a valid ES|QL column reference (for example service.keyword or `host.name.keyword`)
+ from ad_get_job_datafeed_config. Compare distinct_count with total_*_count from ad_ts_model_memory_health; if source
+ cardinality is much larger, entities may be dropped. For CCS, run per cluster and sum. Prefer ad_estimate_memory_requirement
+ for full sizing; this workflow answers how many distinct values the field has in the window.
+enabled: true
+tags: ["anomaly-detection", "diagnostics"]
+
+inputs:
+ - name: source_index
+ type: string
+ description: >
+ Source index name or LIKE pattern (* wildcard). From the job datafeed config (same as ES|QL FROM * METADATA _index filter).
+ - name: split_field_esql
+ type: string
+ description: >
+ Exact ES|QL column expression for COUNT_DISTINCT (not quoted as a string). Examples: service.keyword, `host.name.keyword`.
+ Must match a field on matched documents; use only values taken from job analysis config to avoid query injection.
+ - name: start_time
+ type: string
+ description: Start of time range in ISO 8601 format
+ - name: end_time
+ type: string
+ description: End of time range in ISO 8601 format
+
+triggers:
+ - type: manual
+
+steps:
+ - name: cardinality_esql
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_query
+ body:
+ query: "FROM * METADATA _index | WHERE _index LIKE ?source_index AND @timestamp >= ?start_time AND @timestamp <= ?end_time | STATS distinct_count = COUNT_DISTINCT({{ inputs.split_field_esql }}) | LIMIT 1"
+ params:
+ source_index: "{{ inputs.source_index }}"
+ start_time: "{{ inputs.start_time }}"
+ end_time: "{{ inputs.end_time }}"
+
+ - name: result
+ type: console
+ with:
+ message: |
+ Split-field cardinality ({{ inputs.split_field_esql }}) on indices matching {{ inputs.source_index }} ({{ inputs.start_time }}-{{ inputs.end_time }}):
+ {{ steps.cardinality_esql.output | json:2 }}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/observability-anomaly-expert.md b/plugins/kibana/skills/kibana-anomaly-detection/references/observability-anomaly-expert.md
new file mode 100644
index 0000000..bb69981
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/observability-anomaly-expert.md
@@ -0,0 +1,104 @@
+# Observability / SRE framing — Elastic ML anomaly detection
+
+**Role:** Treat Elasticsearch ML anomaly detection as a reliability signal for SRE and platform work: degradation,
+incident scope, and capacity decisions. Combine the **Investigate**, **Explain**, and **Troubleshoot** modes of the
+parent skill, biased toward reliability interpretation.
+
+---
+
+## Reliability-first interpretation
+
+Interpret anomalies through three reliability lenses:
+
+1. **Incident detection** — is this active degradation? What is the scope?
+2. **Change attribution** — tie signal to deployments, config changes, dependencies when possible.
+3. **Capacity signals** — separate acute incidents from resource-exhaustion trajectories.
+
+## Signal → reliability mapping
+
+| Anomaly pattern | Reliability interpretation | Action |
+| -------------------------------------------------------------- | -------------------------------------------------- | ------------------------------------ |
+| Latency spike + error rate spike (same service, same time) | Service degradation in progress | Incident response |
+| Throughput drop (`actual << typical` with `count`/`low_count`) | Service unavailable or upstream dependency failure | Check dependencies, circuit breakers |
+| Cross-service entity anomalies with temporal chain | Cascading failure / blast propagation | Identify blast radius, isolate |
+| Memory/CPU creep (`multi_bucket_impact ≥ 3`) | Resource exhaustion trajectory | Capacity intervention before OOM |
+| Anomaly onset matches deployment timestamp | Deployment regression | Rollback candidate |
+| Single service anomaly, no related job co-firing | Isolated issue, contained | Service-level investigation |
+| Anomaly during known maintenance window | Expected — suppress via calendar event | `ad_create_calendar_event` |
+
+## SRE investigation protocol
+
+### Phase 1 — Incident scoping (Investigate mode)
+
+1. `ad_get_available_metadata` — identify observability jobs (latency, error rate, throughput, saturation, request
+ count).
+2. `ad_query_anomaly_timeline` (`job_id_pattern='*'`) — establish incident start time and breadth.
+3. `ad_rca_multi_job_entities` (`min_job_count=2`) — co-firing metrics on the same entity = the degraded service.
+4. `ad_rca_blast_radius` — scope: which downstream services are affected.
+
+### Phase 2 — Root cause attribution (Investigate mode)
+
+1. `ad_discover_jobs_by_datafeed_index` — find all jobs monitoring the same infrastructure layer.
+2. `ad_rca_cross_job_entity_match` — confirm which services are actively co-firing.
+3. `ad_rca_correlation` sorted by timestamp — leading metric (first anomaly) = root cause; lagging = symptoms.
+4. `ad_rca_detector_fingerprint` — characterize failure type (latency? saturation? error rate? throughput drop?).
+
+### Phase 3 — Evidence and context (Investigate mode)
+
+1. `ad_get_job_datafeed_config` → source index → `ad_rca_source_evidence` — actual metric values and dimensions.
+2. `ad_query_influencers` — which specific service instances, pods, or hosts are contributing.
+3. `ad_rca_entity_profile` — full behavioral history for the suspect service/host.
+
+### Phase 4 — Deployment regression check (Explain mode)
+
+When incident onset aligns with a recent deployment:
+
+1. `ad_rca_score_reassessment` — confirm whether a score drop reflects renormalization instead of real recovery.
+2. `ad_get_model_plot` — confirm the anomaly sits outside expected bounds instead of being a model artifact.
+3. `ad_rca_source_evidence` — compare metric values before and after the deployment timestamp.
+
+### Phase 5 — Capacity planning (Explain + Troubleshoot modes)
+
+For sustained `multi_bucket_impact ≥ 3` anomalies that look like trajectories instead of spikes:
+
+1. `ad_ts_model_memory_health` — confirm ML memory pressure is not degrading detections.
+2. `ad_query_anomaly_records` filtered to `multi_bucket_impact ≥ 3` — extract resource saturation trends.
+3. `ad_estimate_memory_requirement` — size memory for expanded infrastructure.
+
+### Phase 6 — Maintenance suppression (Troubleshoot mode)
+
+For planned deployments or maintenance windows:
+
+1. `ad_create_calendar_event` — suppress false positives, reduce alert fatigue, protect model health.
+
+## Reliability-specific rules
+
+- **`multi_bucket_impact ≥ 3`** is the primary capacity signal: sustained shifts indicate trajectory. These need
+ capacity planning, not just incident response.
+- **`actual << typical` with throughput detectors** = service unavailability. Treat as SEV-1 until proven otherwise.
+- **Temporal ordering matters**: in cascading failures, the first anomaly timestamp points to root cause, not the
+ highest score.
+- **`initial_record_score >> record_score`**: renormalization — the score dropped because a worse event occurred later.
+ Do not interpret as "resolved." Use Explain mode to communicate this to stakeholders.
+- **All jobs firing simultaneously**: shared infrastructure layer (database, message bus, shared network path).
+ Investigate shared dependencies first.
+- **`ad_validate_ml_tool_permissions`**: run as a preflight when tool calls fail unexpectedly.
+
+## Job health before trusting signals
+
+- `ad_ts_model_memory_health` — a job at `hard_limit` stops learning new entities (new pods/services), risking missed
+ anomalies for those entities.
+- `ad_ts_delayed_data_annotations` — delayed data delays alerts. Raise `query_delay` toward P95 ingest latency + buffer
+ when the pipeline is slow; otherwise expect missed real-time detection.
+- `ad_create_calendar_event` — add maintenance windows to suppress false positives during planned deployments.
+
+## Escalation decision framework
+
+| Signal | SRE action |
+| ------------------------------------------------------- | ------------------------------------------- |
+| Multi-job co-fire + blast radius > 1 service | Declare incident, page on-call |
+| Leading metric identified + deployment timestamp match | Rollback candidate — page owning team |
+| Sustained `multi_bucket_impact ≥ 3` + resource detector | Capacity review, no immediate incident |
+| Single-job anomaly + no downstream impact | Service-level investigation, no incident |
+| Anomaly during known maintenance | Add calendar event, dismiss |
+| Score drop only (renormalization) | Use Explain mode to communicate — no action |
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/permissions-matrix.md b/plugins/kibana/skills/kibana-anomaly-detection/references/permissions-matrix.md
new file mode 100644
index 0000000..4a46869
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/permissions-matrix.md
@@ -0,0 +1,144 @@
+# Permissions Matrix
+
+Maps each tool to the Elasticsearch and Kibana privileges it requires.
+
+Run `ad_validate_ml_tool_permissions` as a preflight check on core `.ml-*` indices (see workflow description). Confirm
+**source data index** privileges separately before previews, `ad_wf_ts_field_cardinality`, or `ad_rca_source_evidence`.
+
+---
+
+## Required Index Privileges
+
+| Index | Privilege | Purpose |
+| --------------------- | --------------------- | ----------------------------------------------------------------------------- |
+| `.ml-anomalies-*` | `read` | Query anomaly records, buckets, influencers, category definitions |
+| `.ml-anomalies-*` | `view_index_metadata` | ESQL `FROM` and metadata access on anomaly indices |
+| `.ml-config` | `read` | Query job and datafeed configurations |
+| `.ml-config` | `view_index_metadata` | ESQL `FROM` against the config index |
+| `.ml-annotations-*` | `read` | Query delayed data annotations |
+| `.ml-annotations-*` | `view_index_metadata` | ESQL `FROM` and metadata access on annotation indices |
+| `.ml-notifications-*` | `read` | Query job messages and notifications |
+| `.ml-notifications-*` | `view_index_metadata` | ESQL `FROM` and metadata access on notification indices |
+| Source data indices | `read` | `ad_rca_source_evidence`, `ad_search_log_category_examples`, cardinality aggs |
+| Source data indices | `view_index_metadata` | ESQL `FROM` with `METADATA _index` on source data |
+
+---
+
+## Tool → Permission Matrix
+
+### ES|QL Tools (Read-only)
+
+| Tool | `.ml-anomalies-*` | `.ml-config` | `.ml-annotations-*` | Source indices |
+| --------------------------------- | ----------------- | --------------- | ------------------- | --------------- |
+| `ad_get_available_metadata` | — | read + metadata | — | — |
+| `ad_get_jobs` | — | read + metadata | — | — |
+| `ad_discover_related_jobs` | — | read + metadata | — | — |
+| `ad_query_anomaly_records` | read + metadata | — | — | — |
+| `ad_query_anomaly_timeline` | read + metadata | — | — | — |
+| `ad_query_influencers` | read + metadata | — | — | — |
+| `ad_rca_multi_job_entities` | read + metadata | — | — | — |
+| `ad_rca_cross_job_entity_match` | read + metadata | — | — | — |
+| `ad_rca_detector_fingerprint` | read + metadata | — | — | — |
+| `ad_rca_correlation` | read + metadata | — | — | — |
+| `ad_rca_blast_radius` | read + metadata | — | — | — |
+| `ad_rca_entity_profile` | read + metadata | — | — | — |
+| `ad_rca_source_evidence` | — | — | — | read + metadata |
+| `ad_rca_score_reassessment` | read + metadata | — | — | — |
+| `ad_get_categories` | read + metadata | — | — | — |
+| `ad_search_log_category_examples` | — | — | — | read + metadata |
+| `ad_get_job_messages` | — | — | — | — |
+| `ad_get_model_snapshots` | read + metadata | — | — | — |
+| `ad_get_model_plot` | read + metadata | — | — | — |
+| `ad_get_forecast_results` | read + metadata | — | — | — |
+| `ad_ts_delayed_data_annotations` | — | — | read + metadata | — |
+| `ad_ts_bucket_event_gaps` | read + metadata | — | — | — |
+| `ad_ts_ingest_latency_estimate` | — | — | — | read |
+| `ad_ts_model_memory_health` | read + metadata | — | — | — |
+
+### Workflow Tools
+
+| Tool | ML API privilege | Source indices | Notes |
+| ------------------------------------- | -------------------------- | -------------- | --------------------------------------------------------------- |
+| `ad_get_job_datafeed_config` | `monitor_ml` | — | Reads job config and stats |
+| `ad_discover_jobs_by_datafeed_index` | `monitor_ml` | — | Reads job config + .ml-config search |
+| `ad_manage_datafeed` | `manage_ml` | — | Start/stop requires write privilege |
+| `ad_preview_datafeed_with_latency` | `monitor_ml` | read | Preview requires source index read |
+| `ad_update_datafeed_query_delay` | `manage_ml` | — | Write operation |
+| `ad_update_delayed_data_check_config` | `manage_ml` | — | Write operation |
+| `ad_estimate_memory_requirement` | `monitor_ml` | read | Cardinality aggs on source indices |
+| `ad_wf_ts_field_cardinality` | `monitor_ml` | read | POST /\_query COUNT_DISTINCT on source split field |
+| `ad_update_model_memory_limit` | `manage_ml` | — | Write operation; job must be closed |
+| `ad_revert_model_snapshot` | `manage_ml` | — | Write operation; job must be closed |
+| `ad_get_calendar_events` | `monitor_ml` | — | — |
+| `ad_create_calendar_event` | `manage_ml` | — | Write operation |
+| `ad_create_job` | `manage_ml` | — | Write operation |
+| `ad_validate_ml_tool_permissions` | `monitor` (cluster) | — | Uses `_security` API |
+| `ad_ts_ccs_diagnostics` | `monitor` (cluster) | — | Uses `_remote/info`, `_cluster/health` |
+| `ad_wf_troubleshoot_anomaly_score` | `monitor_ml` | — | Read-only workflow |
+| `ad_wf_troubleshoot_memory_limit` | `monitor_ml` + `manage_ml` | read | Includes estimate (read) and optional update (write) |
+| `ad_wf_troubleshoot_query_delay` | `monitor_ml` + `manage_ml` | read | Includes latency measurement (read) and optional update (write) |
+
+---
+
+## Privilege Definitions
+
+| Privilege | Scope | What it allows |
+| ----------------------------- | ------- | -------------------------------------------------------------------------------------- | ---------------------- |
+| `read` (index) | Index | Search, GET, ES | QL FROM |
+| `view_index_metadata` (index) | Index | Required for ES | QL FROM and field caps |
+| `monitor_ml` (cluster) | Cluster | GET job configs, stats, snapshots, calendar events |
+| `manage_ml` (cluster) | Cluster | Create/update/delete jobs, datafeeds, calendars; start/stop datafeed; revert snapshots |
+| `monitor` (cluster) | Cluster | Cluster health, remote info, security authenticate |
+
+---
+
+## Minimum Role for Read-only Investigation
+
+```json
+{
+ "cluster": ["monitor_ml"],
+ "indices": [
+ {
+ "names": [".ml-anomalies-*", ".ml-config", ".ml-annotations-*", ".ml-notifications-*"],
+ "privileges": ["read", "view_index_metadata"]
+ },
+ {
+ "names": [""],
+ "privileges": ["read", "view_index_metadata"]
+ }
+ ]
+}
+```
+
+## Minimum Role for Full Remediation
+
+```json
+{
+ "cluster": ["monitor_ml", "manage_ml", "monitor"],
+ "indices": [
+ {
+ "names": [".ml-anomalies-*", ".ml-config", ".ml-annotations-*", ".ml-notifications-*"],
+ "privileges": ["read", "view_index_metadata"]
+ },
+ {
+ "names": [""],
+ "privileges": ["read", "view_index_metadata"]
+ }
+ ]
+}
+```
+
+---
+
+## Troubleshooting Permission Errors
+
+| Symptom | Likely missing privilege |
+| ---------------------------------------------- | ---------------------------------------------- | ------------------------------------------ |
+| ES | QL FROM `.ml-anomalies-*` returns no results | `view_index_metadata` on `.ml-anomalies-*` |
+| `ad_rca_source_evidence` returns empty | `read` on source data indices |
+| Workflow tools return 403 | `monitor_ml` or `manage_ml` cluster privilege |
+| `ad_validate_ml_tool_permissions` fails | `monitor` cluster privilege |
+| `ad_ts_delayed_data_annotations` returns empty | `read` on `.ml-annotations-*` |
+| Job config missing from `ad_get_jobs` | `read` + `view_index_metadata` on `.ml-config` |
+
+Run `ad_validate_ml_tool_permissions` to get a definitive list of which specific privileges are missing.
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/protocols/investigation.md b/plugins/kibana/skills/kibana-anomaly-detection/references/protocols/investigation.md
new file mode 100644
index 0000000..0a29b5d
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/protocols/investigation.md
@@ -0,0 +1,152 @@
+# Investigation Protocol (14 Steps)
+
+Canonical workflow for root cause analysis of Elastic ML anomaly detection events.
+
+> For a worked example, see [../worked-example.md](../worked-example.md).
+
+---
+
+## When to Use This Protocol
+
+- Starting from a single alert and need to determine the root cause
+- Multiple jobs are co-firing and you need to find the common denominator
+- Asked "what broke?", "which entity caused this?", or "why is service X slow?"
+
+For **score explanation questions** (why is my score low/high?), see [../score-reference.md](../score-reference.md)
+instead.
+
+---
+
+## Three-Layer Job Discovery
+
+Before beginning analysis, identify all related jobs using these signals in priority order:
+
+1. **Shared datafeed index patterns** (strongest) — Jobs reading from the same source indices monitor the same
+ underlying system.
+2. **Shared entity field names** (config signal) — Jobs that split by the same field names analyze the same entity
+ dimensions.
+3. **Shared entity values in results** (active incident) — During an active incident, find all jobs where a specific
+ entity is currently co-firing.
+
+---
+
+## The 14 Steps
+
+### Phase 1: Discovery
+
+**Step 1 — Discover** Call `ad_get_available_metadata` to learn available jobs, fields, and functions. Always start here
+when jobs are unknown.
+
+**Step 2 — Find related jobs** Use `ad_discover_jobs_by_datafeed_index` with the job of interest — it retrieves that
+job's `datafeed_config.indices`, then finds all other jobs sharing the same source index. Also use
+`ad_discover_related_jobs` to find jobs sharing entity field names (partition/by/over). Fallback: compare
+`datafeed_config.indices` manually via `ad_get_jobs`.
+
+**Step 3 — Scope** Use `ad_query_anomaly_timeline` with `job_id_pattern` set to the related job group (e.g.,
+`rcaeval-*`) or `*` for all jobs. Identify the incident time window and count of affected jobs. Cross-job composite
+scores reveal coordinated events.
+
+---
+
+### Phase 2: Entity Attribution
+
+**Step 4 — Expand from alert** Extract entity values from the alert (`partition_field_value`, `by_field_value`,
+`over_field_value`). Use `ad_rca_cross_job_entity_match` to find all related jobs with anomalies for that entity. Note
+`first_anomaly` per job for chronology reconstruction.
+
+**Step 5 — Multi-job entities** Use `ad_rca_multi_job_entities` with `min_job_count=2`. Entities anomalous in 2+ jobs
+simultaneously are the strongest root cause signal — they are prime suspects. Single-job entities are likely downstream
+victims.
+
+> Resource faults (CPU, memory, disk) affect multiple metrics → multi-job. Network faults (packet loss) affect latency
+> but not resource metrics → single-job.
+
+**Step 6 — Fingerprint** Use `ad_rca_detector_fingerprint` with the related job group as `job_id_pattern`. Understand
+which system aspects are anomalous: CPU? Latency? Error rate? Memory? The combination of anomalous detectors
+characterizes the fault type.
+
+---
+
+### Phase 3: Deep Analysis
+
+**Step 7 — Drill down per job** Use `ad_query_anomaly_records` with an exact `job_id_pattern` to examine a specific
+job's anomalies in detail, without cross-job noise.
+
+**Step 8 — Attribute** Use `ad_query_influencers` with the related job group as `job_id_pattern` and a low `min_score`
+(25) for shared influencer discovery. Filter for `job_count > 1` to surface entities that are influencers in multiple
+co-firing jobs — the common denominator.
+
+**Step 9 — Profile** Use `ad_rca_entity_profile` to build a complete dossier on the suspect entity: all anomalies across
+all jobs and field types, sorted by timestamp.
+
+**Step 10 — Characterize** Examine `multi_bucket_impact` in results:
+
+- `≥ 3` → sustained behavioral shift (system change), not a transient spike
+- `0–2` → isolated event (one-off anomaly)
+
+---
+
+### Phase 4: Root Cause Confirmation
+
+**Step 11 — Cascade** Use `ad_rca_correlation` sorted by timestamp. The job with the **earliest anomaly** for the
+suspect entity points toward the root cause. Reconstruct chronology: which metric became anomalous first?
+
+**Step 12 — Evidence** Get the source index from `ad_get_job_datafeed_config`, then call `ad_rca_source_evidence` to
+retrieve raw source documents. This shows the actual values that triggered the anomaly at the point of ingestion.
+
+**Step 13 — Log categories** _(only when `by_field_name == "mlcategory"`)_ For log categorization jobs:
+
+1. `ad_get_categories` → find the category matching the anomaly's `by_field_value` (category ID). Examine its terms,
+ regex, and examples.
+2. `ad_search_log_category_examples` twice — once for a **baseline window** (24h before anomaly), once for the **anomaly
+ window**.
+3. Compare: look for changed field values in the variable parts of the log structure (IPs, hostnames, error codes,
+ paths, credentials).
+4. Cross-reference changed entities with influencers from other related jobs to confirm root cause.
+
+---
+
+### Phase 5: Synthesis
+
+**Step 14 — Synthesize** Present findings as a structured RCA report:
+
+| Section | Content |
+| ------------------------ | ------------------------------------------------------------- |
+| **Root cause entity** | The entity (host, service, user) responsible |
+| **Affected systems** | Which jobs/metrics were impacted |
+| **Temporal progression** | Which metric became anomalous first (from Step 11) |
+| **Fault type** | Resource (CPU/memory/disk) / Network / Application / Pipeline |
+| **Severity** | `record_score` range, `multi_bucket_impact`, duration |
+| **Recommended actions** | Remediation steps |
+
+---
+
+## Quick Reference: Tool → Step Mapping
+
+| Tool | Step |
+| ------------------------------------ | ---- |
+| `ad_get_available_metadata` | 1 |
+| `ad_discover_jobs_by_datafeed_index` | 2 |
+| `ad_discover_related_jobs` | 2 |
+| `ad_query_anomaly_timeline` | 3 |
+| `ad_rca_cross_job_entity_match` | 4 |
+| `ad_rca_multi_job_entities` | 5 |
+| `ad_rca_detector_fingerprint` | 6 |
+| `ad_query_anomaly_records` | 7 |
+| `ad_query_influencers` | 8 |
+| `ad_rca_entity_profile` | 9 |
+| `ad_rca_correlation` | 11 |
+| `ad_get_job_datafeed_config` | 12 |
+| `ad_rca_source_evidence` | 12 |
+| `ad_get_categories` | 13 |
+| `ad_search_log_category_examples` | 13 |
+
+---
+
+## Key Decision Rules
+
+- **Low scores across many jobs** > one high score — composite cross-job signal often indicates systemic root cause.
+- **`actual << typical` with count/low_count** → absence/outage, not just a numerically low value.
+- **Entities in 2+ jobs** → prime suspects (resource fault or systemic failure).
+- **Entities in only 1 job** → likely downstream victims or surface-level effects.
+- **`first_anomaly` chronology** → the earliest metric to become anomalous is closest to the root cause.
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/score-reference.md b/plugins/kibana/skills/kibana-anomaly-detection/references/score-reference.md
new file mode 100644
index 0000000..c409573
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/score-reference.md
@@ -0,0 +1,120 @@
+# Anomaly Score Reference
+
+Canonical definitions for all score and impact fields in Elastic ML anomaly detection.
+
+---
+
+## Score Types
+
+| Field | Scope | Range | Description |
+| ----------------------- | --------------------- | ----- | -------------------------------------------------------------------------------------- |
+| `record_score` | Single anomaly record | 0–100 | Current normalized severity. May change over time as the model sees more extreme data. |
+| `initial_record_score` | Single anomaly record | 0–100 | Score at detection time — never changes. Use for alerting on fresh anomalies. |
+| `anomaly_score` | Bucket (time window) | 0–100 | Aggregate severity across all detectors in a bucket. |
+| `initial_anomaly_score` | Bucket | 0–100 | Bucket score at detection time — never changes. |
+| `influencer_score` | Entity × bucket | 0–100 | How anomalous a specific entity (host, user, service) is within that bucket. |
+
+---
+
+## `record_score` Severity Bands
+
+| Band | Range | Interpretation |
+| ------------- | ----- | ---------------------------------------------------------- |
+| Critical | > 75 | High-confidence anomaly; warrants immediate investigation |
+| Warning | 50–75 | Notable deviation; triage and correlate with other signals |
+| Minor | 25–50 | Potentially interesting; aggregate with cross-job signals |
+| Informational | < 25 | Weak signal; useful for context, not standalone action |
+
+> **Cross-job composite signal**: Low scores (25–50) across many jobs simultaneously are often more significant than a
+> single high score. Five jobs each scoring 30 = composite signal 150, pointing to a systemic root cause.
+
+---
+
+## `multi_bucket_impact`
+
+Scale from -5 to +5 indicating whether the anomaly spans multiple consecutive time buckets.
+
+| Value | Meaning |
+| -------- | ---------------------------------------------------- |
+| 0 | One-off event, no sustained pattern |
+| 1–2 | Mild persistence across a few buckets |
+| ≥ 3 | Genuine behavioral shift — not a transient spike |
+| Negative | Anomaly is suppressed by surrounding normal behavior |
+
+Values ≥ 3 strongly suggest a real system change (e.g., a resource exhaustion event that persists) rather than a
+momentary blip.
+
+---
+
+## `initial_record_score` vs `record_score`
+
+Elasticsearch continuously renormalizes scores relative to the most extreme anomaly ever seen by the job. A score of 90
+today may become 60 if a more extreme event appears later — by design, so the "worst ever" event always scores near 100.
+
+**When to use each:**
+
+| Use case | Field |
+| ------------------------------------------------------------ | ------------------------------------------------------------------------------------ |
+| Alerting on newly detected anomalies | `initial_record_score` — captures severity at detection time |
+| Ranking historical anomalies by current importance | `record_score` — reflects how bad this was relative to all history |
+| Detecting renormalization (model calibrated away an anomaly) | Compare: if `initial_record_score >> record_score`, the model saw worse events later |
+
+**Quantify drift:** `score_drift = initial_record_score - record_score`
+
+- Large positive drift = renormalized away (model calibrated)
+- Small drift = score is stable and genuine
+
+---
+
+## `anomaly_score_explanation` Components
+
+When available, this field explains the factors that contributed to the final score.
+
+| Component | Effect on score | What it means |
+| -------------------------------- | --------------- | ------------------------------------------------------------------- |
+| `anomaly_length` | ↑ increases | More consecutive anomalous buckets — sustained deviation |
+| `single_bucket_impact` | ↑ increases | Lower statistical probability → more surprising → higher impact |
+| `multi_bucket_impact` | ↑ increases | Contribution from sustained pattern across multiple buckets |
+| `anomaly_characteristics_impact` | ↑ increases | Mean shift (value moved) vs. variance change (volatility increased) |
+| `high_variance_penalty` | ↓ decreases | Historically noisy data; wide confidence bounds absorb the spike |
+| `incomplete_bucket_penalty` | ↓ decreases | Bucket has less data than expected (ingest lag, sparse events) |
+
+---
+
+## Absence Anomalies
+
+When `actual << typical` with `count`, `low_count`, `low_mean`, or `low_sum` functions, a low or zero value indicates a
+real-world absence — not just a numerically low observation:
+
+- Zero `count` when traffic is normally constant → pipeline stopped, service unavailable
+- `low_mean(response_time)` → requests completing too fast (cache hit storm, bypassed processing)
+- Very low `sum(bytes_sent)` → network partition or data source failure
+
+**Key insight:** A `record_score` of 80 with `actual = 0` and `typical = 5000` is an outage signal, not just a low
+number.
+
+---
+
+## Why a Score Is Unexpectedly Low
+
+1. **`high_variance_penalty`** — Metric is historically noisy; wide model bounds absorb the spike.
+2. **Renormalization** — A more extreme anomaly appeared later, pushing this score down.
+3. **Insufficient training** — Model needs ≥ 3 weeks for weekly seasonality, ≥ 2 full cycles for any period.
+4. **`bucket_span` too large** — Long span smooths short-duration spikes; use smaller span for high-frequency detection.
+5. **Detector function mismatch** — `mean` vs `high_mean`, `count` vs `high_count`. Wrong function = missed direction.
+6. **`incomplete_bucket_penalty`** — Ingest latency or sparse events reduced bucket data volume.
+7. **`custom_rules`** — A detector filter may be suppressing or conditioning the anomaly.
+
+## Why a Score Is Unexpectedly High
+
+1. **Insufficient training history** — Early training: moderate deviations flag as extreme.
+2. **High-cardinality split** — Too few data points per entity per bucket → unreliable probabilities.
+3. **`use_null: true`** — Missing entities produce "null" anomalies that may not be operationally meaningful.
+
+---
+
+## See Also
+
+- [anomaly-detection-functions.md](anomaly-detection-functions.md) — Function selection guide
+- [protocols/investigation.md](protocols/investigation.md) — 14-step investigation workflow
+- [worked-example.md](worked-example.md) — End-to-end investigation walkthrough
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/security-anomaly-expert.md b/plugins/kibana/skills/kibana-anomaly-detection/references/security-anomaly-expert.md
new file mode 100644
index 0000000..4e0f75f
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/security-anomaly-expert.md
@@ -0,0 +1,103 @@
+# Security framing — Elastic ML anomaly detection
+
+**Role:** Treat Elasticsearch ML anomaly detection as a behavioral threat-detection surface. Assume unusual behavior is
+suspicious until benign intent is proven. Combine the **Investigate**, **Explain**, and **Troubleshoot** modes of the
+parent skill, biased toward attack-first interpretation.
+
+---
+
+## Threat-first interpretation
+
+Treat operational monitoring as benign-first; treat security anomalies as attack-first. Then:
+
+1. Map behavioral deviations to known attack patterns.
+2. Reconstruct attacker chains from cross-job signals.
+3. Separate attacker behavior from benign operational noise.
+4. Classify threats with MITRE ATT&CK context.
+
+## Signal mapping
+
+| Anomaly pattern | Threat hypothesis | MITRE tactic |
+| -------------------------------------------------------------- | ----------------------------------------------- | ------------------------------- |
+| Unusual auth failures for a user/host | Brute force, credential stuffing | Credential Access (TA0006) |
+| `actual << typical` with `low_count` on auth/process | Service stop, log clearing, defense evasion | Defense Evasion (TA0005) |
+| New/rare entity (first-seen IP, user, process) | Initial access, new implant, new C2 | Initial Access (TA0001) |
+| Entity anomalous in multiple jobs simultaneously | Active compromise, lateral movement in progress | Lateral Movement (TA0008) |
+| Unusual data volume (bytes_out spike) | Data exfiltration | Exfiltration (TA0010) |
+| Rare process execution (high influencer_score on process name) | Malware execution, living-off-the-land | Execution (TA0002) |
+| Auth success following prior auth failures | Successful credential compromise | Credential Access → Persistence |
+| Privilege escalation patterns (sudo, admin role changes) | Admin abuse, shadow IT, misconfiguration | Privilege Escalation (TA0004) |
+| Regular low-volume network spikes (beaconing) | C2 communication | Command & Control (TA0011) |
+
+## Investigation questions
+
+For each anomalous entity, determine:
+
+1. **Known vs first-seen entity** — treat first-seen entities as higher risk.
+2. **Blast radius** — count how many jobs or systems co-fire.
+3. **Temporal chain** — treat auth failure → auth success → lateral movement as a compromise chain hypothesis.
+4. **Source evidence** — treat raw logs as the ground truth.
+5. **MITRE mapping** — map the pattern to the closest tactic and technique.
+
+## Investigation protocol
+
+### Phase 1 — Triage (Investigate mode)
+
+1. `ad_get_available_metadata` — identify security-relevant jobs (auth, network, process, DNS, endpoint).
+2. `ad_query_anomaly_timeline` — establish incident time window.
+3. `ad_rca_multi_job_entities` (`min_job_count=2`) — multi-job entities in security = active threat actors.
+
+### Phase 2 — Entity attribution (Investigate mode)
+
+1. `ad_rca_cross_job_entity_match` — expand from single alert to full entity activity chain.
+2. `ad_query_influencers` (low `min_score`, broad `job_id_pattern`) — surface all associated entities.
+3. `ad_rca_entity_profile` — complete behavioral dossier on suspect user/host/IP.
+
+### Phase 3 — Attack chain reconstruction (Investigate mode)
+
+1. `ad_rca_correlation` sorted by timestamp — reconstruct chronological order. First anomaly = entry point hypothesis.
+2. `ad_rca_blast_radius` — determine lateral spread: how many systems/accounts affected.
+3. `ad_rca_detector_fingerprint` — which behavioral dimensions are anomalous (auth? process? network? data volume?).
+
+### Phase 4 — Evidence collection (Investigate mode)
+
+1. `ad_get_job_datafeed_config` → source index → `ad_rca_source_evidence` — raw forensic ground truth.
+2. For log categorization jobs: `ad_get_categories` + `ad_search_log_category_examples` — compare baseline vs. incident
+ window for changed IPs, credentials, command-line arguments, file paths.
+
+### Phase 5 — Score validation (Explain mode)
+
+1. If score seems low for a suspicious pattern: `ad_rca_score_reassessment` — check renormalization drift.
+ `initial_record_score` may reveal a threat that was renormalized away.
+2. `ad_get_model_plot` — confirm actual exceeds model bounds.
+
+### Phase 6 — Threat report
+
+- **Threat classification**: attack type + MITRE ATT&CK tactic/technique
+- **Confidence**: High/Medium/Low with reasoning
+- **Affected entities**: users, hosts, IPs, processes
+- **Attack timeline**: reconstructed from `first_anomaly` per job
+- **Evidence summary**: key anomalous values from source documents
+- **Recommended response**: containment, investigation, tuning actions
+
+## Security-specific rules
+
+- **Absence anomalies are high priority**: `actual << typical` on auth or process jobs = log clearing or service killing
+ = defense evasion.
+- **`initial_record_score >> record_score`**: do not dismiss. Score was renormalized after a more extreme event — the
+ original anomaly is still a valid threat indicator.
+- **Low scores across many jobs > one high score**: sophisticated attackers stay below single-job thresholds. Composite
+ cross-job signals are the primary detection mechanism.
+- **New entities**: first-seen host or user is higher priority than a known entity with a moderate score.
+- **`multi_bucket_impact ≥ 3`**: sustained shift = persistent access, beaconing, or ongoing exfiltration.
+- Run `ad_validate_ml_tool_permissions` if tools fail — permission errors are common in multi-tenant security
+ environments.
+
+## Escalation vs. tuning
+
+| Signal | Action |
+| ------------------------------------------------ | ------------------------------------------------- |
+| Multi-job entity + source evidence + MITRE match | Escalate as confirmed threat |
+| Multi-job entity + no source evidence | Escalate for manual log review |
+| Single-job, explainable by operational event | Document and tune (calendar event or custom rule) |
+| Renormalization-only score drop | Explain to stakeholder — use Explain mode |
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/tools.md b/plugins/kibana/skills/kibana-anomaly-detection/references/tools.md
new file mode 100644
index 0000000..95c7773
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/tools.md
@@ -0,0 +1,145 @@
+# Tool Reference — Elastic Anomaly Detection Agent Builder
+
+## ES|QL Tools (24) — read-only, query `.ml-anomalies-*` and `.ml-config`
+
+### Discovery & Metadata
+
+| Tool | Description |
+| --------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| `ad_get_available_metadata` | Discover all jobs, fields, functions, entity fields, bucket_spans. **Call first** when jobs are unknown. Returns a single summary row from `.ml-config`. |
+| `ad_get_jobs` | List all jobs with full config: bucket_span, detectors, field names, memory limit, groups, description. |
+| `ad_discover_related_jobs` | Find groups of jobs sharing `partition_field_name`, `by_field_name`, or `over_field_name`. |
+
+### Anomaly Records & Influencers
+
+| Tool | Params | Description |
+| --------------------------- | ----------------------------------------------- | ------------------------------------------------------------------------------------------ |
+| `ad_query_anomaly_records` | job_id_pattern, min_score, start_time, end_time | Search anomaly records cross-job or scoped. Use `*` for overview, exact ID for drill-down. |
+| `ad_query_anomaly_timeline` | job_id_pattern, start_time, end_time | Bucket timeline / composite signal across jobs. |
+| `ad_query_influencers` | job_id_pattern, min_score, start_time, end_time | Find most anomalous entities. Filter `job_count > 1` for cross-job shared influencers. |
+
+### RCA Tools
+
+| Tool | Description |
+| ------------------------------- | --------------------------------------------------------------------------------------------------------------------------- |
+| `ad_rca_multi_job_entities` | Entities anomalous in multiple jobs (min_job_count=2). Prime root-cause signal. |
+| `ad_rca_cross_job_entity_match` | All jobs where a specific entity value appears anomalous right now. Returns per-job first_anomaly timestamp for chronology. |
+| `ad_rca_detector_fingerprint` | Detector-level incident fingerprint — what aspects of the system are anomalous (CPU? latency?). |
+| `ad_rca_correlation` | Temporal correlation / cascade. Sort by timestamp — earliest anomaly for the entity hints at root cause. |
+| `ad_rca_blast_radius` | Blast radius across jobs for an entity/time window. |
+| `ad_rca_entity_profile` | Complete dossier on a suspect entity. |
+| `ad_rca_source_evidence` | Raw source documents from the original data index. Get index from `ad_get_job_datafeed_config`. |
+| `ad_rca_score_reassessment` | Score drift (`score_drift = initial_record_score - record_score`). Quantify renormalization. |
+
+### Log Categorization
+
+| Tool | Description |
+| --------------------------------- | ------------------------------------------------------------------------------------------------------- |
+| `ad_get_categories` | Category definitions (terms, regex, examples) for categorization jobs. |
+| `ad_search_log_category_examples` | Log samples for a category in a time window. Run twice (baseline + anomaly) and compare variable parts. |
+
+### Model Insight
+
+| Tool | Description |
+| ------------------------- | --------------------------------------------------------------------------------------------------------------- |
+| `ad_get_model_plot` | Model upper/lower/median bounds over time. Most visual explanation for non-technical users. |
+| `ad_get_forecast_results` | Forecast predictions with upper/lower bounds for capacity planning. |
+| `ad_get_model_snapshots` | Available model snapshots for a job. Used before `ad_revert_model_snapshot`. |
+| `ad_get_job_messages` | All notifications from `.ml-notifications-*`: datafeed warnings, delayed data, memory limits, lifecycle events. |
+
+### Time-Series Diagnostics
+
+| Tool | Params | Description |
+| -------------------------------- | ------------------------------------- | --------------------------------------------------------------------------------------------------------- |
+| `ad_ts_model_memory_health` | job_id, limit (1=snapshot, 500=trend) | Memory status time series: model_bytes, limit, memory_status, entity counts. |
+| `ad_ts_ingest_latency_estimate` | source_index, start_time, end_time | P50/P95/P99 ingest latency. Requires `event.ingested` field. If P95 > query_delay → data will be lost. |
+| `ad_ts_bucket_event_gaps` | job_id, start_time, end_time | Buckets with zero or low event counts. Correlate with delayed data annotations. |
+| `ad_ts_delayed_data_annotations` | job_id | All delayed data annotations from `.ml-annotations-*`. Starting point for missing-document investigation. |
+
+---
+
+## Workflow Tools (23 YAML files) — REST API + YAML, management and remediation
+
+### Discovery & Config
+
+| Tool | Description |
+| ------------------------------------ | ------------------------------------------------------------------------------------------------------------------- |
+| `ad_discover_jobs_by_datafeed_index` | Jobs sharing datafeed source indices — strongest related-job signal. Iterates job's indices, queries `.ml-config`. |
+| `ad_get_job_datafeed_config` | Full job + datafeed config: detectors, fields, bucket_span, query_delay, delayed_data_check_config, source indices. |
+
+### Datafeed Operations
+
+| Tool | Description |
+| ---------------------------------- | ------------------------------------------------------------------------------------------- |
+| `ad_manage_datafeed` | Start or stop a datafeed (`_start` / `_stop`). Preview: `ad_preview_datafeed_with_latency`. |
+| `ad_preview_datafeed_with_latency` | Preview datafeed source payload and measure effective latency before tuning query_delay. |
+
+### Config Updates
+
+| Tool | Description |
+| ------------------------------------- | ----------------------------------------------------------------------------------------------------------------------- |
+| `ad_update_datafeed_query_delay` | Update `query_delay`. Stop datafeed first. Set to P95 ingest latency + buffer. |
+| `ad_update_delayed_data_check_config` | Enable/disable delayed data checks or adjust `check_window`. |
+| `ad_update_model_memory_limit` | Update `model_memory_limit`. Requires stop/close/update/open/start sequence. Cannot decrease below current model_bytes. |
+
+### Sizing & Estimation
+
+| Tool | Description |
+| -------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| `ad_wf_ts_field_cardinality` | ES | QL `POST /_query`: COUNT_DISTINCT on a source split field; column name is spliced into the query (`?` params are literals only). Use values from job analysis config only. |
+| `ad_estimate_memory_requirement` | Principled memory sizing: auto-samples cardinality from source data → calls Estimate Model Memory API. Better than `peak_model_bytes * 1.3` (ignores pure influencer memory). |
+
+### Permissions & CCS
+
+| Tool | Description |
+| --------------------------------- | --------------------------------------------------------------------------------------------------------------------- |
+| `ad_validate_ml_tool_permissions` | Preflight: read + view_index_metadata on `.ml-anomalies-*`, `.ml-config`, `.ml-annotations-*`, `.ml-notifications-*`. |
+| `ad_ts_ccs_diagnostics` | CCS diagnostics: remote cluster connectivity, latency, error rates for cross-cluster datafeeds. |
+
+### Lifecycle & Recovery
+
+| Tool | Description |
+| -------------------------- | -------------------------------------------------------------- |
+| `ad_revert_model_snapshot` | Revert to a previous model snapshot. Job must be closed first. |
+| `ad_validate_job_spec` | POST `/_ml/anomaly_detectors/_validate` — validate job JSON |
+| `ad_create_job` | PUT `/_ml/anomaly_detectors/{job_id}` — create job |
+| `ad_create_datafeed` | PUT `/_ml/datafeeds/{datafeed_id}` — create datafeed |
+| `ad_open_job` | POST `/_ml/anomaly_detectors/{job_id}/_open` |
+
+### Calendars
+
+| Tool | Description |
+| -------------------------- | ------------------------------------------------------------------------ |
+| `ad_get_calendar_events` | Get scheduled events from calendars (maintenance windows, holidays). |
+| `ad_create_calendar_event` | Add a scheduled event to suppress false positives during known downtime. |
+
+### Stored Troubleshooting Workflows (decision trees from support runbooks)
+
+| Tool | Trigger |
+| ---------------------------------- | ----------------------------------------------------------------------------- |
+| `ad_wf_troubleshoot_anomaly_score` | "Why is my score low?", "Expected anomaly not detected", "Score too high/low" |
+| `ad_wf_troubleshoot_query_delay` | "My job reports missing documents", "Datafeed has missed X documents" |
+| `ad_wf_troubleshoot_memory_limit` | "My job hit memory limit", "hard_limit", "soft_limit" |
+
+---
+
+## Key System Indices
+
+| Index | `result_type` values |
+| --------------------- | ----------------------------------------------------------------------------------------------------------------------------- |
+| `.ml-anomalies-*` | `bucket`, `record`, `influencer`, `model_size_stats`, `model_plot`, `model_forecast`, `model_snapshot`, `category_definition` |
+| `.ml-annotations-*` | delayed data (`event == "delayed_data"`) |
+| `.ml-notifications-*` | job messages (datafeed warnings, memory limits, lifecycle) |
+| `.ml-config` | job/datafeed documents (used for discovery — all jobs visible, even never-run) |
+
+## Registration
+
+Requires Node.js 18+. Defaults to `elastic`/`changeme` when no credentials supplied.
+
+```bash
+cd skills/kibana/kibana-anomaly-detection
+node scripts/kibana-agent-builder.mjs all register --kibana-url http://localhost:5601
+```
+
+Workflow tools are skipped automatically until Elastic Workflows (preview) is enabled. Configure exclusions in
+`scripts/agent_builder_constants.json`.
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/troubleshoot-anomaly-tool-reference.md b/plugins/kibana/skills/kibana-anomaly-detection/references/troubleshoot-anomaly-tool-reference.md
new file mode 100644
index 0000000..5787d8d
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/troubleshoot-anomaly-tool-reference.md
@@ -0,0 +1,431 @@
+# Troubleshoot mode — tool reference
+
+ES|QL and workflow tool parameters, REST fallbacks, and decision trees for the **Troubleshoot** mode of the parent
+[SKILL.md](../SKILL.md).
+
+## ES|QL Tools (8)
+
+### `ad_get_available_metadata`
+
+Discover all jobs and metadata. **Call first.**
+
+```esql
+FROM .ml-config
+| WHERE job_type == "anomaly_detector"
+| STATS job_count = COUNT(*), job_ids = VALUES(job_id),
+ functions = VALUES(`analysis_config.detectors.function`),
+ bucket_spans = VALUES(`analysis_config.bucket_span`)
+```
+
+_No parameters._
+
+---
+
+### `ad_get_jobs`
+
+List all jobs with config, memory limit, and state context.
+
+```esql
+FROM .ml-config
+| WHERE job_type == "anomaly_detector"
+| KEEP job_id, `analysis_config.bucket_span`, `analysis_config.detectors.function`,
+ `analysis_config.detectors.partition_field_name`,
+ `analysis_config.detectors.by_field_name`,
+ `analysis_limits.model_memory_limit`, groups, description
+| SORT job_id ASC | LIMIT 100
+```
+
+_No parameters._
+
+---
+
+### `ad_get_job_messages`
+
+All notifications from `.ml-notifications-*`: datafeed warnings, delayed data, memory limits, lifecycle events, errors.
+
+| Parameter | Type | Description |
+| --------- | ---- | ----------- |
+| `job_id` | text | Job ID |
+
+```esql
+FROM .ml-notifications-*
+| WHERE job_id == ?job_id
+| SORT timestamp DESC
+| KEEP timestamp, level, message, node_name, job_id
+| LIMIT 50
+```
+
+---
+
+### `ad_get_model_snapshots`
+
+Available model snapshots. Review before using `ad_revert_model_snapshot`.
+
+| Parameter | Type | Description |
+| --------- | ---- | ----------- |
+| `job_id` | text | Job ID |
+
+```esql
+FROM .ml-anomalies-*
+| WHERE result_type == "model_snapshot" AND job_id == ?job_id
+| SORT timestamp DESC
+| KEEP job_id, timestamp, description, snapshot_doc_count
+| LIMIT 20
+```
+
+---
+
+### `ad_ts_model_memory_health`
+
+Memory status time series. `limit=1` for current snapshot; `limit=500` for trend and trajectory.
+
+**Interpretation:** `hard_limit` = CRITICAL (job blind to new entities). `soft_limit` = WARNING (pruning).
+`model_bytes / model_bytes_memory_limit > 0.8` = APPROACHING LIMIT.
+
+| Parameter | Type | Description |
+| --------- | ------- | ----------------------------- |
+| `job_id` | text | Job ID |
+| `limit` | integer | 1 = current, 500 = full trend |
+
+```esql
+FROM .ml-anomalies-*
+| WHERE result_type == "model_size_stats" AND job_id == ?job_id
+| SORT timestamp DESC
+| KEEP job_id, timestamp, model_bytes, peak_model_bytes, model_bytes_memory_limit,
+ model_bytes_exceeded, memory_status,
+ total_by_field_count, total_over_field_count, total_partition_field_count,
+ bucket_allocation_failures_count
+| LIMIT ?limit
+```
+
+---
+
+### `ad_ts_ingest_latency_estimate`
+
+Measure actual ingest latency via `event.ingested`. If P95 > `query_delay` → data is being lost.
+
+| Parameter | Type | Description |
+| -------------- | ---- | --------------------------------- |
+| `source_index` | text | LIKE pattern from datafeed config |
+| `start_time` | text | ISO 8601 |
+| `end_time` | text | ISO 8601 |
+
+```esql
+FROM * METADATA _index
+| WHERE _index LIKE ?source_index
+ AND @timestamp >= ?start_time AND @timestamp <= ?end_time
+ AND event.ingested IS NOT NULL
+| EVAL latency_seconds = DATE_DIFF("second", @timestamp, event.ingested)
+| STATS p50_latency = PERCENTILE(latency_seconds, 50),
+ p95_latency = PERCENTILE(latency_seconds, 95),
+ p99_latency = PERCENTILE(latency_seconds, 99),
+ max_latency = MAX(latency_seconds), doc_count = COUNT(*)
+| LIMIT 1
+```
+
+---
+
+### `ad_ts_bucket_event_gaps`
+
+Buckets with zero or low event counts. Correlate with delayed data annotations to confirm false positives from missing
+data.
+
+| Parameter | Type | Description |
+| ------------ | ---- | ----------- |
+| `job_id` | text | Job ID |
+| `start_time` | text | ISO 8601 |
+| `end_time` | text | ISO 8601 |
+
+```esql
+FROM .ml-anomalies-*
+| WHERE result_type == "bucket"
+ AND job_id == ?job_id
+ AND timestamp >= ?start_time AND timestamp <= ?end_time
+| SORT timestamp ASC
+| KEEP timestamp, event_count, anomaly_score, bucket_span, is_interim
+| LIMIT 500
+```
+
+---
+
+### `ad_ts_delayed_data_annotations`
+
+Delayed data annotations — starting point for any missing documents investigation.
+
+| Parameter | Type | Description |
+| --------- | ---- | ----------- |
+| `job_id` | text | Job ID |
+
+```esql
+FROM .ml-annotations-*
+| WHERE job_id == ?job_id AND event == "delayed_data"
+| SORT timestamp DESC
+| KEEP job_id, timestamp, end_timestamp, annotation
+| LIMIT 100
+```
+
+---
+
+## Workflow Tools (15)
+
+### `ad_get_job_datafeed_config`
+
+Full job + datafeed config: `bucket_span`, `query_delay`, `delayed_data_check_config`, source indices. Essential for all
+troubleshooting and memory estimation.
+
+| Parameter | Type | Required |
+| --------- | ------ | -------- |
+| `job_id` | string | yes |
+
+```text
+Step 1: GET _ml/anomaly_detectors/{job_id}
+Step 2: GET _ml/anomaly_detectors/{job_id}/_stats
+```
+
+---
+
+### `ad_manage_datafeed`
+
+Start or stop a datafeed via `POST _ml/datafeeds/{datafeed_id}/{_start|_stop}`. For preview, use
+`ad_preview_datafeed_with_latency` (`GET _ml/datafeeds/{datafeed_id}/_preview`).
+
+| Parameter | Type | Required | Description |
+| ------------- | ------ | -------- | --------------------------- |
+| `datafeed_id` | string | yes | Usually `datafeed-{job_id}` |
+| `action` | string | yes | `_start` or `_stop` only |
+
+```text
+POST _ml/datafeeds/{datafeed_id}/{action}
+```
+
+---
+
+### `ad_preview_datafeed_with_latency`
+
+Preview datafeed payload and measure effective latency before tuning query_delay.
+
+| Parameter | Type | Required |
+| ------------- | ------ | -------- |
+| `datafeed_id` | string | yes |
+
+```text
+GET _ml/datafeeds/{datafeed_id}/_preview
+```
+
+---
+
+### `ad_update_datafeed_query_delay`
+
+Update `query_delay`. **Stop datafeed first.** Set to P95 ingest latency + buffer.
+
+| Parameter | Type | Required | Description |
+| ----------------- | ------ | -------- | ----------------- |
+| `datafeed_id` | string | yes | |
+| `new_query_delay` | string | yes | e.g. `3m`, `120s` |
+
+```text
+POST _ml/datafeeds/{datafeed_id}/_update
+Body: { "query_delay": "{new_query_delay}" }
+```
+
+---
+
+### `ad_update_delayed_data_check_config`
+
+Enable/disable delayed data checks or adjust `check_window`.
+
+| Parameter | Type | Required | Description |
+| -------------- | ------- | -------- | --------------------------------------------------------------------- |
+| `job_id` | string | yes | |
+| `enabled` | boolean | yes | |
+| `check_window` | string | no | e.g. `2h`; omitted from the update body when empty or whitespace-only |
+
+```text
+Step 1: data.parseJson — build update body (enabled as JSON boolean; check_window only if non-blank after trim)
+Step 2: POST _ml/anomaly_detectors/{job_id}/_update with that body
+```
+
+---
+
+### `ad_update_model_memory_limit`
+
+Update `model_memory_limit`. **Cannot decrease below current `model_bytes`** — clone job to shrink. Requires
+stop/close/update/open/start sequence.
+
+| Parameter | Type | Required | Description |
+| ----------- | ------ | -------- | ------------------- |
+| `job_id` | string | yes | |
+| `new_limit` | string | yes | e.g. `256mb`, `1gb` |
+
+```text
+POST _ml/anomaly_detectors/{job_id}/_update
+Body: { "analysis_limits": { "model_memory_limit": "{new_limit}" } }
+```
+
+---
+
+### `ad_estimate_memory_requirement`
+
+**Best practice for memory sizing.** Auto-samples cardinality from source → calls
+`POST _ml/anomaly_detectors/_estimate_model_memory`. More accurate than `peak_model_bytes * 1.3`.
+
+| Parameter | Type | Required | Description |
+| -------------- | ------ | -------- | --------------------------------- |
+| `job_id` | string | yes | |
+| `sample_start` | string | no | ISO 8601, defaults to 30 days ago |
+| `sample_end` | string | no | ISO 8601, defaults to now |
+
+```text
+Step 1: GET _ml/anomaly_detectors/{job_id}
+Step 2: Extract split fields and pure influencers
+Step 3: POST {datafeed_indices}/_search?size=0 → overall_cardinality per field
+Step 4: POST {datafeed_indices}/_search?size=0 → max_bucket_cardinality
+Step 5: POST _ml/anomaly_detectors/_estimate_model_memory
+Step 6: Compare estimate vs model_memory_limit vs model_bytes
+Step 7: Recommend (increase / reduce data / restructure)
+```
+
+---
+
+### `ad_wf_ts_field_cardinality`
+
+Cardinality of a split field in **source** data. If source cardinality >> the model's `total_*_count` from
+`ad_ts_model_memory_health`, entities may be dropped.
+
+This is a **workflow** (not an ES|QL Agent Builder tool): it runs `POST /_query` and **splices** `split_field_esql` into
+the query text as the `COUNT_DISTINCT` column. ES|QL `?` parameters bind **literals only**, so field names cannot be
+passed as `?` parameters.
+
+| Parameter | Type | Required | Description |
+| ------------------ | ------ | -------- | ------------------------------------------------------------------------------------------------ |
+| `source_index` | string | yes | LIKE pattern from datafeed config |
+| `split_field_esql` | string | yes | Column expression for `COUNT_DISTINCT` (for example `service.keyword`); from job analysis config |
+| `start_time` | string | yes | ISO 8601 |
+| `end_time` | string | yes | ISO 8601 |
+
+```text
+POST /_query
+Body includes "query" with COUNT_DISTINCT() and "params" for ?source_index, ?start_time, ?end_time
+```
+
+---
+
+### `ad_validate_ml_tool_permissions`
+
+Preflight check on core `.ml-*` indices (`read` + `view_index_metadata` on `.ml-anomalies-*`, `.ml-config`,
+`.ml-annotations-*`, `.ml-notifications-*`). Does **not** assert privileges on job source data indices — check those
+before previews or `ad_rca_source_evidence`.
+
+_No parameters._
+
+```text
+Step 1: GET _security/_authenticate
+Step 2: POST _security/user/_has_privileges
+ (.ml-anomalies-*, .ml-config, .ml-annotations-*, .ml-notifications-*)
+```
+
+---
+
+### `ad_ts_ccs_diagnostics`
+
+Cross-cluster search diagnostics: remote cluster connectivity, latency, error rates.
+
+_No parameters._
+
+```text
+Step 1: GET _remote/info
+Step 2: GET _cluster/health
+```
+
+---
+
+### `ad_revert_model_snapshot`
+
+Revert to a previous model snapshot to "unlearn" bad data. Job must be **closed** first. After reverting, reopen and
+restart datafeed from the snapshot timestamp.
+
+| Parameter | Type | Required |
+| ------------- | ------ | -------- |
+| `job_id` | string | yes |
+| `snapshot_id` | string | yes |
+
+```text
+POST _ml/anomaly_detectors/{job_id}/model_snapshots/{snapshot_id}/_revert
+```
+
+---
+
+### `ad_get_calendar_events`
+
+Get scheduled events from a calendar (maintenance windows, holidays).
+
+| Parameter | Type | Required |
+| ------------- | ------ | -------- |
+| `calendar_id` | string | yes |
+
+```text
+GET _ml/calendars/{calendar_id}/events
+```
+
+---
+
+### `ad_create_calendar_event`
+
+Add a scheduled event to suppress false positives during known downtime.
+
+| Parameter | Type | Required | Description |
+| ------------- | ------ | -------- | --------------------------------- |
+| `calendar_id` | string | yes | |
+| `description` | string | yes | e.g. `Planned maintenance window` |
+| `start_time` | string | yes | ISO 8601 or epoch_millis |
+| `end_time` | string | yes | ISO 8601 or epoch_millis |
+
+```text
+POST _ml/calendars/{calendar_id}/events
+Body: { "events": [{ "description": ..., "start_time": ..., "end_time": ... }] }
+```
+
+---
+
+### `ad_wf_troubleshoot_query_delay`
+
+Full automated decision tree for missing documents diagnosis.
+
+| Parameter | Type | Required |
+| --------- | ------ | -------- |
+| `job_id` | string | yes |
+
+Decision tree:
+
+1. Retrieve config + `memory_status`.
+2. **Gate:** Categorization job + `memory_status == hard_limit` → EXIT EARLY. Missing-doc warning is a false alarm from
+ memory exhaustion. Fix memory first.
+3. `ad_ts_delayed_data_annotations` — frequency, severity, affected ranges.
+4. `ad_ts_bucket_event_gaps` — zero/low event count buckets.
+5. `ad_ts_ingest_latency_estimate` — P50/P95/P99. If P95 > `query_delay` → data lost.
+6. Recommend `query_delay` = P95 + buffer.
+7. Additional remediation: add ingest pipeline, use ingest timestamp as `time_field`, revert snapshot + backfill.
+
+---
+
+### `ad_wf_troubleshoot_memory_limit`
+
+Full automated decision tree for memory limit diagnosis.
+
+| Parameter | Type | Required |
+| --------- | ------ | -------- |
+| `job_id` | string | yes |
+
+Decision tree:
+
+1. `ad_ts_model_memory_health` (limit=1): classify as `ok` / `soft_limit` / `hard_limit`.
+2. If `hard_limit`: check notifications for "missed documents" warnings — these are SYMPTOMS of hard_limit, not ingest
+ lag.
+3. `ad_ts_model_memory_health` (limit=500): stable / linear / exponential growth. Predict time-to-limit.
+4. Inspect `model_size_stats`: `total_by_field_count > 100K`, `total_partition_field_count > 10K`,
+ `total_category_count > 10K` → identify dominant driver.
+5. `ad_estimate_memory_requirement`: principled sizing.
+6. Recommend: **A** — Increase limit; **B** — Reduce data (filter datafeed, reduce influencers); **C** — Multi-GB
+ architectural restructuring.
+
+---
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/worked-example.md b/plugins/kibana/skills/kibana-anomaly-detection/references/worked-example.md
new file mode 100644
index 0000000..3219870
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/worked-example.md
@@ -0,0 +1,263 @@
+# Worked Investigation Example
+
+End-to-end walkthrough of a multi-job anomaly investigation using the [14-step protocol](protocols/investigation.md).
+
+---
+
+## Scenario
+
+**Alert received:** `rcaeval-ob-cpu` — `partition_field_value: frontend` — `record_score: 72` — `2024-03-15T14:30:00Z`
+
+The on-call engineer receives this alert and needs to determine: Is `frontend` the root cause, or a victim of something
+upstream?
+
+---
+
+## Phase 1: Discovery
+
+### Step 1 — Discover available jobs
+
+Call `ad_get_available_metadata` (no parameters).
+
+**Result summary:**
+
+```text
+job_count: 6
+job_ids: [rcaeval-ob-cpu, rcaeval-ob-latency, rcaeval-ob-memory,
+ rcaeval-ob-errors, rcaeval-nw-throughput, rcaeval-logs-app]
+functions: [mean, high_mean, count, rare]
+partition_fields: [service]
+bucket_spans: [5m]
+```
+
+**Interpretation:** 6 jobs all partitioned by `service`. All likely monitor the same system from different angles.
+
+---
+
+### Step 2 — Find related jobs
+
+Call `ad_discover_jobs_by_datafeed_index` with `job_id: rcaeval-ob-cpu`.
+
+**Result:**
+
+```text
+Jobs sharing index rcaeval-re1-ob:
+ rcaeval-ob-cpu, rcaeval-ob-latency, rcaeval-ob-memory, rcaeval-ob-errors
+ match_count: 4
+
+Jobs sharing index rcaeval-re1-nw:
+ rcaeval-nw-throughput
+ match_count: 1
+```
+
+Then call `ad_discover_related_jobs` with `job_id: rcaeval-ob-cpu`.
+
+**Result:**
+
+```text
+entity_field: service
+ jobs: [rcaeval-ob-cpu, rcaeval-ob-latency, rcaeval-ob-memory,
+ rcaeval-ob-errors, rcaeval-nw-throughput]
+ job_count: 5
+```
+
+**Interpretation:** 5 jobs all split by `service`. The `rcaeval-logs-app` job uses `mlcategory` (log categorization),
+not `service`. The 5 observability jobs are our related group.
+
+---
+
+### Step 3 — Scope the incident
+
+Call `ad_query_anomaly_timeline` with:
+
+- `job_id_pattern: rcaeval-*`
+- `min_score: 25`
+- `start_time: 2024-03-15T13:00:00Z`
+- `end_time: 2024-03-15T16:00:00Z`
+
+**Result:**
+
+```text
+timestamp max_score job_count composite_score jobs
+2024-03-15T14:00:00Z 48 2 76 [rcaeval-ob-cpu, rcaeval-ob-latency]
+2024-03-15T14:15:00Z 61 3 138 [rcaeval-ob-cpu, rcaeval-ob-latency, rcaeval-ob-memory]
+2024-03-15T14:30:00Z 78 4 198 [rcaeval-ob-cpu, rcaeval-ob-latency, rcaeval-ob-memory, rcaeval-ob-errors]
+2024-03-15T14:45:00Z 72 4 201 [rcaeval-ob-cpu, rcaeval-ob-latency, rcaeval-ob-memory, rcaeval-ob-errors]
+2024-03-15T15:00:00Z 45 3 112 [rcaeval-ob-latency, rcaeval-ob-memory, rcaeval-ob-errors]
+```
+
+**Interpretation:** Peak at 14:30 with 4 jobs co-firing (composite score 198). Incident started ~14:00, peak 14:30,
+declining by 15:00. This is a 1-hour incident, not a transient spike. CPU was first to fire (14:00).
+
+---
+
+## Phase 2: Entity Attribution
+
+### Step 4 — Expand from alert
+
+Call `ad_rca_cross_job_entity_match` with:
+
+- `entity_value: frontend`
+- `min_score: 10`
+- `start_time: 2024-03-15T13:30:00Z`
+- `end_time: 2024-03-15T15:30:00Z`
+
+**Result:**
+
+```text
+job_id max_score anomaly_count first_anomaly functions
+rcaeval-ob-cpu 78 8 2024-03-15T14:00:00Z [high_mean]
+rcaeval-ob-latency 74 7 2024-03-15T14:05:00Z [high_mean]
+rcaeval-ob-memory 61 5 2024-03-15T14:15:00Z [high_mean]
+rcaeval-ob-errors 52 4 2024-03-15T14:20:00Z [high_count]
+```
+
+**Interpretation:** `frontend` is anomalous in 4 jobs. CPU was first (14:00), then latency (14:05), then memory (14:15),
+then errors (14:20). This cascading pattern suggests CPU is upstream — high CPU caused latency, which caused memory
+pressure, which caused errors.
+
+---
+
+### Step 5 — Multi-job entities
+
+Call `ad_rca_multi_job_entities` with:
+
+- `min_score: 25`
+- `min_job_count: 2`
+- `start_time: 2024-03-15T14:00:00Z`
+- `end_time: 2024-03-15T15:00:00Z`
+
+**Result:**
+
+```text
+partition_field_value job_count max_score functions
+frontend 4 78 [high_mean, high_count]
+backend 1 31 [high_mean]
+```
+
+**Interpretation:** `frontend` is the only entity anomalous in 4+ jobs — confirmed prime suspect. `backend` appears in
+only 1 job at score 31 — likely incidental.
+
+---
+
+### Step 6 — Fingerprint
+
+Call `ad_rca_detector_fingerprint` with:
+
+- `job_id_pattern: rcaeval-ob-*`
+- `min_score: 25`
+- `start_time: 2024-03-15T14:00:00Z`
+- `end_time: 2024-03-15T15:00:00Z`
+
+**Result:**
+
+```text
+job_id function field_name max_score count
+rcaeval-ob-cpu high_mean system.cpu.percent 78 8
+rcaeval-ob-latency high_mean http.response_time 74 7
+rcaeval-ob-memory high_mean system.memory.used 61 5
+rcaeval-ob-errors high_count http.error_count 52 4
+```
+
+**Interpretation:** Classic resource exhaustion pattern: CPU spike → latency increase → memory pressure → errors. All
+metrics elevated on `frontend` simultaneously.
+
+---
+
+## Phase 3: Deep Analysis
+
+### Steps 7–10 — Drill down, attribute, profile, characterize
+
+`ad_query_anomaly_records` for `rcaeval-ob-cpu` with `min_score: 50`:
+
+```text
+record_score: 78
+actual: 94.7 (% CPU)
+typical: 31.2
+multi_bucket_impact: 4
+initial_record_score: 79
+```
+
+`multi_bucket_impact: 4` → sustained shift across 4+ buckets, not a transient spike. `initial ≈ current` → no
+renormalization; score is stable and genuine.
+
+---
+
+## Phase 4: Root Cause Confirmation
+
+### Step 11 — Cascade
+
+Call `ad_rca_correlation` with:
+
+- `job_id_pattern: rcaeval-ob-*`
+- `min_score: 25`
+- Window: 13:30–15:30
+
+**Result (sorted by timestamp):**
+
+```text
+timestamp job_id record_score function field_name
+2024-03-15T14:00:00Z rcaeval-ob-cpu 78 high_mean cpu.percent → FIRST
+2024-03-15T14:05:00Z rcaeval-ob-latency 74 high_mean response_time
+2024-03-15T14:15:00Z rcaeval-ob-memory 61 high_mean memory.used
+2024-03-15T14:20:00Z rcaeval-ob-errors 52 high_count error_count
+```
+
+**Interpretation:** CPU anomaly at 14:00 is the earliest — this is the root cause signal.
+
+### Step 12 — Evidence
+
+Get source index from `ad_get_job_datafeed_config` → `rcaeval-re1-ob`.
+
+Call `ad_rca_source_evidence`:
+
+- `source_index: rcaeval-re1-ob`
+- Window: 13:50–14:10
+
+**Sample raw docs (14:00):**
+
+```json
+{"service": "frontend", "system.cpu.percent": 96.2, "event": "metricset", "@timestamp": "2024-03-15T14:00:34Z"}
+{"service": "frontend", "system.cpu.percent": 93.8, "event": "metricset", "@timestamp": "2024-03-15T14:01:05Z"}
+{"service": "frontend", "system.cpu.percent": 94.1, "event": "metricset", "@timestamp": "2024-03-15T14:01:41Z"}
+```
+
+**Interpretation:** CPU is genuinely at ~95% on `frontend` starting at 14:00, confirming the ML anomaly matches reality.
+
+---
+
+## Step 14 — RCA Report
+
+**Root cause:** `frontend` service experienced CPU exhaustion starting at 2024-03-15T14:00Z.
+
+**Evidence:**
+
+- CPU rose from baseline 31% to 94–97% at 14:00 (score: 78, `multi_bucket_impact: 4` — sustained shift)
+- CPU anomaly preceded all other anomalies by 5–20 minutes
+- `frontend` was the only entity anomalous in 4 jobs simultaneously — no other service implicated
+
+**Cascading impact:**
+
+1. **14:00** — CPU saturates (94%+)
+2. **14:05** — HTTP latency climbs (CPU-bound request processing)
+3. **14:15** — Memory rises (queued/retried requests consuming heap)
+4. **14:20** — Error rate spikes (timeouts from downstream callers)
+
+**Fault type:** Resource exhaustion — CPU-bound processing on `frontend` pod(s)
+
+**Recommended actions:**
+
+1. Check `frontend` deployment for runaway process or CPU-hungry code path (profiling)
+2. Review recent deploys to `frontend` in the 30 min before 14:00
+3. Horizontal scale or vertical CPU limit increase as immediate mitigation
+4. Add CPU throttling alert at 80% to catch before next saturation
+
+**Severity:** Score 78, `multi_bucket_impact: 4`, duration ~1 hour — **high severity**
+
+---
+
+## See Also
+
+- [protocols/investigation.md](protocols/investigation.md) — Full 14-step protocol
+- [score-reference.md](score-reference.md) — Score field definitions and severity bands
+- [anomaly-detection-functions.md](anomaly-detection-functions.md) — Function selection guide
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/references/workflow-tools.md b/plugins/kibana/skills/kibana-anomaly-detection/references/workflow-tools.md
new file mode 100644
index 0000000..7118945
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/references/workflow-tools.md
@@ -0,0 +1,536 @@
+# Workflow Tools Reference
+
+Full documentation for all workflow tools. Most call Elasticsearch ML HTTP APIs (REST) via the Kibana Agent Builder. A
+few run ES|QL via `POST /_query` when a column identifier must be spliced into the query text (ES|QL `?` parameters bind
+literals only, not field names).
+
+> For ES|QL read tools, see the skill SKILL.md files. For permissions required, see
+> [permissions-matrix.md](permissions-matrix.md).
+
+---
+
+## Availability Note
+
+Workflow tools require **Kibana Elastic Agent Workflows** to be enabled. Not all deployments have this feature. If a
+workflow tool is unavailable, the description includes a manual fallback — prefer ES|QL queries first, then REST API
+calls if ES|QL cannot express the operation.
+
+---
+
+## Job & Datafeed Configuration
+
+### `ad_get_job_datafeed_config`
+
+Fetch complete job and datafeed configuration in one call.
+
+| Parameter | Type | Required | Description |
+| --------- | ------ | -------- | ---------------------------- |
+| `job_id` | string | yes | The anomaly detection job ID |
+
+**API calls:**
+
+```text
+GET _ml/anomaly_detectors/{job_id}
+GET _ml/anomaly_detectors/{job_id}/_stats
+```
+
+**Returns:** `analysis_config` (detectors, by/over/partition fields, bucket_span, frequency), `datafeed_config` (source
+indices, query, query_delay, delayed_data_check_config), and runtime stats (memory_status, model_bytes, data_counts,
+state).
+
+**When to use:** Required before calling `ad_rca_source_evidence` (to get the source index). Also used in
+troubleshooting workflows to inspect query_delay, bucket_span, and custom_rules.
+
+---
+
+### `ad_discover_jobs_by_datafeed_index`
+
+Given a job of interest, find all other jobs whose datafeed reads from the same source indices.
+
+| Parameter | Type | Required | Description |
+| --------- | ------ | -------- | ---------------------------------------- |
+| `job_id` | string | yes | The job ID whose source indices to match |
+
+**API calls:**
+
+```text
+GET _ml/anomaly_detectors/{job_id} ← get target job's datafeed_config.indices
+GET .ml-config/_search ← find all other jobs with overlapping indices
+```
+
+**Returns:** Jobs grouped by shared index pattern, with match count per group.
+
+**When to use:** Step 2 of every investigation. Jobs reading from the same source index monitor the same underlying
+system — strongest config-level relatedness signal.
+
+\*\*Fallback if unavailable: Use ESQL to find get target job's datafeed_config.indices then all other jobs with
+overlapping indices
+
+```esql
+FROM .ml-config
+| WHERE job_type == "anomaly_detector"
+```
+
+---
+
+## Datafeed Lifecycle
+
+### `ad_manage_datafeed`
+
+Start or stop a datafeed (`POST _ml/datafeeds/{datafeed_id}/_start` or `/_stop`). Preview uses a separate GET endpoint;
+use `ad_preview_datafeed_with_latency` instead of `_preview` on this workflow.
+
+| Parameter | Type | Required | Description |
+| ------------- | ------ | -------- | ----------------------------- |
+| `datafeed_id` | string | yes | Typically `datafeed-{job_id}` |
+| `action` | string | yes | `_start` or `_stop` only |
+
+**API call:**
+
+```text
+POST _ml/datafeeds/{datafeed_id}/{action}
+```
+
+**When to use:** Required as part of remediation sequences:
+
+- Stop datafeed before updating `query_delay` or `model_memory_limit`
+- Restart datafeed after updating job config
+- To preview extracted rows before starting, call `ad_preview_datafeed_with_latency`
+ (`GET _ml/datafeeds/{datafeed_id}/_preview`)
+
+---
+
+### `ad_preview_datafeed_with_latency`
+
+Preview a datafeed's output to inspect data quality and measure effective latency.
+
+| Parameter | Type | Required | Description |
+| ------------- | ------ | -------- | -------------------------- |
+| `datafeed_id` | string | yes | The datafeed ID to preview |
+
+**API call:**
+
+```text
+GET _ml/datafeeds/{datafeed_id}/_preview
+```
+
+**Returns:** Sample documents that the datafeed would extract, including field values and timestamps. Use to verify: Is
+`event.ingested` available? Are all expected fields present? What does the data look like at query time?
+
+**When to use:** Before tuning `query_delay` — understand what data the datafeed sees and which timestamp fields are
+available for latency measurement.
+
+---
+
+### `ad_update_datafeed_query_delay`
+
+Update the `query_delay` on a datafeed to capture more late-arriving data.
+
+| Parameter | Type | Required | Description |
+| ----------------- | ------ | -------- | ----------------------------------------- |
+| `datafeed_id` | string | yes | The datafeed ID |
+| `new_query_delay` | string | yes | New delay value, e.g., `3m`, `120s`, `5m` |
+
+**API call:**
+
+```text
+POST _ml/datafeeds/{datafeed_id}/_update
+Body: {"query_delay": "{new_query_delay}"}
+```
+
+**Prerequisites:** Datafeed must be stopped first (`ad_manage_datafeed` with `_stop`).
+
+**Trade-off:** Larger `query_delay` = more late-arriving data captured = slower anomaly alerts. Set to P95 ingest
+latency + a buffer (e.g., if P95 latency is 90s, set to `2m`).
+
+---
+
+### `ad_update_delayed_data_check_config`
+
+Control how aggressively delayed data is detected and annotated.
+
+| Parameter | Type | Required | Description |
+| -------------- | ------- | -------- | -------------------------------------------------------------------------------------------------------- |
+| `job_id` | string | yes | The anomaly detection job ID |
+| `enabled` | boolean | yes | Enable or disable delayed data checks |
+| `check_window` | string | no | Time window to scan for delayed data, e.g., `2h`; omitted from the request when empty or whitespace-only |
+
+**API call:**
+
+```text
+Step 1: data.parseJson — build JSON (enabled as boolean; check_window only when non-blank after trim)
+Step 2: POST _ml/anomaly_detectors/{job_id}/_update
+Body (example with window): {"analysis_config": {"delayed_data_check_config": {"enabled": true, "check_window": "2h"}}}
+Body (example without window): {"analysis_config": {"delayed_data_check_config": {"enabled": true}}}
+```
+
+**When to use:** Disable if delayed data checks are generating false positives. Increase `check_window` if late-arriving
+data is coming in very late (>1h after event time).
+
+---
+
+## Memory Management
+
+### `ad_estimate_memory_requirement`
+
+Compute a principled `model_memory_limit` estimate using the same algorithm Elasticsearch uses internally.
+
+| Parameter | Type | Required | Description |
+| -------------- | ------ | -------- | ------------------------------------------------------------------------- |
+| `job_id` | string | yes | The anomaly detection job ID |
+| `sample_start` | string | no | Start of cardinality sampling period (ISO 8601). Defaults to 30 days ago. |
+| `sample_end` | string | no | End of cardinality sampling period (ISO 8601). Defaults to now. |
+
+**API calls (7 steps):**
+
+```text
+1. GET _ml/anomaly_detectors/{job_id} ← get config
+2. [identify cardinality fields]
+3. POST {indices}/_search?size=0 ← overall_cardinality (aggs)
+4. POST {indices}/_search?size=0 ← max_bucket_cardinality (date_histogram)
+5. POST _ml/anomaly_detectors/_estimate_model_memory ← official estimate
+6. [compare estimate vs current limit vs actual usage]
+7. [produce recommendation]
+```
+
+**Returns:** Estimated memory requirement, comparison against current `model_memory_limit` and `peak_model_bytes`, and a
+recommendation (increase / current is appropriate / reduce data).
+
+**When to use:** Before increasing `model_memory_limit`. Much more accurate than `peak_model_bytes * 1.3` because it
+uses the actual cardinality of your data.
+
+---
+
+### `ad_wf_ts_field_cardinality`
+
+Approximate **distinct value count** for a split field (`partition_field`, `by_field`, or `over_field`) in **source**
+data over a time window via ES|QL `POST /_query`. `source_index`, `start_time`, and `end_time` are sent as **named
+literal** `params` for `?` placeholders; `split_field_esql` is **interpolated** into the query as the `COUNT_DISTINCT`
+column (for example `service.keyword` or `` `host.name.keyword` `` from `ad_get_job_datafeed_config`). If source
+cardinality is much larger than `total_*_count` from `ad_ts_model_memory_health`, entities may be dropped. For CCS, run
+per cluster and sum. Prefer `ad_estimate_memory_requirement` for full sizing.
+
+#### Parameters
+
+- `source_index` (string, required): index name or LIKE pattern from the datafeed config (same filter as
+ `FROM * METADATA _index`).
+- `split_field_esql` (string, required): valid ES|QL column expression for `COUNT_DISTINCT` — not a string literal; use
+ values from job analysis config only.
+- `start_time` (string, required): ISO 8601 start of window.
+- `end_time` (string, required): ISO 8601 end of window.
+
+**API call:**
+
+```text
+POST /_query
+Body: { "query": "... COUNT_DISTINCT() ...", "params": { "source_index": "...", "start_time": "...", "end_time": "..." } }
+```
+
+**When to use:** After `ad_ts_model_memory_health` shows memory pressure or `total_*_count` lower than expected — check
+whether the source still has more distinct split values than the model retains.
+
+---
+
+### `ad_update_model_memory_limit`
+
+Update the `model_memory_limit` on a job.
+
+| Parameter | Type | Required | Description |
+| ----------- | ------ | -------- | ------------------------------------- |
+| `job_id` | string | yes | The anomaly detection job ID |
+| `new_limit` | string | yes | New limit value, e.g., `256mb`, `1gb` |
+
+**API call:**
+
+```text
+POST _ml/anomaly_detectors/{job_id}/_update
+Body: {"analysis_limits": {"model_memory_limit": "{new_limit}"}}
+```
+
+**Prerequisites:** Job must be closed first. Full remediation sequence:
+
+1. `ad_manage_datafeed` with `_stop`
+2. `POST _ml/anomaly_detectors/{job_id}/_close`
+3. `ad_update_model_memory_limit`
+4. `POST _ml/anomaly_detectors/{job_id}/_open`
+5. `ad_manage_datafeed` with `_start`
+
+**Constraint:** Cannot decrease below current `model_bytes`. To shrink, clone the job with a lower limit.
+
+---
+
+## Model Snapshots
+
+### `ad_revert_model_snapshot`
+
+Revert a job's model to a previous snapshot to "unlearn" bad data.
+
+| Parameter | Type | Required | Description |
+| ------------- | ------ | -------- | ------------------------------------------------------------ |
+| `job_id` | string | yes | The anomaly detection job ID |
+| `snapshot_id` | string | yes | The snapshot ID to revert to (from `ad_get_model_snapshots`) |
+
+**API call:**
+
+```text
+POST _ml/anomaly_detectors/{job_id}/model_snapshots/{snapshot_id}/_revert
+```
+
+**Prerequisites:** Job must be closed before reverting.
+
+**Post-revert steps:**
+
+1. Reopen the job: `POST _ml/anomaly_detectors/{job_id}/_open`
+2. Restart datafeed from the snapshot timestamp to reprocess data (prevents gap in analysis)
+
+**When to use:** When bad/anomalous training data has corrupted the model — e.g., a 2-week outage that the model
+"learned" as normal. Revert to a snapshot from before the bad data, then reprocess.
+
+---
+
+## Calendar Management
+
+### `ad_get_calendar_events`
+
+Retrieve scheduled events from a calendar.
+
+| Parameter | Type | Required | Description |
+| ------------- | ------ | -------- | --------------- |
+| `calendar_id` | string | yes | The calendar ID |
+
+**API call:**
+
+```text
+GET _ml/calendars/{calendar_id}/events
+```
+
+**Returns:** List of events with `description`, `start_time`, `end_time`.
+
+**When to use:** Before adding a new event, verify what's already scheduled. Also use to audit whether a score anomaly
+was suppressed by an existing calendar event.
+
+---
+
+### `ad_create_calendar_event`
+
+Add a scheduled event to suppress false positives during known downtime.
+
+| Parameter | Type | Required | Description |
+| ------------- | ------ | -------- | -------------------------------------------------------------- |
+| `calendar_id` | string | yes | The calendar ID (must already exist) |
+| `description` | string | yes | Human-readable description, e.g., `Planned maintenance window` |
+| `start_time` | string | yes | ISO 8601 or epoch_millis |
+| `end_time` | string | yes | ISO 8601 or epoch_millis |
+
+**API call:**
+
+```text
+POST _ml/calendars/{calendar_id}/events
+Body: {"events": [{"description": "...", "start_time": "...", "end_time": "..."}]}
+```
+
+**Effect:** During the event window, the ML model does not produce anomaly results — it continues learning but
+suppresses output. Results resume normally after the window ends.
+
+**When to use:** Planned maintenance, deployments, known data pipeline downtime, holidays with predictable traffic
+changes.
+
+---
+
+## Job Creation
+
+Workflows `ad_validate_job_spec`, `ad_create_job`, and `ad_create_datafeed` accept JSON **text** inputs (`job_body`,
+`datafeed_body`). Each step uses Liquid `json_parse` with typed `${{ }}` interpolation so Elasticsearch receives a
+structured JSON object in the HTTP body, not a JSON value that is itself a quoted string.
+
+### `ad_validate_job_spec`
+
+Validate a job configuration before create.
+
+| Parameter | Type | Required | Description |
+| ---------- | ------ | -------- | --------------------------------------------------------------------------- |
+| `job_body` | string | yes | Full job JSON text (PUT create shape); parsed to an object in the workflow. |
+
+**API call:**
+
+```text
+POST _ml/anomaly_detectors/_validate
+Body: {full job configuration JSON document}
+```
+
+### `ad_create_job`
+
+Create a new anomaly detection job from a configuration.
+
+| Parameter | Type | Required | Description |
+| ---------- | ------ | -------- | -------------------------------------------------------- |
+| `job_id` | string | yes | The new job ID to create |
+| `job_body` | string | yes | Full job JSON text; parsed to an object in the workflow. |
+
+**API call:**
+
+```text
+PUT _ml/anomaly_detectors/{job_id}
+Body: {full job configuration JSON document}
+```
+
+**When to use:** Advanced scenario — when the agent has explored the data and designed a job configuration. Requires a
+complete `analysis_config` including detectors and `data_description`; create the datafeed with `ad_create_datafeed`
+before open/start.
+
+### `ad_create_datafeed`
+
+Create or replace a datafeed.
+
+| Parameter | Type | Required | Description |
+| --------------- | ------ | -------- | ------------------------------------------------------------- |
+| `datafeed_id` | string | yes | Typically `datafeed-{job_id}` |
+| `datafeed_body` | string | yes | Full datafeed JSON text; parsed to an object in the workflow. |
+
+**API call:**
+
+```text
+PUT _ml/datafeeds/{datafeed_id}
+Body: {full datafeed configuration JSON document}
+```
+
+### `ad_open_job`
+
+Open a job after configuration exists.
+
+| Parameter | Type | Required | Description |
+| --------- | ------ | -------- | -------------- |
+| `job_id` | string | yes | Job ID to open |
+
+**API call:**
+
+```text
+POST _ml/anomaly_detectors/{job_id}/_open
+```
+
+---
+
+## Permissions & Diagnostics
+
+### `ad_validate_ml_tool_permissions`
+
+Preflight check for core `.ml-*` indices used by packaged tools (see workflow YAML for exact `_has_privileges` payload).
+Does **not** cover job-specific source indices — validate those separately.
+
+| Parameter | Type | Required | Description |
+| --------- | ---- | -------- | ----------- |
+| _(none)_ | — | — | — |
+
+**API calls:**
+
+```text
+GET _security/_authenticate ← identify current user/API key
+POST _security/user/_has_privileges ← check index permissions
+Body: {
+ "index": [
+ {"names": [".ml-anomalies-*"], "privileges": ["read", "view_index_metadata"]},
+ {"names": [".ml-config"], "privileges": ["read", "view_index_metadata"]},
+ {"names": [".ml-annotations-*"], "privileges": ["read", "view_index_metadata"]},
+ {"names": [".ml-notifications-*"], "privileges": ["read", "view_index_metadata"]}
+ ]
+}
+```
+
+**Returns:** Current identity and a boolean per index/privilege combination.
+
+**When to use:** Early sanity check for `.ml-*` access; still verify source index privileges before datafeed preview or
+evidence queries.
+
+---
+
+### `ad_ts_ccs_diagnostics`
+
+Diagnose cross-cluster search (CCS) issues for datafeeds querying remote clusters.
+
+| Parameter | Type | Required | Description |
+| --------- | ---- | -------- | ----------- |
+| _(none)_ | — | — | — |
+
+**API calls:**
+
+```text
+GET _remote/info ← list configured remote clusters and connection status
+GET _cluster/health ← check local cluster health
+```
+
+**Returns:** Remote cluster connectivity status (connected/disconnected), latency, and error rates.
+
+**When to use:** When datafeed source indices use CCS patterns (e.g., `remote1:logs-*`) and the job reports missing
+documents or delayed data that can't be explained by local ingest latency.
+
+---
+
+## Troubleshooting Workflows (Decision Trees)
+
+### `ad_wf_troubleshoot_anomaly_score`
+
+Branching decision tree for unexpectedly high or low anomaly scores.
+
+| Parameter | Type | Required | Description |
+| ------------------ | ------ | -------- | --------------------------------------------------------- |
+| `job_id` | string | yes | The anomaly detection job ID |
+| `record_timestamp` | string | no | ISO 8601 timestamp of the specific anomaly to investigate |
+| `entity_value` | string | no | Entity value to focus investigation on |
+
+**Decision tree steps:**
+
+1. **Gate checks**: sufficient training data? memory status ok? delayed data present?
+2. **Score comparison**: compare `record_score` vs `initial_record_score` — large gap = renormalization
+3. **Job config analysis**: `bucket_span`, detector function, `custom_rules`, `use_null`
+4. **Model learning**: model plot (wide bounds = high variance), score reassessment for drift
+5. **Score factor education**: explain each `anomaly_score_explanation` component
+
+**Trigger phrases:** "Why is my score low?", "Expected anomaly not detected", "Score too high"
+
+---
+
+### `ad_wf_troubleshoot_memory_limit`
+
+Branching decision tree for `soft_limit` / `hard_limit` memory issues.
+
+| Parameter | Type | Required | Description |
+| --------- | ------ | -------- | -------------------------------------------- |
+| `job_id` | string | yes | The anomaly detection job ID to troubleshoot |
+
+**Decision tree steps:**
+
+1. **Memory status**: `model_bytes`, limit, `memory_status`, entity counts, allocation failures
+2. **False alarm check**: if `hard_limit`, check for missing-doc warnings that are symptoms of memory (not ingest lag)
+3. **Growth trend**: classify as stable plateau / linear growth / exponential growth
+4. **Threshold inspection**: `total_by_field_count > 100K`? `total_partition_field_count > 10K`? Identify which field
+ drives memory.
+5. **Principled estimate**: `ad_estimate_memory_requirement` — compare vs current limit vs actual usage
+6. **CCS check**: if datafeed uses CCS, account for cross-cluster cardinality
+7. **Recommendation**: increase limit / reduce data (filter datafeed, exclude datasets, reduce influencers) /
+ restructure job
+
+**Trigger phrases:** "My job hit memory limit", "hard_limit", "soft_limit"
+
+---
+
+### `ad_wf_troubleshoot_query_delay`
+
+Branching decision tree for missing documents and query_delay warnings.
+
+| Parameter | Type | Required | Description |
+| --------- | ------ | -------- | -------------------------------------------- |
+| `job_id` | string | yes | The anomaly detection job ID to troubleshoot |
+
+**Decision tree steps:**
+
+1. **Hard_limit false alarm**: if job has `categorization_field_name` AND `memory_status == hard_limit` → fix memory
+ first (missing docs = symptom, not cause)
+2. **Delayed data annotations**: frequency, severity, affected time ranges
+3. **Bucket event gaps**: buckets with zero or abnormally low event counts
+4. **Ingest latency measurement**: if `event.ingested` available, use it; otherwise use date_histogram comparison
+ fallback
+5. **Recommend `query_delay`**: P95 latency + buffer; trade-off: larger delay = slower alerts
+6. **Additional remediation**: add ingest pipeline timestamp, consider `event.ingested` as `time_field`, revert model
+ snapshot if catastrophically late data
+
+**Trigger phrases:** "Missing documents", "Datafeed has missed X documents", "query_delay"
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/scripts/agent_builder_constants.json b/plugins/kibana/skills/kibana-anomaly-detection/scripts/agent_builder_constants.json
new file mode 100644
index 0000000..bf490f5
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/scripts/agent_builder_constants.json
@@ -0,0 +1,33 @@
+{
+ "default_agent_id": "elastic-ai-agent",
+ "default_agent_enable_elastic_capabilities": true,
+ "workflow_tool_exclusions": [
+ "ad_get_job_datafeed_config",
+ "ad_ts_ccs_diagnostics",
+ "ad_get_calendar_events",
+ "ad_create_calendar_event",
+ "ad_discover_jobs_by_datafeed_index"
+ ],
+ "fallback_tools": {
+ "ad_get_index_mappings": "platform.core.get_index_mapping"
+ },
+ "workflow_prefixes": [
+ "ad_wf_",
+ "ad_create_",
+ "ad_manage_",
+ "ad_open_",
+ "ad_update_",
+ "ad_revert_",
+ "ad_preview_",
+ "ad_validate_",
+ "ad_estimate_"
+ ],
+ "builtin_tools": [
+ "platform.core.search",
+ "platform.core.list_indices",
+ "platform.core.get_index_mapping",
+ "platform.core.execute_esql",
+ "platform.core.generate_esql",
+ "platform.core.product_documentation"
+ ]
+}
diff --git a/plugins/kibana/skills/kibana-anomaly-detection/scripts/kibana-agent-builder.mjs b/plugins/kibana/skills/kibana-anomaly-detection/scripts/kibana-agent-builder.mjs
new file mode 100755
index 0000000..8170482
--- /dev/null
+++ b/plugins/kibana/skills/kibana-anomaly-detection/scripts/kibana-agent-builder.mjs
@@ -0,0 +1,1481 @@
+#!/usr/bin/env node
+/**
+ * Kibana Agent Builder helpers for anomaly-detection — connection pattern aligned with
+ * elastic/agent-skills `kibana-dashboards.js`.
+ *
+ * Requires Node.js 18+ (global fetch). Optional `node:undici` / `undici` for TLS bypass without
+ * mutating process-wide NODE_TLS_REJECT_UNAUTHORIZED when available.
+ *
+ * Usage (run from repo root; script lives under skills/kibana/kibana-anomaly-detection/scripts/):
+ * node skills/kibana/kibana-anomaly-detection/scripts/kibana-agent-builder.mjs test
+ * node skills/kibana/kibana-anomaly-detection/scripts/kibana-agent-builder.mjs tools register [--dry-run]
+ * node skills/kibana/kibana-anomaly-detection/scripts/kibana-agent-builder.mjs skills register [--dry-run]
+ * node skills/kibana/kibana-anomaly-detection/scripts/kibana-agent-builder.mjs all register [--dry-run] # tools → workflows → skills → merge skill_ids on elastic-ai-agent
+ *
+ * Kibana compatibility: targets **9.4+** Agent Builder / Workflows. Tool and skill updates use PUT when the API
+ * supports it (9.5+); if PUT is missing or returns 404/405, registration falls back to DELETE + POST so idempotent
+ * runs still succeed on older minors.
+ *
+ * Environment variables (same as kibana-dashboards.js):
+ * KIBANA_URL, KIBANA_CLOUD_ID / ELASTICSEARCH_CLOUD_ID,
+ * KIBANA_USERNAME / ELASTICSEARCH_USERNAME,
+ * KIBANA_PASSWORD / ELASTICSEARCH_PASSWORD,
+ * KIBANA_API_KEY / ELASTICSEARCH_API_KEY,
+ * KIBANA_SPACE_ID, KIBANA_INSECURE
+ */
+
+import { readFileSync, readdirSync, existsSync } from "fs";
+import { join, dirname, basename } from "path";
+import { fileURLToPath } from "url";
+import { createRequire } from "module";
+
+const require = createRequire(import.meta.url);
+
+const __dirname = dirname(fileURLToPath(import.meta.url));
+const KIBANA_REF_DIR = join(__dirname, "..", "references", "kibana");
+const AGENT_DIR = join(KIBANA_REF_DIR, "agent");
+const TOOLS_DIR = join(KIBANA_REF_DIR, "tools");
+const WORKFLOWS_DIR = join(KIBANA_REF_DIR, "workflows");
+const PLUGIN_ROOT = join(__dirname, "..");
+const SKILLS_DIR = join(PLUGIN_ROOT, "skills");
+const CONSTANTS_PATH = join(__dirname, "agent_builder_constants.json");
+
+const CONSTANTS = JSON.parse(readFileSync(CONSTANTS_PATH, "utf8"));
+const DEFAULT_AGENT_ID = CONSTANTS.default_agent_id ?? "elastic-ai-agent";
+/** When true (default), PUT sets configuration.enable_elastic_capabilities so bundle skills are active in the Elastic AI Agent. Set false in agent_builder_constants.json to skip. */
+const DEFAULT_AGENT_ENABLE_ELASTIC_CAPABILITIES = CONSTANTS.default_agent_enable_elastic_capabilities !== false;
+const WORKFLOW_TOOL_EXCLUSIONS = new Set(CONSTANTS.workflow_tool_exclusions);
+const WORKFLOW_PREFIXES = CONSTANTS.workflow_prefixes;
+const BUILTIN_TOOLS = CONSTANTS.builtin_tools;
+const FALLBACK_TOOLS = CONSTANTS.fallback_tools || {};
+const BUILTIN_TOOLS_SET = new Set(BUILTIN_TOOLS);
+
+/** Kibana Agent Builder rejects skills with more than this many tool_ids. */
+const MAX_SKILL_TOOL_IDS = 5;
+
+let kibanaFetchImpl = globalThis.fetch.bind(globalThis);
+let insecureDispatcher = null;
+try {
+ const u = require("node:undici");
+ kibanaFetchImpl = u.fetch.bind(u);
+ insecureDispatcher = new u.Agent({ connect: { rejectUnauthorized: false } });
+} catch {
+ try {
+ const u = require("undici");
+ kibanaFetchImpl = u.fetch.bind(u);
+ insecureDispatcher = new u.Agent({ connect: { rejectUnauthorized: false } });
+ } catch {
+ insecureDispatcher = null;
+ }
+}
+
+/**
+ * HTTP fetch for Kibana: uses undici Agent for insecure TLS when possible (no global env toggle).
+ */
+async function kibanaHttpFetch(url, config, init = {}) {
+ const merged = {
+ ...init,
+ headers: { ...init.headers },
+ };
+
+ if (config.insecure && insecureDispatcher) {
+ merged.dispatcher = insecureDispatcher;
+ return kibanaFetchImpl(url, merged);
+ }
+
+ if (config.insecure && !insecureDispatcher) {
+ const prev = process.env.NODE_TLS_REJECT_UNAUTHORIZED;
+ try {
+ process.env.NODE_TLS_REJECT_UNAUTHORIZED = "0";
+ return await globalThis.fetch(url, merged);
+ } finally {
+ if (prev === undefined) {
+ delete process.env.NODE_TLS_REJECT_UNAUTHORIZED;
+ } else {
+ process.env.NODE_TLS_REJECT_UNAUTHORIZED = prev;
+ }
+ }
+ }
+
+ return globalThis.fetch(url, merged);
+}
+
+/**
+ * Read JSON from a fetch Response body; keeps raw text if JSON.parse fails.
+ */
+async function readJsonBody(res) {
+ const text = await res.text();
+ if (!text?.trim()) return { text: "", parsed: null };
+ try {
+ return { text, parsed: JSON.parse(text) };
+ } catch {
+ return { text, parsed: null };
+ }
+}
+
+// -----------------------------------------------------------------------------
+// Kibana client (aligned with kibana-dashboards.js)
+// -----------------------------------------------------------------------------
+
+function kibanaUrlFromCloudId(cloudId) {
+ try {
+ const parts = cloudId.split(":");
+ if (parts.length !== 2) return null;
+ const decoded = Buffer.from(parts[1], "base64").toString("utf8");
+ const decodedParts = decoded.split("$");
+ if (decodedParts.length < 3 || !decodedParts[2]) return null;
+ const domain = decodedParts[0];
+ const kibanaUuid = decodedParts[2];
+ let host = domain;
+ let port = "";
+ if (domain.includes(":")) {
+ const splitDomain = domain.split(":");
+ host = splitDomain[0];
+ port = `:${splitDomain[1]}`;
+ } else {
+ port = ":443";
+ }
+ return `https://${kibanaUuid}.${host}${port}`;
+ } catch {
+ return null;
+ }
+}
+
+function resolveApiKey(cli) {
+ if (cli.apiKeyFromCli) {
+ const t = (cli.apiKey ?? "").trim();
+ return { apiKey: t || undefined, apiKeyCliEmpty: !t };
+ }
+ for (const name of ["KIBANA_API_KEY", "ELASTICSEARCH_API_KEY"]) {
+ const raw = process.env[name];
+ if (raw == null) continue;
+ const t = String(raw).trim();
+ if (t) return { apiKey: t, apiKeyCliEmpty: false };
+ }
+ return { apiKey: undefined, apiKeyCliEmpty: false };
+}
+
+const DEFAULT_KIBANA_URL = "http://localhost:5601";
+
+export function getKibanaConfig(cli = {}) {
+ const cloudId = process.env.KIBANA_CLOUD_ID || process.env.ELASTICSEARCH_CLOUD_ID;
+ let url = cli.kibanaUrl || process.env.KIBANA_URL;
+
+ if (!url && cloudId) {
+ url = kibanaUrlFromCloudId(cloudId);
+ }
+
+ const usingDefaults = !url && !cloudId;
+ if (usingDefaults) url = DEFAULT_KIBANA_URL;
+
+ const { apiKey, apiKeyCliEmpty } = resolveApiKey(cli);
+ const username =
+ cli.username ??
+ process.env.KIBANA_USERNAME ??
+ process.env.ELASTICSEARCH_USERNAME ??
+ (apiKey ? undefined : "elastic");
+ const password =
+ cli.password ??
+ process.env.KIBANA_PASSWORD ??
+ process.env.ELASTICSEARCH_PASSWORD ??
+ (apiKey ? undefined : "changeme");
+ let spaceId = cli.spaceId ?? process.env.KIBANA_SPACE_ID;
+ if (spaceId === "default") spaceId = undefined;
+
+ const insecureFlag =
+ cli.insecure === true || ["1", "true", "yes"].includes((process.env.KIBANA_INSECURE || "").toLowerCase());
+
+ return {
+ url,
+ username,
+ password,
+ apiKey,
+ apiKeyCliEmpty,
+ spaceId,
+ insecure: insecureFlag,
+ usingDefaults,
+ };
+}
+
+function warnUsingDefaults() {
+ console.log(`No Kibana URL configured — using default: ${DEFAULT_KIBANA_URL} (elastic/changeme)`);
+ console.log("");
+ console.log("If you want to override Kibana configuration, you can set one of:");
+ console.log(" 1. Elastic Cloud: KIBANA_CLOUD_ID + KIBANA_API_KEY");
+ console.log(" 2. URL + API Key: KIBANA_URL + KIBANA_API_KEY");
+ console.log(" 3. Basic Auth: KIBANA_URL + KIBANA_USERNAME + KIBANA_PASSWORD");
+ console.log(" 4. CLI flags: --kibana-url, --username, --password, --api-key");
+ console.log("");
+}
+
+function getHeaders(config) {
+ const headers = {
+ "Content-Type": "application/json",
+ "kbn-xsrf": "true",
+ "x-elastic-internal-origin": "kibana",
+ "User-Agent": "elastic-agentic-anomaly-detection",
+ };
+
+ if (config.apiKey) {
+ headers.Authorization = `ApiKey ${config.apiKey}`;
+ } else if (config.username && config.password) {
+ const auth = Buffer.from(`${config.username}:${config.password}`).toString("base64");
+ headers.Authorization = `Basic ${auth}`;
+ }
+
+ return headers;
+}
+
+function getBasePath(config) {
+ let basePath = (config.url || "").replace(/\/$/, "");
+ if (config.spaceId && config.spaceId !== "default") {
+ basePath += `/s/${config.spaceId}`;
+ }
+ return basePath;
+}
+
+function validateConfig(config, { dryRun }) {
+ if (config.apiKeyCliEmpty) {
+ console.error("Error: --api-key cannot be empty or whitespace-only.");
+ return false;
+ }
+ if (config.apiKey) return true;
+ if (config.username && config.password) return true;
+ if (dryRun) return true;
+ console.error("Error: Authentication required (API key or username + password).");
+ return false;
+}
+
+async function kibanaFetch(config, path, options = {}) {
+ const basePath = getBasePath(config);
+ const url = `${basePath}${path}`;
+
+ const fetchOptions = {
+ ...options,
+ headers: {
+ ...getHeaders(config),
+ ...options.headers,
+ },
+ };
+
+ const response = await kibanaHttpFetch(url, config, fetchOptions);
+ const contentType = response.headers.get("content-type");
+ let data;
+ if (contentType && contentType.includes("application/json")) {
+ data = await response.json();
+ } else {
+ data = await response.text();
+ }
+
+ return { ok: response.ok, status: response.status, data };
+}
+
+async function kibanaEsRequest(config, method, esPath, body) {
+ const query = new URLSearchParams({
+ path: esPath,
+ method: method.toUpperCase(),
+ }).toString();
+
+ const options = { method: "POST" };
+ if (body !== undefined) {
+ options.body = JSON.stringify(body);
+ }
+
+ return kibanaFetch(config, `/api/console/proxy?${query}`, options);
+}
+
+// -----------------------------------------------------------------------------
+// Tool registration
+// -----------------------------------------------------------------------------
+
+function loadToolDefs() {
+ const tools = [];
+ for (const subdir of ["esql"]) {
+ const dir = join(TOOLS_DIR, subdir);
+ if (!existsSync(dir)) continue;
+ for (const name of readdirSync(dir).sort()) {
+ if (!name.endsWith(".json")) continue;
+ try {
+ const def = JSON.parse(readFileSync(join(dir, name), "utf8"));
+ def._sourceFile = `${subdir}/${name}`;
+ tools.push(def);
+ } catch (e) {
+ console.warn(`Skipping ${subdir}/${name}: ${e.message}`);
+ }
+ }
+ }
+ return tools;
+}
+
+function transformToolDef(def) {
+ const { _sourceFile: _, ...rest } = def;
+ const payload = {
+ id: rest.name || rest.id || "unnamed",
+ type: rest.type || "esql",
+ description: rest.description || "",
+ };
+ if (rest.tags) payload.tags = rest.tags;
+ const config = { ...(rest.configuration || {}) };
+ if (payload.type === "esql") config.params = rest.parameters || {};
+ payload.configuration = config;
+ return payload;
+}
+
+/**
+ * True when PUT failed because the update route/method is unavailable (older Kibana), not request validation.
+ * In that case we fall back to DELETE + POST with the full create payload.
+ */
+function agentBuilderPutUnsupported(status, errorText) {
+ const t = (errorText || "").slice(0, 800);
+ if (status === 404 || status === 405 || status === 501) return true;
+ if (status === 400 && /\b(route not found|method not allowed|no handler|cannot\s+(PUT|patch))\b/i.test(t)) {
+ return true;
+ }
+ return false;
+}
+
+function sleep(ms) {
+ return new Promise((resolve) => setTimeout(resolve, ms));
+}
+
+/**
+ * Retry Kibana Agent Builder writes when SNAPSHOT / testcontainers stacks return transient errors during bulk
+ * registration (common after many consecutive tool POSTs).
+ */
+async function kibanaHttpFetchWithRetry(url, config, init, { attempts = 5 } = {}) {
+ let res;
+ for (let attempt = 0; attempt < attempts; attempt++) {
+ res = await kibanaHttpFetch(url, config, init);
+ if (res.ok) return res;
+ if (![408, 429, 502, 503, 504].includes(res.status)) return res;
+ try {
+ await res.text();
+ } catch {
+ /* ignore body read errors */
+ }
+ const backoff = 150 * 2 ** attempt + Math.floor(Math.random() * 100);
+ await sleep(backoff);
+ }
+ return res;
+}
+
+async function registerTool(config, def, dryRun) {
+ const payload = transformToolDef(def);
+ console.log(`Tool: ${payload.id} (${def._sourceFile})`);
+
+ if (dryRun) {
+ console.log(JSON.stringify(payload, null, 2));
+ return true;
+ }
+
+ const basePath = getBasePath(config);
+ const toolsUrl = `${basePath}/api/agent_builder/tools`;
+ const toolByIdUrl = `${toolsUrl}/${encodeURIComponent(payload.id)}`;
+ const headers = { ...getHeaders(config), "kbn-xsrf": "true" };
+
+ /** PUT body — id comes from path; `type` is immutable (POST-only). */
+ const updateBody = {
+ description: payload.description,
+ configuration: payload.configuration,
+ };
+ if (payload.tags) updateBody.tags = payload.tags;
+
+ let res = await kibanaHttpFetchWithRetry(
+ toolsUrl,
+ config,
+ {
+ method: "POST",
+ headers,
+ body: JSON.stringify(payload),
+ },
+ { attempts: 5 },
+ );
+
+ if (res.ok) {
+ console.log(` Registered: ${payload.id}`);
+ return true;
+ }
+
+ const text = await res.text();
+ const alreadyExists = res.status === 409 || (res.status === 400 && /already exists|duplicate/i.test(text));
+
+ if (alreadyExists) {
+ console.log(` Already exists — updating (PUT): ${payload.id}`);
+ const putRes = await kibanaHttpFetchWithRetry(
+ toolByIdUrl,
+ config,
+ {
+ method: "PUT",
+ headers,
+ body: JSON.stringify(updateBody),
+ },
+ { attempts: 5 },
+ );
+ if (putRes.ok) {
+ console.log(` Updated: ${payload.id}`);
+ return true;
+ }
+ const putText = await putRes.text();
+ if (agentBuilderPutUnsupported(putRes.status, putText)) {
+ console.log(` PUT unsupported — deleting and re-creating: ${payload.id}`);
+ await kibanaHttpFetchWithRetry(toolByIdUrl, config, { method: "DELETE", headers }, { attempts: 3 });
+ const recRes = await kibanaHttpFetchWithRetry(
+ toolsUrl,
+ config,
+ {
+ method: "POST",
+ headers,
+ body: JSON.stringify(payload),
+ },
+ { attempts: 5 },
+ );
+ if (recRes.ok) {
+ console.log(` Re-created: ${payload.id}`);
+ return true;
+ }
+ const recText = await recRes.text();
+ console.error(` Failed: ${recRes.status} ${recText.slice(0, 200)}`);
+ return false;
+ }
+ console.error(` Failed to update tool: ${putRes.status} ${putText.slice(0, 200)}`);
+ return false;
+ }
+
+ console.error(` Failed: ${res.status} ${text.slice(0, 200)}`);
+ return false;
+}
+
+async function cmdToolsRegister(argv) {
+ let dryRun = false;
+ const cli = {};
+ for (let i = 0; i < argv.length; i++) {
+ const a = argv[i];
+ if (a === "--dry-run") dryRun = true;
+ else if (a === "--kibana-url") cli.kibanaUrl = argv[++i];
+ else if (a === "--username") cli.username = argv[++i];
+ else if (a === "--password") cli.password = argv[++i];
+ else if (a === "--api-key") {
+ cli.apiKeyFromCli = true;
+ cli.apiKey = argv[++i];
+ } else if (a === "--space-id") cli.spaceId = argv[++i];
+ else if (a === "--insecure") cli.insecure = true;
+ }
+
+ const config = getKibanaConfig(cli);
+ if (!validateConfig(config, { dryRun })) process.exit(1);
+ if (config.usingDefaults && !dryRun) warnUsingDefaults();
+
+ const defs = loadToolDefs();
+ console.log(`Loaded ${defs.length} tool definitions`);
+
+ if (!dryRun) {
+ const st = await kibanaFetch(config, "/api/status");
+ if (!st.ok) {
+ console.error("Cannot reach Kibana:", st.status, st.data);
+ process.exit(1);
+ }
+ console.log("Connected to Kibana", st.data?.version?.number || "?");
+ }
+
+ let succeeded = 0,
+ failed = 0,
+ skipped = 0;
+ for (const def of defs) {
+ const primaryBuiltin = FALLBACK_TOOLS[def.name];
+ if (primaryBuiltin && BUILTIN_TOOLS_SET.has(primaryBuiltin)) {
+ console.log(` Skipping fallback tool: ${def.name} (${primaryBuiltin} is available as a builtin)`);
+ skipped++;
+ continue;
+ }
+ const ok = await registerTool(config, def, dryRun);
+ ok ? succeeded++ : failed++;
+ if (!dryRun) await sleep(120);
+ }
+
+ if (!dryRun) {
+ console.log(`\nRegistration complete: ${succeeded} succeeded, ${failed} failed, ${skipped} skipped`);
+ }
+}
+
+// -----------------------------------------------------------------------------
+// ML job scaffolding
+// -----------------------------------------------------------------------------
+
+function parseJobsCreateServiceHealthArgs(argv) {
+ const cli = {};
+ const opts = {
+ prefix: "svc",
+ metricsIndex: "metrics-*",
+ logsIndex: "logs-*",
+ apmIndex: "apm-*",
+ bucketSpan: "15m",
+ queryDelay: "120s",
+ memoryLimit: "256mb",
+ dryRun: false,
+ };
+
+ for (let i = 0; i < argv.length; i++) {
+ const a = argv[i];
+ if (a === "--dry-run") opts.dryRun = true;
+ else if (a === "--prefix") opts.prefix = argv[++i];
+ else if (a === "--metrics-index") opts.metricsIndex = argv[++i];
+ else if (a === "--logs-index") opts.logsIndex = argv[++i];
+ else if (a === "--apm-index") opts.apmIndex = argv[++i];
+ else if (a === "--bucket-span") opts.bucketSpan = argv[++i];
+ else if (a === "--query-delay") opts.queryDelay = argv[++i];
+ else if (a === "--memory-limit") opts.memoryLimit = argv[++i];
+ else if (a === "--kibana-url") cli.kibanaUrl = argv[++i];
+ else if (a === "--username") cli.username = argv[++i];
+ else if (a === "--password") cli.password = argv[++i];
+ else if (a === "--api-key") {
+ cli.apiKeyFromCli = true;
+ cli.apiKey = argv[++i];
+ } else if (a === "--space-id") cli.spaceId = argv[++i];
+ else if (a === "--insecure") cli.insecure = true;
+ }
+
+ return { cli, opts };
+}
+
+function serviceHealthPlans(opts) {
+ const makePlan = (suffix, description, indices, datafeedQuery, detectors, influencers) => {
+ const jobId = `${opts.prefix}-${suffix}`;
+ const datafeedId = `datafeed-${jobId}`;
+ return {
+ jobId,
+ datafeedId,
+ jobBody: {
+ description,
+ analysis_config: {
+ bucket_span: opts.bucketSpan,
+ detectors,
+ influencers,
+ },
+ analysis_limits: {
+ model_memory_limit: opts.memoryLimit,
+ },
+ data_description: {
+ time_field: "@timestamp",
+ },
+ },
+ datafeedBody: {
+ datafeed_id: datafeedId,
+ job_id: jobId,
+ indices,
+ query_delay: opts.queryDelay,
+ scroll_size: 1000,
+ query: datafeedQuery,
+ },
+ };
+ };
+
+ const serviceInfluencers = ["service.name", "host.name", "kubernetes.pod.name"];
+
+ return [
+ makePlan(
+ "cpu-high-mean",
+ "Detect sustained CPU pressure per service from metrics data.",
+ [opts.metricsIndex],
+ {
+ bool: {
+ filter: [
+ { exists: { field: "@timestamp" } },
+ { exists: { field: "service.name" } },
+ { exists: { field: "system.cpu.total.norm.pct" } },
+ ],
+ },
+ },
+ [
+ {
+ function: "high_mean",
+ field_name: "system.cpu.total.norm.pct",
+ partition_field_name: "service.name",
+ },
+ ],
+ serviceInfluencers,
+ ),
+ makePlan(
+ "memory-high-mean",
+ "Detect sustained memory pressure per service from metrics data.",
+ [opts.metricsIndex],
+ {
+ bool: {
+ filter: [
+ { exists: { field: "@timestamp" } },
+ { exists: { field: "service.name" } },
+ { exists: { field: "system.memory.actual.used.pct" } },
+ ],
+ },
+ },
+ [
+ {
+ function: "high_mean",
+ field_name: "system.memory.actual.used.pct",
+ partition_field_name: "service.name",
+ },
+ ],
+ serviceInfluencers,
+ ),
+ makePlan(
+ "latency-high-mean",
+ "Detect service latency spikes from APM transactions.",
+ [opts.apmIndex],
+ {
+ bool: {
+ filter: [
+ { exists: { field: "@timestamp" } },
+ { exists: { field: "service.name" } },
+ { term: { "processor.event": "transaction" } },
+ { exists: { field: "transaction.duration.us" } },
+ ],
+ },
+ },
+ [
+ {
+ function: "high_mean",
+ field_name: "transaction.duration.us",
+ partition_field_name: "service.name",
+ },
+ ],
+ ["service.name", "transaction.type", "host.name"],
+ ),
+ makePlan(
+ "error-rate-high-count",
+ "Detect service-level error surges from logs and failed transactions.",
+ [opts.logsIndex, opts.apmIndex],
+ {
+ bool: {
+ filter: [{ exists: { field: "@timestamp" } }, { exists: { field: "service.name" } }],
+ should: [{ term: { "log.level": "error" } }, { term: { "event.outcome": "failure" } }],
+ minimum_should_match: 1,
+ },
+ },
+ [
+ {
+ function: "high_count",
+ partition_field_name: "service.name",
+ },
+ ],
+ ["service.name", "event.dataset", "host.name", "kubernetes.pod.name"],
+ ),
+ ];
+}
+
+async function upsertServiceHealthPlan(config, plan, dryRun) {
+ const requests = [
+ ["PUT", `/_ml/anomaly_detectors/${encodeURIComponent(plan.jobId)}`, plan.jobBody],
+ ["PUT", `/_ml/datafeeds/${encodeURIComponent(plan.datafeedId)}`, plan.datafeedBody],
+ ["POST", `/_ml/anomaly_detectors/${encodeURIComponent(plan.jobId)}/_open`],
+ ["POST", `/_ml/datafeeds/${encodeURIComponent(plan.datafeedId)}/_start`],
+ ];
+
+ console.log(`\nJob: ${plan.jobId}`);
+ if (dryRun) {
+ for (const [method, path, body] of requests) {
+ console.log(`${method} ${path}`);
+ if (body !== undefined) console.log(JSON.stringify(body, null, 2));
+ }
+ return true;
+ }
+
+ for (const [method, path, body] of requests) {
+ const res = await kibanaEsRequest(config, method, path, body);
+ if (res.ok) {
+ console.log(` ${method} ${path} -> ${res.status}`);
+ continue;
+ }
+ const text = typeof res.data === "string" ? res.data : JSON.stringify(res.data);
+ const alreadyExists =
+ res.status === 409 || (res.status === 400 && /already exists|resource_already_exists_exception/i.test(text));
+ const alreadyStarted =
+ res.status === 409 || (res.status === 400 && /already open|already started|is started/i.test(text));
+ if (alreadyExists || alreadyStarted) {
+ console.log(` ${method} ${path} -> exists/already started; continuing`);
+ continue;
+ }
+ console.error(` ${method} ${path} -> failed (${res.status}): ${text.slice(0, 350)}`);
+ return false;
+ }
+
+ return true;
+}
+
+async function cmdJobsCreateServiceHealth(argv) {
+ const { cli, opts } = parseJobsCreateServiceHealthArgs(argv);
+ const config = getKibanaConfig(cli);
+ if (!validateConfig(config, { dryRun: opts.dryRun })) process.exit(1);
+ if (config.usingDefaults && !opts.dryRun) warnUsingDefaults();
+
+ if (!opts.dryRun) {
+ const st = await kibanaFetch(config, "/api/status");
+ if (!st.ok) {
+ console.error("Cannot reach Kibana:", st.status, st.data);
+ process.exit(1);
+ }
+ console.log("Connected to Kibana", st.data?.version?.number || "?");
+ }
+
+ const plans = serviceHealthPlans(opts);
+ console.log(
+ `Creating ${plans.length} service-health jobs (prefix='${opts.prefix}', bucket_span='${opts.bucketSpan}', query_delay='${opts.queryDelay}')`,
+ );
+
+ let okCount = 0;
+ let failCount = 0;
+ for (const plan of plans) {
+ const ok = await upsertServiceHealthPlan(config, plan, opts.dryRun);
+ if (ok) okCount++;
+ else failCount++;
+ }
+
+ console.log(`\nDone: ${okCount} succeeded, ${failCount} failed`);
+ if (!opts.dryRun) {
+ console.log("Check jobs in Kibana: Stack Management > Machine Learning > Anomaly Detection Jobs");
+ }
+}
+
+// -----------------------------------------------------------------------------
+// Tool classification (skill tool_ids + agent JSON manifests)
+// -----------------------------------------------------------------------------
+
+function isWorkflowTool(toolId) {
+ if (WORKFLOW_TOOL_EXCLUSIONS.has(toolId)) return true;
+ if (toolId in FALLBACK_TOOLS) return true;
+ return WORKFLOW_PREFIXES.some((p) => toolId.startsWith(p));
+}
+
+/** ES|QL tool IDs that `tools register` would POST (same skip rules as cmdToolsRegister). */
+function getRegisterableEsqlToolIds() {
+ const ids = new Set();
+ for (const def of loadToolDefs()) {
+ const primaryBuiltin = FALLBACK_TOOLS[def.name];
+ if (primaryBuiltin && BUILTIN_TOOLS_SET.has(primaryBuiltin)) continue;
+ ids.add(transformToolDef(def).id);
+ }
+ return ids;
+}
+
+function loadAgentToolNames(jsonBasename) {
+ const path = join(AGENT_DIR, jsonBasename);
+ if (!existsSync(path)) return [];
+ try {
+ const data = JSON.parse(readFileSync(path, "utf8"));
+ return Array.isArray(data.tools) ? data.tools : [];
+ } catch {
+ return [];
+ }
+}
+
+function mergedCoreAgentToolNames() {
+ const a = loadAgentToolNames("anomaly_detective.json");
+ const b = loadAgentToolNames("anomaly_explainer.json");
+ const c = loadAgentToolNames("anomaly_maintainer.json");
+ return [...new Set([...a, ...b, ...c])];
+}
+
+/** Maps skill folder id → tool lists from references/kibana/agent/*.json (used only to derive skill tool_ids). */
+const SKILL_ID_AGENT_JSON = {
+ "investigate-anomaly": "anomaly_detective.json",
+ "explain-anomaly-results": "anomaly_explainer.json",
+ "troubleshoot-anomaly-detection-jobs": "anomaly_maintainer.json",
+ "manage-anomaly-detection-job": "anomaly_maintainer.json",
+};
+
+function toolNamesDeclaredForSkill(skillId) {
+ if (
+ skillId === "kibana-anomaly-detection" ||
+ skillId === "observability-anomaly-expert" ||
+ skillId === "security-anomaly-expert"
+ ) {
+ return mergedCoreAgentToolNames();
+ }
+ const agentJson = SKILL_ID_AGENT_JSON[skillId];
+ if (agentJson) return loadAgentToolNames(agentJson);
+ return [];
+}
+
+/**
+ * Full preference list: builtins (in order) then custom ES|QL tools from agent manifests that `tools register`
+ * created (workflow-backed tools omitted).
+ */
+function buildSkillToolIdsFallback(skillId, registeredEsqlIds, missingAccumulator) {
+ const declared = toolNamesDeclaredForSkill(skillId);
+ const nonWorkflow = declared.filter((t) => !isWorkflowTool(t));
+ const nonBuiltin = nonWorkflow.filter((t) => !BUILTIN_TOOLS_SET.has(t));
+ const attached = [];
+ for (const t of nonBuiltin) {
+ if (registeredEsqlIds.has(t)) attached.push(t);
+ else missingAccumulator.add(t);
+ }
+ return [...BUILTIN_TOOLS, ...attached];
+}
+
+const TOOL_ID_SCAN_RE = /\b(ad_[a-z][a-z0-9_]*)\b|\b(platform\.core\.[a-z0-9_]+)\b|\b(observability\.[a-z0-9_]+)\b/g;
+
+/**
+ * Tool ids appearing in *skillMarkdown* (order = first mention wins). Only includes ids Kibana can attach:
+ * builtins from this bundle, or registerable ES|QL tools (non-workflow).
+ */
+function extractMentionedToolIdsInOrder(skillMarkdown, registeredEsqlIds) {
+ const ordered = [];
+ const seen = new Set();
+ if (!skillMarkdown) return ordered;
+ let m;
+ TOOL_ID_SCAN_RE.lastIndex = 0;
+ while ((m = TOOL_ID_SCAN_RE.exec(skillMarkdown)) !== null) {
+ const id = m[1] || m[2] || m[3];
+ if (seen.has(id)) continue;
+ if (isWorkflowTool(id)) continue;
+ if (BUILTIN_TOOLS_SET.has(id)) {
+ seen.add(id);
+ ordered.push(id);
+ continue;
+ }
+ if (registeredEsqlIds.has(id)) {
+ seen.add(id);
+ ordered.push(id);
+ }
+ }
+ return ordered;
+}
+
+/**
+ * At most {@link MAX_SKILL_TOOL_IDS} tools: prefer those mentioned in the skill markdown (full SKILL.md text),
+ * then fill from the manifest-derived fallback list.
+ */
+function buildSkillToolIdsCapped(skillId, skillMarkdownScan, registeredEsqlIds, missingAccumulator) {
+ const mentioned = extractMentionedToolIdsInOrder(skillMarkdownScan, registeredEsqlIds);
+ const fallback = buildSkillToolIdsFallback(skillId, registeredEsqlIds, missingAccumulator);
+ const out = [];
+ const seen = new Set();
+ for (const id of mentioned) {
+ if (out.length >= MAX_SKILL_TOOL_IDS) break;
+ if (seen.has(id)) continue;
+ seen.add(id);
+ out.push(id);
+ }
+ for (const id of fallback) {
+ if (out.length >= MAX_SKILL_TOOL_IDS) break;
+ if (seen.has(id)) continue;
+ seen.add(id);
+ out.push(id);
+ }
+ return out;
+}
+
+function enrichSkillDefsWithToolIds(defs) {
+ const registered = getRegisterableEsqlToolIds();
+ const missing = new Set();
+ for (const def of defs) {
+ const scan = def._toolScanMarkdown ?? "";
+ delete def._toolScanMarkdown;
+ def.tool_ids = buildSkillToolIdsCapped(def.id, scan, registered, missing);
+ }
+ if (missing.size > 0) {
+ const sample = [...missing].sort().slice(0, 16).join(", ");
+ console.warn(
+ ` Note: ${missing.size} tool id(s) declared in agent JSON but not in ES|QL bundle (workflows / skipped fallbacks) — omitted from skill tool_ids: ${sample}${missing.size > 16 ? " …" : ""}`,
+ );
+ }
+ return defs;
+}
+
+async function cmdWorkflowsRegister(argv) {
+ let dryRun = false;
+ const cli = {};
+ for (let i = 0; i < argv.length; i++) {
+ const a = argv[i];
+ if (a === "--dry-run") dryRun = true;
+ else if (a === "--kibana-url") cli.kibanaUrl = argv[++i];
+ else if (a === "--username") cli.username = argv[++i];
+ else if (a === "--password") cli.password = argv[++i];
+ else if (a === "--api-key") {
+ cli.apiKeyFromCli = true;
+ cli.apiKey = argv[++i];
+ } else if (a === "--space-id") cli.spaceId = argv[++i];
+ else if (a === "--insecure") cli.insecure = true;
+ }
+
+ const config = getKibanaConfig(cli);
+ if (!validateConfig(config, { dryRun })) process.exit(1);
+ if (config.usingDefaults && !dryRun) warnUsingDefaults();
+
+ const files = readdirSync(WORKFLOWS_DIR)
+ .filter((f) => f.endsWith(".yaml") || f.endsWith(".yml"))
+ .sort();
+
+ if (files.length === 0) {
+ console.error("No YAML files found in", WORKFLOWS_DIR);
+ process.exit(1);
+ }
+
+ const workflows = files.map((f) => ({
+ _file: f,
+ yaml: readFileSync(join(WORKFLOWS_DIR, f), "utf8"),
+ }));
+
+ console.log(`Loaded ${workflows.length} workflow YAML files`);
+
+ if (dryRun) {
+ for (const { _file, yaml } of workflows) {
+ console.log(`\n--- ${_file} ---`);
+ console.log(yaml.slice(0, 200) + (yaml.length > 200 ? "\n ..." : ""));
+ }
+ return;
+ }
+
+ const st = await kibanaFetch(config, "/api/status");
+ if (!st.ok) {
+ console.error("Cannot reach Kibana:", st.status, st.data);
+ process.exit(1);
+ }
+ console.log("Connected to Kibana", st.data?.version?.number || "?");
+
+ const basePath = getBasePath(config);
+ const url = `${basePath}/api/workflows?overwrite=true`;
+ const payload = { workflows: workflows.map(({ yaml }) => ({ yaml })) };
+
+ const res = await kibanaHttpFetch(url, config, {
+ method: "POST",
+ headers: { ...getHeaders(config), "kbn-xsrf": "true" },
+ body: JSON.stringify(payload),
+ });
+
+ let body;
+ try {
+ body = await res.json();
+ } catch {
+ const text = await res.text().catch(() => "");
+ console.error(`Unexpected response ${res.status}: ${text.slice(0, 300)}`);
+ process.exit(1);
+ }
+
+ if (!res.ok) {
+ console.error(`Failed: ${res.status}`, JSON.stringify(body).slice(0, 400));
+ process.exit(1);
+ }
+
+ for (const { id, name } of body.created ?? []) {
+ console.log(` Registered: ${name} (${id})`);
+ }
+ for (const failure of body.failures ?? []) {
+ console.error(` Failed: ${JSON.stringify(failure)}`);
+ }
+
+ console.log(`\nDone: ${body.created?.length ?? 0} registered, ${body.failures?.length ?? 0} failed`);
+}
+
+async function cmdTest(config) {
+ if (!validateConfig(config, { dryRun: false })) process.exit(1);
+ if (config.usingDefaults) warnUsingDefaults();
+
+ const basePath = getBasePath(config);
+ const res = await kibanaHttpFetch(`${basePath}/api/status`, config, {
+ headers: { ...getHeaders(config), "kbn-xsrf": "true" },
+ });
+ const data = await res.json().catch(() => ({}));
+
+ if (!res.ok) {
+ console.error("Connection failed:", res.status, data);
+ process.exit(1);
+ }
+
+ const version = data.version?.number || "unknown";
+ console.log("Connected to Kibana", version);
+ console.log(" Base URL:", config.url);
+ console.log(" Space:", config.spaceId || "(default)");
+ console.log(" Auth:", config.apiKey ? "ApiKey" : "Basic");
+}
+
+// -----------------------------------------------------------------------------
+// Skill registration
+// -----------------------------------------------------------------------------
+
+function parseSkillFrontmatter(raw) {
+ const fmMatch = raw.match(/^---\n([\s\S]*?)\n---/);
+ if (!fmMatch) return {};
+ const lines = fmMatch[1].split("\n");
+ const result = {};
+ let i = 0;
+ while (i < lines.length) {
+ const keyMatch = lines[i].match(/^(\w+):\s*(.*)/);
+ if (!keyMatch) {
+ i++;
+ continue;
+ }
+ const key = keyMatch[1];
+ const valueStart = keyMatch[2].trim();
+ if (valueStart === ">-" || valueStart === ">") {
+ const parts = [];
+ i++;
+ while (i < lines.length && /^\s/.test(lines[i])) {
+ parts.push(lines[i].trim());
+ i++;
+ }
+ result[key] = parts.join(" ");
+ } else {
+ result[key] = valueStart;
+ i++;
+ }
+ }
+ return result;
+}
+
+function extractSkillContent(raw) {
+ const match = raw.match(/^---\n[\s\S]*?\n---\n([\s\S]*)/);
+ return match ? match[1].trim() : raw.trim();
+}
+
+function readSkillDefFromDir(skillDir, id) {
+ const mdPath = join(skillDir, "SKILL.md");
+ if (!existsSync(mdPath)) return null;
+ let raw;
+ try {
+ raw = readFileSync(mdPath, "utf8");
+ } catch (e) {
+ console.warn(`Skipping ${id}: ${e.message}`);
+ return null;
+ }
+ const fm = parseSkillFrontmatter(raw);
+ if (!fm.name && !fm.description) {
+ console.warn(`Skipping ${id}: no name/description in frontmatter`);
+ return null;
+ }
+ return {
+ id,
+ name: fm.name || id,
+ description: fm.description || "",
+ content: extractSkillContent(raw),
+ /** Full SKILL.md (for tool mention scan); stripped before POST. */
+ _toolScanMarkdown: raw,
+ };
+}
+
+function loadSkillDefs() {
+ const defs = [];
+
+ // Hub skill at plugin root (skills/kibana/kibana-anomaly-detection/SKILL.md)
+ const hubId = basename(PLUGIN_ROOT);
+ const hubDef = readSkillDefFromDir(PLUGIN_ROOT, hubId);
+ if (hubDef) defs.push(hubDef);
+
+ if (!existsSync(SKILLS_DIR)) {
+ console.warn(`Skills directory not found: ${SKILLS_DIR}`);
+ return defs;
+ }
+ for (const entry of readdirSync(SKILLS_DIR).sort()) {
+ const skillPath = join(SKILLS_DIR, entry);
+ const sub = readSkillDefFromDir(skillPath, entry);
+ if (sub) defs.push(sub);
+ }
+ return defs;
+}
+
+async function registerSkill(config, def, dryRun) {
+ const toolIds = def.tool_ids ?? [];
+ const payload = {
+ id: def.id,
+ name: def.name,
+ description: def.description,
+ content: def.content,
+ tool_ids: toolIds,
+ };
+ console.log(`Skill: ${payload.id} (${toolIds.length} tool_ids)`);
+
+ if (dryRun) {
+ const preview = {
+ ...payload,
+ content: payload.content.slice(0, 120) + (payload.content.length > 120 ? "..." : ""),
+ tool_ids: toolIds,
+ };
+ console.log(JSON.stringify(preview, null, 2));
+ return true;
+ }
+
+ const basePath = getBasePath(config);
+ const skillsUrl = `${basePath}/api/agent_builder/skills`;
+ const skillByIdUrl = `${skillsUrl}/${encodeURIComponent(payload.id)}`;
+ const headers = { ...getHeaders(config), "kbn-xsrf": "true" };
+
+ /** PUT body — path carries skill id (see Kibana Agent Builder API). */
+ const updateBody = {
+ name: payload.name,
+ description: payload.description,
+ content: payload.content,
+ tool_ids: toolIds,
+ };
+
+ let res = await kibanaHttpFetch(skillsUrl, config, {
+ method: "POST",
+ headers,
+ body: JSON.stringify(payload),
+ });
+
+ if (res.ok) {
+ console.log(` Registered: ${payload.id}`);
+ return true;
+ }
+
+ const text = await res.text();
+ const alreadyExists =
+ res.status === 409 ||
+ (res.status === 400 && /already exists|duplicate/i.test(text)) ||
+ /already exists/i.test(text);
+
+ if (alreadyExists) {
+ console.log(` Already exists — updating (PUT): ${payload.id}`);
+ const putRes = await kibanaHttpFetch(skillByIdUrl, config, {
+ method: "PUT",
+ headers,
+ body: JSON.stringify(updateBody),
+ });
+ if (putRes.ok) {
+ console.log(` Updated: ${payload.id}`);
+ return true;
+ }
+ const putText = await putRes.text();
+ if (agentBuilderPutUnsupported(putRes.status, putText)) {
+ console.log(` PUT unsupported — deleting and re-creating: ${payload.id}`);
+ await kibanaHttpFetch(skillByIdUrl, config, { method: "DELETE", headers });
+ const recRes = await kibanaHttpFetch(skillsUrl, config, {
+ method: "POST",
+ headers,
+ body: JSON.stringify(payload),
+ });
+ if (recRes.ok) {
+ console.log(` Re-created: ${payload.id}`);
+ return true;
+ }
+ const recText = await recRes.text();
+ console.error(` Failed: ${recRes.status} ${recText.slice(0, 200)}`);
+ return false;
+ }
+ console.error(` Failed to update skill: ${putRes.status} ${putText.slice(0, 200)}`);
+ return false;
+ }
+
+ console.error(` Failed: ${res.status} ${text.slice(0, 200)}`);
+ return false;
+}
+
+/** True if object looks like an Agent Builder agent record (GET /api/agent_builder/agents/{id}). */
+function looksLikeAgentRecord(o) {
+ if (!o || typeof o !== "object") return false;
+ if (typeof o.id === "string") return true;
+ if (typeof o.name === "string") return true;
+ if (o.configuration != null && typeof o.configuration === "object") return true;
+ if (Array.isArray(o.skill_ids)) return true;
+ return false;
+}
+
+/** Normalize GET /api/agent_builder/agents/{id} JSON (envelope or raw agent). */
+function unwrapAgentPayload(data) {
+ if (!data || typeof data !== "object") return null;
+
+ const direct = looksLikeAgentRecord(data) ? data : null;
+ if (direct) return direct;
+
+ const nestedKeys = ["agent", "item", "attributes", "record", "result"];
+ for (const key of nestedKeys) {
+ const v = data[key];
+ if (looksLikeAgentRecord(v)) return v;
+ }
+
+ const d = data.data;
+ if (d != null && typeof d === "object") {
+ if (looksLikeAgentRecord(d)) return d;
+ if (Array.isArray(d) && d.length === 1 && looksLikeAgentRecord(d[0])) return d[0];
+ }
+
+ for (const v of Object.values(data)) {
+ if (looksLikeAgentRecord(v)) return v;
+ if (Array.isArray(v) && v.length && looksLikeAgentRecord(v[0])) return v[0];
+ if (v != null && typeof v === "object" && !Array.isArray(v)) {
+ for (const inner of Object.values(v)) {
+ if (looksLikeAgentRecord(inner)) return inner;
+ }
+ }
+ }
+
+ return null;
+}
+
+function buildAgentPutBody(agent, configuration) {
+ const payload = { configuration };
+ if (agent.name != null) payload.name = agent.name;
+ if (agent.description != null) payload.description = agent.description;
+ if (agent.avatar_color != null) payload.avatar_color = agent.avatar_color;
+ if (agent.avatar_symbol != null) payload.avatar_symbol = agent.avatar_symbol;
+ if (agent.labels != null) payload.labels = agent.labels;
+ if (agent.visibility != null) payload.visibility = agent.visibility;
+ return payload;
+}
+
+/**
+ * Build agent `configuration` from GET response (supports nested `configuration` or flat fields).
+ * Some Kibana versions expose skill_ids / workflow_ids / enable_elastic_capabilities at the top level.
+ */
+function extractAgentConfiguration(agent) {
+ const cfg = {
+ ...(agent.configuration && typeof agent.configuration === "object" ? agent.configuration : {}),
+ };
+ if (!Array.isArray(cfg.skill_ids) && Array.isArray(agent.skill_ids)) {
+ cfg.skill_ids = [...agent.skill_ids];
+ }
+ if (cfg.enable_elastic_capabilities === undefined && typeof agent.enable_elastic_capabilities === "boolean") {
+ cfg.enable_elastic_capabilities = agent.enable_elastic_capabilities;
+ }
+ if (!Array.isArray(cfg.workflow_ids) && Array.isArray(agent.workflow_ids)) {
+ cfg.workflow_ids = [...agent.workflow_ids];
+ }
+ if (!Array.isArray(cfg.skill_ids)) cfg.skill_ids = [];
+ if (!Array.isArray(cfg.workflow_ids)) cfg.workflow_ids = [];
+ return cfg;
+}
+
+/**
+ * Merge bundle skill IDs into the default Elastic AI Agent (PUT /api/agent_builder/agents/{id}).
+ * Requires agentBuilder:manageAgents on top of skill/tool registration privileges.
+ */
+async function attachSkillsToDefaultAgent(config, skillIds, agentId, dryRun) {
+ const basePath = getBasePath(config);
+ const agentUrl = `${basePath}/api/agent_builder/agents/${encodeURIComponent(agentId)}`;
+ const headers = { ...getHeaders(config), "kbn-xsrf": "true" };
+
+ if (dryRun) {
+ const merged = [...new Set(skillIds)];
+ const previewCfg = {
+ skill_ids: merged,
+ enable_elastic_capabilities: DEFAULT_AGENT_ENABLE_ELASTIC_CAPABILITIES === true,
+ workflow_ids: [],
+ };
+ console.log(`PUT ${agentUrl} (dry-run — no GET; preview configuration merge)`);
+ console.log(JSON.stringify({ configuration: previewCfg }, null, 2));
+ return true;
+ }
+
+ const getRes = await kibanaHttpFetch(agentUrl, config, { method: "GET", headers });
+ const { text: getText, parsed: payload } = await readJsonBody(getRes);
+
+ if (!getRes.ok) {
+ const snippet =
+ payload != null && typeof payload === "object" ? JSON.stringify(payload).slice(0, 280) : getText.slice(0, 280);
+ console.warn(` Could not GET agent "${agentId}" (${getRes.status}): ${snippet.slice(0, 220)}`);
+ console.warn(` Skipping default-agent skill attachment (needs read_onechat / agent read on agents).`);
+ return false;
+ }
+
+ const agent = unwrapAgentPayload(payload);
+ if (!agent || typeof agent !== "object") {
+ const keys =
+ payload != null && typeof payload === "object" ? Object.keys(payload).join(", ") : getText.slice(0, 80);
+ console.warn(
+ ` Unexpected GET agent response shape (keys/snippet: ${keys}) — skipping default-agent skill attachment`,
+ );
+ return false;
+ }
+
+ const cfg = extractAgentConfiguration(agent);
+ const existing = [...cfg.skill_ids];
+ const merged = [...new Set([...existing, ...skillIds])];
+ const missing = skillIds.filter((id) => !existing.includes(id));
+
+ const enableWasNotTrue = DEFAULT_AGENT_ENABLE_ELASTIC_CAPABILITIES && cfg.enable_elastic_capabilities !== true;
+ const needsSkillMerge = missing.length > 0;
+
+ if (!needsSkillMerge && !enableWasNotTrue) {
+ console.log(` Default agent "${agentId}" already includes all bundle skills and elastic capabilities are enabled`);
+ return true;
+ }
+
+ if (DEFAULT_AGENT_ENABLE_ELASTIC_CAPABILITIES) {
+ cfg.enable_elastic_capabilities = true;
+ }
+ cfg.skill_ids = merged;
+
+ const putBody = buildAgentPutBody(agent, cfg);
+
+ const reasons = [];
+ if (needsSkillMerge) reasons.push(`adding skills: ${missing.join(", ")}`);
+ if (enableWasNotTrue) reasons.push("enable_elastic_capabilities → true");
+
+ console.log(`Default agent: updating "${agentId}" (${reasons.join("; ")})`);
+
+ const putRes = await kibanaHttpFetch(agentUrl, config, {
+ method: "PUT",
+ headers,
+ body: JSON.stringify(putBody),
+ });
+
+ if (putRes.ok) {
+ console.log(` Updated agent "${agentId}": ${merged.length} skill_id(s) total`);
+ return true;
+ }
+
+ const { text: putErrRaw, parsed: putPayload } = await readJsonBody(putRes);
+ const errText =
+ putPayload != null && typeof putPayload === "object"
+ ? JSON.stringify(putPayload).slice(0, 450)
+ : putErrRaw.slice(0, 450);
+ console.error(` PUT agent "${agentId}" failed (${putRes.status}): ${errText.slice(0, 350)}`);
+ console.error(` (Requires privileges to update agents, e.g. agentBuilder / manage agents.)`);
+ return false;
+}
+
+async function cmdSkillsRegister(argv) {
+ let dryRun = false;
+ let skipDefaultAgent = false;
+ let defaultAgentId = DEFAULT_AGENT_ID;
+ const cli = {};
+ for (let i = 0; i < argv.length; i++) {
+ const a = argv[i];
+ if (a === "--dry-run") dryRun = true;
+ else if (a === "--skip-default-agent") skipDefaultAgent = true;
+ else if (a === "--default-agent-id") defaultAgentId = argv[++i];
+ else if (a === "--kibana-url") cli.kibanaUrl = argv[++i];
+ else if (a === "--username") cli.username = argv[++i];
+ else if (a === "--password") cli.password = argv[++i];
+ else if (a === "--api-key") {
+ cli.apiKeyFromCli = true;
+ cli.apiKey = argv[++i];
+ } else if (a === "--space-id") cli.spaceId = argv[++i];
+ else if (a === "--insecure") cli.insecure = true;
+ }
+
+ const config = getKibanaConfig(cli);
+ if (!validateConfig(config, { dryRun })) process.exit(1);
+ if (config.usingDefaults && !dryRun) warnUsingDefaults();
+
+ const defs = enrichSkillDefsWithToolIds(loadSkillDefs());
+ console.log(
+ `Loaded ${defs.length} skill definitions (tool_ids capped at ${MAX_SKILL_TOOL_IDS}; skill text first, then manifest fallback)`,
+ );
+
+ if (defs.length === 0) {
+ console.error(`No skills found under ${PLUGIN_ROOT} or ${SKILLS_DIR}`);
+ process.exit(1);
+ }
+
+ if (!dryRun) {
+ const st = await kibanaFetch(config, "/api/status");
+ if (!st.ok) {
+ console.error("Cannot reach Kibana:", st.status, st.data);
+ process.exit(1);
+ }
+ console.log("Connected to Kibana", st.data?.version?.number || "?");
+ }
+
+ let succeeded = 0,
+ failed = 0;
+ const succeededSkillIds = [];
+ for (const def of defs) {
+ const ok = await registerSkill(config, def, dryRun);
+ if (ok) {
+ succeeded++;
+ succeededSkillIds.push(def.id);
+ } else {
+ failed++;
+ }
+ }
+
+ if (!dryRun) {
+ console.log(`\nRegistration complete: ${succeeded} succeeded, ${failed} failed`);
+ }
+
+ if (!skipDefaultAgent && succeededSkillIds.length > 0) {
+ console.log("\nAttaching registered skills to default agent…");
+ await attachSkillsToDefaultAgent(config, succeededSkillIds, defaultAgentId, dryRun);
+ } else if (!skipDefaultAgent && succeededSkillIds.length === 0 && !dryRun) {
+ console.log("\nSkipping default-agent attachment (no skills registered successfully).");
+ }
+}
+
+/** Register tools, then workflows, then skills (skills attach `tool_ids` for tools POSTed in step 1). */
+async function cmdAllRegister(argv) {
+ console.log("=== 1/3 tools register ===\n");
+ await cmdToolsRegister(argv);
+ console.log("\n=== 2/3 workflows register ===\n");
+ await cmdWorkflowsRegister(argv);
+ console.log("\n=== 3/3 skills register (+ default agent) ===\n");
+ await cmdSkillsRegister(argv);
+ console.log("\n=== all register complete ===");
+}
+
+function printUsage() {
+ console.log(`
+Kibana Agent Builder (anomaly-detection)
+
+Commands:
+ test GET /api/status — verify URL and credentials
+ tools register POST all ES|QL tools to Agent Builder
+ workflows register POST all YAML workflow definitions to Kibana Workflows engine
+ skills register POST hub + skills/; PUT default agent configuration (skill_ids merge, enable_elastic_capabilities, workflow_ids)
+ all register Run tools register, workflows register, skills register (+ default agent)
+ jobs create-service-health Create baseline service issue-detection ML jobs + datafeeds
+
+Options (workflows/tools/skills/all register):
+ --dry-run Print JSON payloads only
+ --skip-default-agent Do not PUT skill_ids on the default Elastic AI Agent
+ --default-agent-id ID Agent id to update (default: from agent_builder_constants.json, usually elastic-ai-agent)
+ --kibana-url URL
+ --username USER
+ --password PASS
+ --api-key KEY
+ --space-id ID
+ --insecure Skip TLS verification (also KIBANA_INSECURE=true)
+
+Options (jobs create-service-health):
+ --prefix ID Job ID prefix (default: svc)
+ --metrics-index PAT Metrics index pattern (default: metrics-*)
+ --logs-index PAT Logs index pattern (default: logs-*)
+ --apm-index PAT APM index pattern (default: apm-*)
+ --bucket-span SPAN Bucket span (default: 15m)
+ --query-delay DELAY Datafeed query_delay (default: 120s)
+ --memory-limit SIZE model_memory_limit (default: 256mb)
+
+Environment variables match kibana-dashboards.js — see script header.
+`);
+}
+
+async function main() {
+ const argv = process.argv.slice(2);
+ if (argv.length === 0 || ["-h", "--help", "help"].includes(argv[0])) {
+ printUsage();
+ process.exit(argv.length === 0 ? 1 : 0);
+ }
+
+ const [cmd, sub] = argv;
+ if (cmd === "test") {
+ await cmdTest(getKibanaConfig({}));
+ return;
+ }
+ if (cmd === "tools" && sub === "register") {
+ await cmdToolsRegister(argv.slice(2));
+ return;
+ }
+ if (cmd === "workflows" && sub === "register") {
+ await cmdWorkflowsRegister(argv.slice(2));
+ return;
+ }
+ if (cmd === "skills" && sub === "register") {
+ await cmdSkillsRegister(argv.slice(2));
+ return;
+ }
+ if (cmd === "all" && sub === "register") {
+ await cmdAllRegister(argv.slice(2));
+ return;
+ }
+ if (cmd === "jobs" && sub === "create-service-health") {
+ await cmdJobsCreateServiceHealth(argv.slice(2));
+ return;
+ }
+
+ console.error(`Unknown command: ${cmd}${sub ? ` ${sub}` : ""}`);
+ printUsage();
+ process.exit(1);
+}
+
+main().catch((e) => {
+ console.error(e);
+ process.exit(1);
+});
diff --git a/plugins/kibana/skills/kibana-dashboards/SKILL.md b/plugins/kibana/skills/kibana-dashboards/SKILL.md
index 7472155..4a9e3aa 100644
--- a/plugins/kibana/skills/kibana-dashboards/SKILL.md
+++ b/plugins/kibana/skills/kibana-dashboards/SKILL.md
@@ -6,7 +6,7 @@ description: >
deployment.
metadata:
author: elastic
- version: 0.1.1
+ version: 0.1.2
---
# Kibana Dashboards and Visualizations
@@ -231,8 +231,8 @@ scrolling. Design for density—place primary KPIs and key trends above the fold
| `region_map` | Region/choropleth maps | Yes |
| `pie`, `treemap`, `mosaic`, `waffle` | Partition charts | Yes |
-> **Note:** To create donut charts, use `pie` with `donut_hole` set to `"s"`, `"m"`, or `"l"` (small, medium, large
-> hole). Use `"none"` for a solid pie.
+> **Note:** To create donut charts, use `pie` with `styling.donut_hole` set to `"s"`, `"m"`, or `"l"` (small, medium,
+> large hole). Use `"none"` for a solid pie. Example: `"styling": { "donut_hole": "m" }`.
### Dataset Types
@@ -325,7 +325,7 @@ For detailed schemas and all chart type options, see [Chart Types Reference](ref
{
"title": "Top Hosts",
"type": "xy",
- "axis": { "x": { "title": { "visible": false } }, "y": { "anchor": "start", "title": { "visible": false } } },
+ "axis": { "x": { "title": { "visible": false } }, "y": { "title": { "visible": false } } },
"layers": [
{
"type": "bar_horizontal",
@@ -345,7 +345,7 @@ For detailed schemas and all chart type options, see [Chart Types Reference](ref
"type": "xy",
"axis": {
"x": { "title": { "visible": false }, "scale": "temporal", "domain": { "type": "fit", "rounding": false } },
- "y": { "anchor": "start", "title": { "visible": false } }
+ "y": { "title": { "visible": false } }
},
"layers": [
{
diff --git a/plugins/kibana/skills/kibana-dashboards/assets/bar-chart-esql.json b/plugins/kibana/skills/kibana-dashboards/assets/bar-chart-esql.json
index 5e85649..51995b8 100644
--- a/plugins/kibana/skills/kibana-dashboards/assets/bar-chart-esql.json
+++ b/plugins/kibana/skills/kibana-dashboards/assets/bar-chart-esql.json
@@ -3,7 +3,7 @@
"type": "xy",
"axis": {
"x": { "title": { "visible": false } },
- "y": { "anchor": "start", "title": { "visible": false } }
+ "y": { "title": { "visible": false } }
},
"layers": [
{
diff --git a/plugins/kibana/skills/kibana-dashboards/assets/demo-dashboard.json b/plugins/kibana/skills/kibana-dashboards/assets/demo-dashboard.json
index 2da503c..c5eba97 100644
--- a/plugins/kibana/skills/kibana-dashboards/assets/demo-dashboard.json
+++ b/plugins/kibana/skills/kibana-dashboards/assets/demo-dashboard.json
@@ -91,7 +91,7 @@
"type": "xy",
"axis": {
"x": { "title": { "visible": false } },
- "y": { "anchor": "start", "title": { "visible": false } }
+ "y": { "title": { "visible": false } }
},
"layers": [
{
@@ -118,7 +118,7 @@
"type": "xy",
"axis": {
"x": { "title": { "visible": false } },
- "y": { "anchor": "start", "title": { "visible": false } }
+ "y": { "title": { "visible": false } }
},
"layers": [
{
@@ -195,7 +195,7 @@
"type": "xy",
"axis": {
"x": { "title": { "visible": false } },
- "y": { "anchor": "start", "title": { "visible": false } }
+ "y": { "title": { "visible": false } }
},
"layers": [
{
@@ -223,7 +223,7 @@
"type": "xy",
"axis": {
"x": { "title": { "visible": false } },
- "y": { "anchor": "start", "title": { "visible": false } }
+ "y": { "title": { "visible": false } }
},
"layers": [
{
diff --git a/plugins/kibana/skills/kibana-dashboards/assets/ecommerce-analytics-dashboard.json b/plugins/kibana/skills/kibana-dashboards/assets/ecommerce-analytics-dashboard.json
index 8a32f0a..c4eb1b5 100644
--- a/plugins/kibana/skills/kibana-dashboards/assets/ecommerce-analytics-dashboard.json
+++ b/plugins/kibana/skills/kibana-dashboards/assets/ecommerce-analytics-dashboard.json
@@ -123,7 +123,6 @@
}
},
"y": {
- "anchor": "start",
"title": {
"visible": false
}
@@ -170,7 +169,6 @@
}
},
"y": {
- "anchor": "start",
"title": {
"visible": false
}
@@ -217,7 +215,6 @@
}
},
"y": {
- "anchor": "start",
"title": {
"visible": false
}
@@ -265,7 +262,6 @@
}
},
"y": {
- "anchor": "start",
"title": {
"visible": false
}
@@ -313,7 +309,6 @@
}
},
"y": {
- "anchor": "start",
"title": {
"visible": false
}
@@ -453,7 +448,6 @@
}
},
"y": {
- "anchor": "start",
"title": {
"visible": false
}
@@ -500,7 +494,6 @@
}
},
"y": {
- "anchor": "start",
"title": {
"visible": false
}
diff --git a/plugins/kibana/skills/kibana-dashboards/assets/line-chart-timeseries.json b/plugins/kibana/skills/kibana-dashboards/assets/line-chart-timeseries.json
index c0886fc..53b5cfc 100644
--- a/plugins/kibana/skills/kibana-dashboards/assets/line-chart-timeseries.json
+++ b/plugins/kibana/skills/kibana-dashboards/assets/line-chart-timeseries.json
@@ -3,7 +3,7 @@
"type": "xy",
"axis": {
"x": { "title": { "visible": false }, "scale": "temporal", "domain": { "type": "fit", "rounding": false } },
- "y": { "anchor": "start", "title": { "visible": false } }
+ "y": { "title": { "visible": false } }
},
"layers": [
{
diff --git a/plugins/kibana/skills/kibana-dashboards/references/chart-types-reference.md b/plugins/kibana/skills/kibana-dashboards/references/chart-types-reference.md
index 1468262..3342ab9 100644
--- a/plugins/kibana/skills/kibana-dashboards/references/chart-types-reference.md
+++ b/plugins/kibana/skills/kibana-dashboards/references/chart-types-reference.md
@@ -11,8 +11,8 @@ Complete schema reference for each supported chart type via the Kibana dashboard
- `tag_cloud` — Tag/word cloud
- `data_table` — Data tables
- `region_map` — Region/choropleth maps
-- `pie`, `treemap`, `mosaic`, `waffle` — Partition charts (use `pie` with `donut_hole` for donuts: `"s"`, `"m"`, or
- `"l"`)
+- `pie`, `treemap`, `mosaic`, `waffle` — Partition charts (use `pie` with `styling.donut_hole` for donuts: `"s"`, `"m"`,
+ or `"l"`)
## DataView Aggregation Operations
@@ -323,8 +323,8 @@ For ES|QL, uses `metrics` and `rows` arrays. Each entry uses `{ column: "..." }`
Partition charts display parts of a whole. Uses a flat structure (no `layers`) with `metrics` for the slice sizes and
`group_by` for the rings or groupings. The schema is identical for all partition types—simply change `"type": "pie"` to
-`"treemap"`, `"mosaic"`, or `"waffle"`. To create a donut, use `"type": "pie"` with `"donut_hole"` set to `"s"`, `"m"`,
-or `"l"`.
+`"treemap"`, `"mosaic"`, or `"waffle"`. To create a donut, use `"type": "pie"` with `"styling": { "donut_hole": "m" }`.
+Valid `donut_hole` values are `"none"`, `"s"`, `"m"`, or `"l"`.
**ES|QL Example:**
diff --git a/plugins/kibana/skills/kibana-dashboards/references/dashboard-api-reference.md b/plugins/kibana/skills/kibana-dashboards/references/dashboard-api-reference.md
index 1d79229..8e9b7cb 100644
--- a/plugins/kibana/skills/kibana-dashboards/references/dashboard-api-reference.md
+++ b/plugins/kibana/skills/kibana-dashboards/references/dashboard-api-reference.md
@@ -369,7 +369,7 @@ node scripts/kibana-dashboards.js dashboard create dashboard.json
| Datatable structure | ES\|QL data_table requires `metrics` + `rows` arrays |
| XY chart fails | Put `data_source` inside each layer (for both dataView and ES\|QL) |
| Heatmap property names | Heatmap uses `x`, `y`, `metric` for axes and value |
-| XY axis config | Use `axis` (singular); `y` with `anchor: "start"` for left axis |
+| XY axis config | Use `axis` (singular); `y` for left axis, `y2` for right axis |
| ref_id panels missing | Prefer inline definitions (properties in `config`) over `ref_id` |
## Testing from Dev Tools
diff --git a/plugins/kibana/skills/kibana-dashboards/scripts/kibana-dashboards.js b/plugins/kibana/skills/kibana-dashboards/scripts/kibana-dashboards.js
index 08ce0d2..85a3124 100644
--- a/plugins/kibana/skills/kibana-dashboards/scripts/kibana-dashboards.js
+++ b/plugins/kibana/skills/kibana-dashboards/scripts/kibana-dashboards.js
@@ -143,7 +143,6 @@ function getHeaders(config) {
const headers = {
"Content-Type": "application/json",
"kbn-xsrf": "true",
- "x-elastic-internal-origin": "kibana",
"User-Agent": "elastic-agentic",
};
diff --git a/plugins/observability/.claude-plugin/plugin.json b/plugins/observability/.claude-plugin/plugin.json
index 6950bb8..acad7a6 100644
--- a/plugins/observability/.claude-plugin/plugin.json
+++ b/plugins/observability/.claude-plugin/plugin.json
@@ -1,5 +1,5 @@
{
- "version": "0.2.4",
+ "version": "0.3.0",
"name": "elastic-observability",
"description": "Elastic Observability skills - OpenTelemetry instrumentation and migration (.NET, Java, Python), LLM observability, log search, SLOs, and service health",
"author": {
diff --git a/plugins/observability/plugin.json b/plugins/observability/plugin.json
index 582717c..e895ad8 100644
--- a/plugins/observability/plugin.json
+++ b/plugins/observability/plugin.json
@@ -1,7 +1,7 @@
{
"name": "elastic-observability",
"description": "Elastic Observability skills - OpenTelemetry instrumentation and migration (.NET, Java, Python), LLM observability, log search, SLOs, and service health",
- "version": "0.2.4",
+ "version": "0.3.0",
"author": {
"name": "Elastic",
"url": "https://www.elastic.co"
diff --git a/plugins/observability/skills/k8s-investigation/SKILL.md b/plugins/observability/skills/k8s-investigation/SKILL.md
new file mode 100644
index 0000000..33c5903
--- /dev/null
+++ b/plugins/observability/skills/k8s-investigation/SKILL.md
@@ -0,0 +1,465 @@
+---
+name: observability-k8s-investigation
+description: >
+ Investigate Kubernetes workload, node, and control-plane issues using OTel telemetry
+ (EDOT). Use when diagnosing pod failures (CrashLoopBackOff, OOMKilled, Error), node
+ pressure, resource exhaustion, image pull failures, admission rejections, autoscaling
+ anomalies, or correlating K8s state with application signals. OTel ingest path only
+ — the legacy ECS Kubernetes integration shape is out of scope.
+metadata:
+ author: elastic
+ version: 0.2.0
+---
+
+# Kubernetes Investigation
+
+Diagnose Kubernetes issues using OTel telemetry collected via EDOT (Elastic Distribution of OpenTelemetry) and the
+kube-stack collector. Correlate cluster state, pod runtime metrics, K8s events, application logs, and APM to identify
+root cause across the workload, node, and control-plane layers.
+
+## Scope
+
+**In scope:** OTel-receiver-namespaced indices (`metrics-kubeletstatsreceiver.otel-*`,
+`metrics-k8sclusterreceiver.otel-*`, `logs-k8seventsreceiver.otel-*`, `logs-k8sobjectsreceiver.otel-*`) and OTel
+semantic conventions (`k8s.pod.name`, `k8s.namespace.name`, `k8s.container.restarts`).
+
+**Out of scope:**
+
+- The legacy Elastic Agent Kubernetes integration (`metrics-kubernetes.*`, `logs-kubernetes.*`, `kubernetes.*` fields).
+ Being deprecated — do not author queries against these paths.
+- APM-layer analysis (service SLO breaches, transaction error rates, upstream dependency health). Different domain —
+ once a K8s root cause is ruled in or out, APM investigation continues outside this skill.
+- Cluster provisioning, capacity planning, cost optimization. Different domain.
+
+## Guidelines
+
+These apply to every investigation. When in doubt, re-read them before writing the synthesis.
+
+**Absence of evidence is not evidence. Do not confabulate from empty results.** If log queries return 0 rows, logs are
+likely not collected or the pod has no recent lines — this does _not_ mean "dependency unavailable" or any other
+specific failure mode. Report `no_logs_available` and weight remaining signals accordingly.
+
+**Empty dependency data ≠ upstream healthy.** Services without APM instrumentation (load generators, workers) emit no
+destination metrics. Report `insufficient_dependency_data`, not "upstreams OK."
+
+**Co-symptoms are not causes.** Two services degrading simultaneously usually share an upstream, not a causal link. Only
+attribute causation when (a) one service's degradation clearly precedes the other's, and (b) the delta is large (>5×
+error rate, >3× latency).
+
+**OOMKilled ≠ memory leak by default.** The limit may simply be undersized for the workload's working set. Compare
+against a 7-day baseline at the same hour-of-day before claiming a leak.
+
+**Error-termination ≠ application bug by default.** Check `k8s.pod.cpu_limit_utilization` first. CFS throttling driving
+liveness probe timeouts is the most common misdiagnosis in this space.
+
+**Average CPU hides throttling.** A pod can look healthy at 40–60% average `cpu_limit_utilization` while being throttled
+severely at p99. Linux enforces CPU limits in 100ms periods; bursty workloads hit quota mid-period and stall. Look at
+max and p95, not just average.
+
+**Restart count is boolean, not a counter.** `k8s.container.restarts` is pulled directly from the K8s API and may be
+pruned by the kubelet at any time, so the absolute value is unreliable. Treat it as `== 0` (no recent restarts) vs `> 0`
+(recently restarting); do not derive backoff timing or "linear vs exponential" patterns from it. Confirm the restart
+pattern via K8s `Killing` / `BackOff` events instead.
+
+**Prefer to report uncertainty over manufacturing confidence.** If the evidence is ambiguous, the synthesis should say
+so. Competing hypotheses are a valid output.
+
+## Indices and fields
+
+### Where to look
+
+| Signal | Index pattern | Use |
+| --------------------- | --------------------------------------------------- | ------------------------------------------------------------------- |
+| Pod/container runtime | `metrics-kubeletstatsreceiver.otel-*` | CPU, memory, network, filesystem. Utilization ratios. |
+| Cluster state | `metrics-k8sclusterreceiver.otel-*` | Restarts, phase, last-terminated reason, HPA, quota, node condition |
+| K8s events | `logs-k8seventsreceiver.otel-*` | Killing, BackOff, FailedScheduling, Evicted, image pull events |
+| K8s object snapshots | `logs-k8sobjectsreceiver.otel-*` | Deployment/service/configmap state over time |
+| Application logs | `logs-*.otel-*` | `body.text`, `severity_text`, filtered by `k8s.pod.name` |
+| APM | `traces-*.otel-*`, `metrics-service_*.otel-default` | Correlate via `service.name` + K8s resource attrs |
+| ML anomalies | `.ml-anomalies-*` | Memory-growth, restart-rate, throttle jobs (if configured) |
+
+### Key fields
+
+Flat OTel paths work in ES|QL. Prefer the flat form for readability; the nested `resource.attributes.*` form is for raw
+log documents only.
+
+| Field | Index | What it is |
+| ------------------------------------------------ | --------------------------- | ------------------------------------------------------- |
+| `k8s.pod.name` | all k8s | Pod name |
+| `k8s.namespace.name` | all k8s | Namespace |
+| `k8s.container.name` | all k8s | Container within pod |
+| `k8s.deployment.name` | k8sclusterreceiver + others | Parent deployment |
+| `k8s.pod.phase` | k8sclusterreceiver | Pending=1/Running=2/Succeeded=3/Failed=4/Unknown=5 |
+| `k8s.container.restarts` | k8sclusterreceiver | Total container restart count |
+| `k8s.container.status.last_terminated_reason` | k8sclusterreceiver | `OOMKilled`, `Error`, `Completed`, `ContainerCannotRun` |
+| `k8s.pod.status_reason` | k8sclusterreceiver | Pod-level reason (`Evicted`, `NodeLost`) |
+| `k8s.pod.memory_limit_utilization` | kubeletstatsreceiver | 0.0–1.0+ (can exceed 1 transiently before OOM) |
+| `k8s.pod.cpu_limit_utilization` | kubeletstatsreceiver | 0.0–N (frequently >1 under CFS throttling) |
+| `k8s.pod.memory.usage` / `.working_set` | kubeletstatsreceiver | Bytes |
+| `k8s.node.condition_memory_pressure` | k8sclusterreceiver | 1 = pressure, 0 = ok |
+| `k8s.node.condition_ready` | k8sclusterreceiver | 0 = NotReady |
+| `k8s.hpa.current_replicas` / `.desired_replicas` | k8sclusterreceiver | HPA state |
+| `attributes.k8s.event.reason` | k8seventsreceiver | Event reason (filter on this) |
+| `body.text` | k8seventsreceiver / logs | Event message / log message |
+| `k8s.object.name` | k8seventsreceiver | involvedObject name (log attribute, use flat form) |
+
+### Field availability
+
+Several fields above are off by default in stock kube-stack collectors and require explicit configuration. Verify
+presence before relying on them; if absent, fall back as noted and call out the substitution in the synthesis.
+
+| Field | Why it might be missing | Fall-back |
+| ------------------------------------------------------------ | ---------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------- |
+| `k8s.container.status.last_terminated_reason` | Optional metric in k8sclusterreceiver; gated behind `metrics_collected.metadata` config. | Infer from K8s `Killing` / `OOMKilling` events in `logs-k8seventsreceiver.otel-*` and exit codes in app logs. |
+| `k8s.pod.status_reason` | Same — optional metric on k8sclusterreceiver. | Infer from events: `Evicted`, `NodeLost`, `Preempted`. |
+| `k8s.pod.cpu_limit_utilization` / `memory_limit_utilization` | Only emitted when the pod has the corresponding limit set, and the kubeletstatsreceiver metric is enabled. | Compute manually as `k8s.pod.cpu.usage / ` from k8sclusterreceiver, or use absolute usage trending against a baseline. |
+| `k8s.node.condition_memory_pressure` | Gated behind k8sclusterreceiver `node_conditions_to_report` (default omits this). | Compare `k8s.node.memory.usage` against `k8s.node.allocatable_memory`, or look for `Evicted` events on the node. |
+
+If a fall-back is used, note it in the synthesis (e.g. `(via memory.usage; limit_utilization not collected)`) so the
+reader knows the signal is indirect.
+
+## ES|QL gotchas
+
+Before writing queries, know these. Each of them silently produces wrong answers rather than failing loudly.
+
+**`VALUES()` returns scalar for single distinct value, array for multiple.** Templating that assumes array shape (e.g.
+`| first`) extracts the first character of the string when scalar. Use `MV_FIRST(VALUES(...))` or handle both.
+
+**`PERCENTILE` does not work on OTel `histogram` type** (as of 8.15). For APM duration percentiles, use `AVG` on the
+`aggregate_metric_double` summary field (`AVG(transaction.duration.summary)` divides sum by value_count). For true
+percentiles, fall back to Kibana Query DSL.
+
+**`COUNT(agg_metric_double)` returns `value_count` (events), not doc count.** `SUM(field)` gives the sum component;
+`AVG(field)` gives sum/value_count. Do not use `SUM(transaction.duration.summary)` as an event-count proxy — it returns
+total duration.
+
+**K8s metrics use flat OTel field paths in ES|QL.** `k8s.pod.name`, not `resource.attributes.k8s.pod.name`. The nested
+form is for raw log documents.
+
+## Failure-mode taxonomy
+
+Vocabulary for classification, not a decision tree. Use the pivotal-signal column to recognize which mode you're looking
+at; use "Investigate" to know what else should corroborate.
+
+### Workload layer
+
+| Mode | Pivotal signal | Investigate |
+| ----------------------------------- | -------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| **OOMKilled** | `last_terminated_reason == "OOMKilled"` + `memory_limit_utilization → 1.0` | Monotonic rise (leak) vs. load-driven spike? Compare current trend to 7-day baseline. Check heap metrics (JVM, Go, Node) for GC pressure. |
+| **CPU throttling → Error exit** | `cpu_limit_utilization > 1.0` + `last_terminated_reason == "Error"` | Liveness/readiness probe timeouts from CFS throttling. Average CPU can look fine (40–60%) while p99 throttle is severe. Check probe timeouts vs observed startup/health latency. |
+| **Liveness probe misconfiguration** | Restarts without resource pressure; `initialDelaySeconds` < startup time | K8s events show `Unhealthy` / `Killing`. `kubectl logs --previous` typically shows healthy startup before kill. |
+| **CrashLoopBackOff (generic)** | `BackOff` events + rising `k8s.container.restarts` | Branch on `last_terminated_reason` — this is a meta-mode. OOMKilled → memory path; Error → logs + throttling; ContainerCannotRun → image/exec. |
+| **ImagePullBackOff** | K8s events `Failed` with image name + `429` or `not found` | Registry rate limit? Missing tag? Wrong imagePullSecret? Check recency of `Pulling`/`Pulled` events. |
+| **Stuck rollout** | New pods `Pending`/not-Ready > `progressDeadlineSeconds`; old pods still serving | Check `k8s.deployment.available` vs `.desired`. Admission rejection? Readiness probe failing on new pods? HPA not scaling? |
+| **Termination signal race** | Brief 5xx bursts correlated with rolling deploys | Endpoint removal races termination. New requests can hit the pod after SIGTERM starts. NGINX gotcha: `STOPSIGNAL SIGTERM` triggers _fast_ shutdown, not graceful — use `STOPSIGNAL SIGQUIT` for graceful drain. Check ingress 502 rate vs rollout timing. |
+
+### Node layer
+
+| Mode | Pivotal signal | Investigate |
+| ----------------------------------- | ----------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------- |
+| **Node NotReady cascade** | `k8s.node.condition_ready == 0` + mass `Evicted` events | Memory pressure? Disk pressure? Network partition from API server? Inspect kubelet logs, `k8s.node.condition_*` history. |
+| **Resource eviction** | `status_reason == "Evicted"` + `condition_memory_pressure == 1` on node | Node-level noisy neighbor. QoS order: BestEffort → Burstable → Guaranteed. Identify which pod drove node memory up. |
+| **Node affinity/selector conflict** | Mass unschedulable pods after label change | K8s events show `FailedScheduling`. Often triggered by cluster upgrades (e.g. `node-role.kubernetes.io/master` → `control-plane`). |
+
+### Control plane
+
+| Mode | Pivotal signal | Investigate |
+| ----------------------------- | ------------------------------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------- |
+| **etcd I/O cascade** | API server latency spike + cluster-wide kubelet heartbeat failures | Disk IOPS, fsync latency (must be <10ms). Cloud-burst-credit exhaustion is common. |
+| **Admission webhook block** | Mass `FailedCreate` across namespaces; deployments frozen | `failurePolicy:Fail` webhook pod crashed. Check webhook pod health + API server TCP connection cache (caches dead connections ~15 min). |
+| **Priority preemption storm** | Production pods terminating with `preempted-by` annotation | New `PriorityClass` with `globalDefault:true` caused cascade. Check `kube-scheduler` events. |
+| **PDB drain deadlock** | Node drain stuck indefinitely; HTTP 429 from Eviction API | PDB `minAvailable`/`maxUnavailable` too strict. No default drain timeout. Manual PDB deletion unblocks. |
+
+### Autoscaling & admission
+
+| Mode | Pivotal signal | Investigate |
+| ----------------------------- | ------------------------------------------------------------------ | -------------------------------------------------------------------------------------------------------------------------------------------------- |
+| **HPA unready-pod dampening** | Load rising, HPA not scaling; unready pods included in calculation | HPA averages CPU across all replicas including unready (0% contribution). Check `k8s.hpa.current_replicas` vs `.desired_replicas` + pod readiness. |
+| **Resource quota silent 403** | Deployment stuck at n-1/n; `FailedCreate` on ReplicaSet | Namespace quota exhausted (often CronJob accumulation). Check `k8s.resource_quota.used` vs `.hard_limit`. |
+
+### Networking
+
+| Mode | Pivotal signal | Investigate |
+| --------------------------- | -------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------- |
+| **StatefulSet split-brain** | Duplicate pod identities across partitioned nodes | Network partition + eviction timeout race. Two instances of same ordinal running. No fencing by default. |
+| **CoreDNS OOMKill** | CoreDNS restarts + cluster-wide DNS timeouts in app logs | Default CoreDNS memory (~170Mi) insufficient under query amplification (ndots:5, each external lookup → ~10 lookups). |
+
+### When classification is ambiguous
+
+Real incidents often match two modes. Examples:
+
+- OOMKilled pod with simultaneous CPU throttling — memory usually drives the kill, but verify by checking whether memory
+ or CPU hit limit first.
+- Stuck rollout with HPA dampening and resource quota near-exhaustion — both can freeze a deploy. Check which constraint
+ is binding.
+- Node NotReady with pods that were already crashing — the node issue may be incidental.
+
+When two modes fit, name both in the synthesis and say which one you believe is causal and why. Do not force a single
+hypothesis when the evidence supports two.
+
+## Signal interpretation
+
+### Memory
+
+- **Monotonic rise over 30–60 min** → leak. Check GC metrics for the language: JVM `jvm.gc.duration`, Go
+ `process.runtime.go.gc.pause_ns`, Node `v8js_gc_duration`. Rising GC frequency/pause with stable live-set is the
+ canonical leak signature.
+- **Diurnal / load-correlated spikes** → load-driven, not leak. Consider HPA tuning or limit increase.
+- **Hits 1.0, then restart** → OOMKilled confirmed. Exit code 137 (SIGKILL) in app logs consistent.
+
+### CPU
+
+- `cpu_limit_utilization > 1.0` sustained → CFS throttling. Node has spare CPU; the pod is quota-blocked.
+- Symptoms of throttling (not the throttle metric itself): liveness probe timeouts, p99 latency 4–16× p50, queue
+ backpressure upstream, Error-reason container terminations.
+- Average can look healthy while p95 is throttled. Do not trust average alone.
+
+### Restart patterns
+
+- `restarts > 0` recently → workload has been restarting. Don't read magnitude into the count (see _Restart count is
+ boolean_); confirm the pattern from K8s `Killing` / `BackOff` event timestamps in `logs-k8seventsreceiver.otel-*`.
+- Restarts correlated with memory pressure (`memory_limit_utilization → 1.0`) → OOMKilled path.
+- Restarts without memory/CPU pressure → probe misconfig, app bug, or startup dependency failure. Pull events for
+ `Unhealthy` and `Killing`.
+
+### Termination reasons
+
+- `OOMKilled` → memory path.
+- `Error` → non-zero exit. Check app logs; if empty/minimal, check CPU throttling before attributing to app logic.
+- `Completed` → ran to completion. Normal for Jobs/CronJobs/init containers; anomalous otherwise.
+- `ContainerCannotRun` → runtime/image/exec issue. Check image pull events.
+
+## Investigation flow
+
+> An investigation is not a checklist. The sections below describe a _typical_ arc — **compress, skip, or revisit them
+> based on what you find.** Terminate as soon as you have enough evidence to synthesize at a known confidence. Chasing
+> signals past the point of diminishing returns is a failure mode, not thoroughness.
+
+### Orient
+
+Resolve the target: `k8s.pod.name`, `k8s.namespace.name`, optionally `k8s.deployment.name` and `service.name`. If no
+time window is given, default to the last hour for pod-level investigations, last 2 hours for event correlation, last 6
+hours for ongoing/unresolved incidents.
+
+If the alert payload already tells you the failure mode (e.g., it fires specifically on `OOMKilled`), note that and skip
+classification; move to confirmation and baseline comparison.
+
+### Characterize
+
+Get the shape of the workload's recent behavior: restart count, termination reasons, phase, utilization. One or two
+queries usually suffice.
+
+```esql
+FROM metrics-k8sclusterreceiver.otel-*
+| WHERE k8s.pod.name == "" AND k8s.namespace.name == ""
+ AND @timestamp > NOW() - 1 hour
+| STATS restarts = MAX(k8s.container.restarts),
+ term_reasons = VALUES(k8s.container.status.last_terminated_reason),
+ phase = MAX(k8s.pod.phase)
+```
+
+```esql
+FROM metrics-kubeletstatsreceiver.otel-*
+| WHERE k8s.pod.name == "" AND @timestamp > NOW() - 15 minutes
+| STATS mem_pct = ROUND(MAX(k8s.pod.memory_limit_utilization) * 100, 1),
+ cpu_pct = ROUND(MAX(k8s.pod.cpu_limit_utilization) * 100, 1)
+```
+
+### Classify
+
+Use the taxonomy. The pivotal signal should match; the "Investigate" column tells you what corroboration to seek.
+
+When two modes fit, note both and proceed with the one that has the stronger pivotal signal. You may revise during
+corroboration.
+
+### Corroborate
+
+Pull the evidence your classification predicts you'll find. Typical sources:
+
+**K8s events** for the namespace and window:
+
+```esql
+FROM logs-k8seventsreceiver.otel-*
+| WHERE k8s.namespace.name == ""
+ AND @timestamp > NOW() - 2 hours
+ AND attributes.k8s.event.reason IN (
+ "BackOff", "Killing", "Unhealthy", "Failed",
+ "FailedScheduling", "Evicted", "SuccessfulRescale",
+ "Pulling", "Pulled", "Started", "Created"
+ )
+| SORT @timestamp DESC
+| KEEP @timestamp, attributes.k8s.event.reason, body.text, k8s.object.name
+| LIMIT 30
+```
+
+**Application logs** if available — look at the 200 most recent lines before the termination timestamp. If absent, flag
+`no_logs_available`; do not invent a log pattern.
+
+**APM** if the pod runs an instrumented service — resolve `service.name` from pod resource attributes for later
+correlation. SLO / latency / error-rate analysis itself is APM-layer work and out of scope for this skill.
+
+**Baseline comparison** — for utilization-based findings, compare current values to 7-day-prior at the same hour-of-day.
+"High memory" is meaningful only relative to what's normal for this workload.
+
+### Check for upstream cause (conditional)
+
+Only pursue if the symptom pattern suggests it. Threshold: upstream error rate >5× baseline _or_ latency >3× baseline,
+AND degradation started before the symptom on the target service. Co-symptoms do not establish causation.
+
+If `metrics-service_destination.1m.otel-default` has no rows for the service, report `insufficient_dependency_data` —
+not "upstreams healthy."
+
+### Check for recent change (conditional)
+
+`SuccessfulCreate` / `Pulled` events in the last 2 hours often correlate with deploys. `logs-k8sobjectsreceiver.otel-*`
+shows configmap/secret/deployment spec changes. A change within 15 minutes of the symptom onset is a strong correlation,
+but still a correlation — verify it plausibly explains the mode you've classified.
+
+### Synthesize and stop
+
+Synthesize as soon as you have enough evidence to support a hypothesis at known confidence. You do not need to complete
+every section above — investigation terminates when either:
+
+- You have a high-confidence hypothesis with corroboration, or
+- You have a low/medium-confidence hypothesis and further queries are unlikely to change the picture (e.g., logs are
+ unavailable, APM isn't instrumented, no recent changes found).
+
+## Synthesis
+
+Default structure:
+
+```text
+HYPOTHESIS (confidence: high | medium | low)
+
+
+EVIDENCE
+-
+-
+-
+
+CONFIDENCE NOTE
+
+
+RECOMMENDED NEXT STEPS
+1.
+2.
+
+DOWNSTREAM IMPACT
+
+```
+
+**When two hypotheses are live:** replace HYPOTHESIS with COMPETING HYPOTHESES; list both, say which you lean toward and
+why, and list the evidence that would disambiguate them.
+
+**When no incident is found** (symptom resolved, or alert appears spurious): say so directly.
+`ALERT FIRED BUT SYSTEM APPEARS HEALTHY` is a valid output. List what you checked and what you didn't find.
+
+### Confidence calibration
+
+Start at **high** and downgrade based on what's missing:
+
+- Downgrade to **medium** if: primary signal is clear but corroboration is missing (no logs, no APM, no baseline
+ comparison possible). Or: two modes fit and you can't disambiguate.
+- Downgrade to **low** if: only a single signal supports the hypothesis, signals conflict, or the mode requires evidence
+ you couldn't fetch.
+
+Never return **high** when application log data was absent and the hypothesis depends on application behavior. Absence
+of evidence does not corroborate a hypothesis.
+
+## Query recipes
+
+### Most-restarting pods in a namespace
+
+```esql
+FROM metrics-k8sclusterreceiver.otel-*
+| WHERE k8s.namespace.name == "" AND @timestamp > NOW() - 1 hour
+| STATS restarts = MAX(k8s.container.restarts) BY k8s.pod.name, k8s.container.status.last_terminated_reason
+| WHERE restarts > 0
+| SORT restarts DESC
+| LIMIT 20
+```
+
+### CPU throttling check for a pod
+
+```esql
+FROM metrics-kubeletstatsreceiver.otel-*
+| WHERE k8s.pod.name == "" AND @timestamp > NOW() - 30 minutes
+| STATS max_cpu_ratio = ROUND(MAX(k8s.pod.cpu_limit_utilization), 2),
+ avg_cpu_ratio = ROUND(AVG(k8s.pod.cpu_limit_utilization), 2),
+ max_cpu_cores = ROUND(MAX(k8s.pod.cpu.usage), 3)
+```
+
+Sustained ratio >1.0 = throttling. Transient >1.0 with avg <0.5 is usually benign burst.
+
+### Nodes under memory pressure (right now)
+
+```esql
+FROM metrics-k8sclusterreceiver.otel-*
+| WHERE @timestamp > NOW() - 15 minutes AND k8s.node.condition_memory_pressure == 1
+| STATS ts = MAX(@timestamp) BY k8s.node.name
+| SORT ts DESC
+```
+
+### Admission denials (webhook or quota) last hour
+
+```esql
+FROM logs-k8seventsreceiver.otel-*
+| WHERE @timestamp > NOW() - 1 hour
+ AND (attributes.k8s.event.reason == "FailedCreate"
+ OR body.text LIKE "*admission webhook*"
+ OR body.text LIKE "*exceeded quota*")
+| SORT @timestamp DESC
+| KEEP @timestamp, k8s.namespace.name, attributes.k8s.event.reason, body.text
+| LIMIT 30
+```
+
+### Firing K8s alerts
+
+```text
+GET /api/alerting/rules/_find?search=k8s&search_fields=tags&filter=alert.attributes.executionStatus.status:active
+```
+
+## Examples
+
+### "Why is my pod CrashLoopBackOff-ing?"
+
+Characterize first: get restart count, termination reason, memory and CPU utilization.
+
+- If `last_terminated_reason == "OOMKilled"` and memory utilization hit 1.0 → memory path. Corroborate with 7-day
+ baseline: monotonic rise over days = leak; spiky = load-driven. Check GC metrics if language is known.
+- If `last_terminated_reason == "Error"` and `cpu_limit_utilization > 1.0` → CPU throttling path. Corroborate with
+ liveness probe config (initialDelaySeconds, timeoutSeconds) and K8s events for `Unhealthy`.
+- If `last_terminated_reason == "Error"` and CPU is fine → application-logic path. Pull recent logs before termination.
+- If `last_terminated_reason == "ContainerCannotRun"` → image/exec path. Check K8s events for `Failed` pull events.
+
+Synthesize with appropriate confidence. If logs were unavailable on the Error path, downgrade to medium and say so.
+
+### "Is my rollout stuck?"
+
+Authoritative signal: `k8s.deployment.available < k8s.deployment.desired` for > 10 minutes.
+
+Diagnose the constraint:
+
+- K8s events on the new ReplicaSet: `FailedCreate` → admission rejection (quota, webhook, PSP). `FailedScheduling` → no
+ node fits.
+- New-pod utilization: all at 0% memory → never started (image pull failure); high CPU with low memory → slow startup
+ hitting readiness probe.
+- HPA state: stable `current_replicas < desired_replicas` under load → unready-pod dampening.
+
+### "Alert fired but everything looks healthy"
+
+Possible and worth naming explicitly. Check:
+
+- Has the symptom resolved? Compare current utilization/restart rate to the alert trigger point.
+- Was the alert a transient spike that's already decayed?
+- Is the alert tuned appropriately (e.g., too-short evaluation window)?
+
+Output: `ALERT FIRED BUT SYSTEM APPEARS HEALTHY` with what you checked. Recommend alert tuning if the pattern is
+recurrent.
+
+## Related
+
+- **Workflow:** `K8s CrashLoopBackOff Investigation` — alert-triggered automated version of the pod-level path above.
+ Runs deterministic ESQL + branches; this skill provides the interpretation layer the workflow lacks.
+- **Forge genome library:** 16 K8s failure scenarios (OOMKill cascade, CPU throttling, probe misconfig, node NotReady,
+ admission webhook block, etc.) validating this skill's coverage.
diff --git a/plugins/security/.claude-plugin/plugin.json b/plugins/security/.claude-plugin/plugin.json
index 5ab3529..ad67618 100644
--- a/plugins/security/.claude-plugin/plugin.json
+++ b/plugins/security/.claude-plugin/plugin.json
@@ -1,5 +1,5 @@
{
- "version": "0.2.4",
+ "version": "0.3.0",
"name": "elastic-security",
"description": "Elastic Security skills - alert triage, case management, detection rule management, and sample data generation",
"author": {
diff --git a/plugins/security/plugin.json b/plugins/security/plugin.json
index 4c2cf02..d74dea1 100644
--- a/plugins/security/plugin.json
+++ b/plugins/security/plugin.json
@@ -1,7 +1,7 @@
{
"name": "elastic-security",
"description": "Elastic Security skills - alert triage, case management, detection rule management, and sample data generation",
- "version": "0.2.4",
+ "version": "0.3.0",
"author": {
"name": "Elastic",
"url": "https://www.elastic.co"
@@ -9,12 +9,5 @@
"repository": "https://github.com/elastic/agent-skills",
"homepage": "https://github.com/elastic/agent-skills/tree/main/plugins/security",
"license": "Apache-2.0",
- "keywords": [
- "security",
- "siem",
- "detection",
- "alerts",
- "cases",
- "elastic"
- ]
+ "keywords": ["security", "siem", "detection", "alerts", "cases", "elastic"]
}
diff --git a/skills/elasticsearch/elasticsearch-esql/SKILL.md b/skills/elasticsearch/elasticsearch-esql/SKILL.md
index 175d901..4d03d81 100644
--- a/skills/elasticsearch/elasticsearch-esql/SKILL.md
+++ b/skills/elasticsearch/elasticsearch-esql/SKILL.md
@@ -6,7 +6,7 @@ description: >
charts and dashboards from ES|QL results.
metadata:
author: elastic
- version: 0.1.1
+ version: 0.3.0
---
# Elasticsearch ES|QL
@@ -120,11 +120,24 @@ node scripts/esql.js test
curl -s "$ELASTICSEARCH_URL//_settings/index.mode" -H "Authorization: ApiKey $ELASTICSEARCH_API_KEY"
```
+ For TSDS indices on 9.4+, prefer the in-language discovery commands `METRICS_INFO` and `TS_INFO` (both GA) over
+ inspecting mappings — they enumerate the metric catalogue and the dimension labels of each time series directly. Both
+ must follow `TS` and must precede `STATS`/`SORT`/`LIMIT`. See
+ [Time Series Queries](references/time-series-queries.md#metric-and-time-series-discovery).
+
+ ```bash
+ node scripts/esql.js raw "TS metrics-tsds | METRICS_INFO | SORT metric_name" --tsv
+ node scripts/esql.js raw "TS metrics-tsds | TS_INFO | KEEP metric_name, dimensions | SORT metric_name" --tsv
+ ```
+
3. **Choose the right ES|QL feature for the task**: Before writing queries, match the user's intent to the most
appropriate ES|QL feature. Prefer a single advanced query over multiple basic ones.
- "find patterns," "categorize," "group similar messages" → `CATEGORIZE(field)`
- "spike," "dip," "anomaly," "when did X change" → `CHANGE_POINT value ON key`
- "trend over time," "time series" → `STATS ... BY BUCKET(@timestamp, interval)` or `TS` for TSDB
+ - "PromQL", "Prometheus query/dashboard/alert", `sum by (instance) (...)`, label matchers like `{cluster="prod"}` →
+ `PROMQL` source command (9.4+ preview); see [PROMQL Command](references/promql-command.md). Prefer `TS` for native
+ ES|QL phrasing.
- "search," "find documents matching" → `MATCH` (default), `QSTR` (advanced boolean), `KQL` (Kibana migration). For
content/document relevance search, follow the [ES|QL Search Strategy](references/esql-search-strategy.md)
- "count," "average," "breakdown" → `STATS` with aggregation functions
@@ -134,6 +147,8 @@ node scripts/esql.js test
CIDR_MATCH), common templates, and ambiguity handling
- [Time Series Queries](references/time-series-queries.md) - **read before any TS query**: inner/outer aggregation
model, TBUCKET syntax, RATE constraints
+ - [PROMQL Command](references/promql-command.md) — **read before any PROMQL query**: options, output schema,
+ limitations, and `PROMQL` vs `TS` decision matrix (9.4+ preview)
- [ES|QL Complete Reference](references/esql-reference.md) - full syntax for all commands and functions
- [ES|QL Search Strategy](references/esql-search-strategy.md) — for content/document relevance search (retrieve →
fuse → rerank)
@@ -281,6 +296,25 @@ TS metrics-tsds
| SORT bucket
```
+**Time series with PromQL syntax (9.4+ preview):** Use the `PROMQL` source command when the user explicitly asks for
+PromQL, references Prometheus syntax (`sum by (instance) (...)`, label matchers like `{cluster="prod"}`), or is
+migrating a Prometheus dashboard or alert. The `PROMQL` command accepts standard PromQL with optional `index`, `step`,
+`buckets`, `start`, `end`, and `scrape_interval` options, and produces a table that the rest of the ES|QL pipeline can
+process. Range selectors are optional — when omitted, the window is `max(step, scrape_interval)`. Otherwise prefer `TS`
+(GA in 9.4). `PROMQL` does **not** support group modifiers, set operators (`or`/`and`/`unless`), or functions like
+`histogram_quantile`, `predict_linear`, and `label_join` — fall back to `TS` for those. See
+[PROMQL Command](references/promql-command.md) for the full reference.
+
+```esql
+// Adaptive Kibana query — date picker drives time range and step
+PROMQL index=metrics-* sum by (instance) (rate(http_requests_total))
+
+// Named result, post-processed with ES|QL
+PROMQL index=k8s step=1h bytes=(max by (cluster) (network.bytes_in))
+| STATS max_bytes = MAX(bytes) BY cluster
+| SORT cluster
+```
+
**Data enrichment with LOOKUP JOIN:** The basic `ON` clause matches fields by name in both indices
(`LOOKUP JOIN idx ON field_name`). When the join key has a different name in the source, use `RENAME` first to align
names. 9.2+ tech preview also supports expression predicates (`ON expr == expr`); see
@@ -351,6 +385,7 @@ For complete ES|QL syntax including all commands, functions, and operators, read
- [Query Patterns](references/query-patterns.md) - Natural language to ES|QL translation
- [Generation Tips](references/generation-tips.md) - Best practices for query generation
- [Time Series Queries](references/time-series-queries.md) - TS command, time series aggregation functions, TBUCKET
+- [PROMQL Command](references/promql-command.md) - PromQL source command for TSDS indices (9.4+ preview)
- [DSL to ES|QL Migration](references/dsl-to-esql-migration.md) - Convert Query DSL to ES|QL
- [Environment Setup](references/environment-setup.md) - Connection configuration options
diff --git a/skills/elasticsearch/elasticsearch-esql/references/dsl-to-esql-migration.md b/skills/elasticsearch/elasticsearch-esql/references/dsl-to-esql-migration.md
index e9ae400..e138a91 100644
--- a/skills/elasticsearch/elasticsearch-esql/references/dsl-to-esql-migration.md
+++ b/skills/elasticsearch/elasticsearch-esql/references/dsl-to-esql-migration.md
@@ -958,28 +958,29 @@ FROM sales
Features not available in ES|QL as of version 9.3:
-| Feature | Query DSL | ES\|QL |
-| ---------------------------- | --------- | ----------------------------------- |
-| Highlighting | ✅ | ❌ |
-| Nested queries | ✅ | ❌ |
-| Parent-child queries | ✅ | ❌ |
-| Scroll/pagination beyond 10k | ✅ | ❌ |
-| Percolate queries | ✅ | ❌ |
-| Complex boosting | ✅ | Limited |
-| Geo distance sorting | ✅ | ❌ |
-| Runtime fields | ✅ | Use EVAL |
-| Suggest API | ✅ | ❌ |
-| Collapse (field collapsing) | ✅ | ❌ |
-| Inner hits | ✅ | ❌ |
-| Timezone in date functions | ✅ | ❌ (UTC only) |
-| JOIN (non-lookup) | N/A | ❌ (only LEFT JOIN on lookup index) |
+| Feature | Query DSL | ES\|QL |
+| ---------------------------- | --------- | ----------------------------------------- |
+| Highlighting | ✅ | ❌ |
+| Nested queries | ✅ | ❌ |
+| Parent-child queries | ✅ | ❌ |
+| Scroll/pagination beyond 10k | ✅ | ❌ |
+| Percolate queries | ✅ | ❌ |
+| Complex boosting | ✅ | Limited |
+| Geo distance sorting | ✅ | ❌ |
+| Runtime fields | ✅ | Use EVAL |
+| Suggest API | ✅ | ❌ |
+| Collapse (field collapsing) | ✅ | ❌ |
+| Inner hits | ✅ | ❌ |
+| Timezone support | ✅ | ✅ `SET time_zone` (Serverless GA) |
+| JOIN (non-lookup) | N/A | ❌ (only LEFT JOIN on lookup index) |
+| Subqueries / UNION ALL | N/A | ✅ `FROM` subqueries (Serverless preview) |
### Unsupported Field Types in ES|QL
- `nested`
- `binary`
- `completion`
-- `flattened`
+- `flattened` (use `METADATA _source` + `JSON_EXTRACT` to access sub-keys)
- Range types (`date_range`, `integer_range`, etc.)
- `rank_feature`, `rank_features`
- `search_as_you_type`
diff --git a/skills/elasticsearch/elasticsearch-esql/references/esql-reference.md b/skills/elasticsearch/elasticsearch-esql/references/esql-reference.md
index 662323a..3597fac 100644
--- a/skills/elasticsearch/elasticsearch-esql/references/esql-reference.md
+++ b/skills/elasticsearch/elasticsearch-esql/references/esql-reference.md
@@ -52,23 +52,26 @@ Query directives modify the behavior of an ES|QL query. They appear before the s
### SET (9.3+, tech preview)
-Controls query-level settings.
+Controls query-level settings. Every `SET` directive must end with a semicolon before the source command.
**Syntax:**
```esql
-SET setting = value; [SET settingN = valueN;]
+SET setting = "value"; [SET setting = "value";]
source-command
| processing-commands
```
**`unmapped_fields`** (9.3+ preview) -- controls how unmapped fields are treated:
-- `FAIL` (default) -- the query fails if it references unmapped fields
-- `NULLIFY` -- treats unmapped fields as null values
+- `"default"` / `"fail"` -- the query fails if it references unmapped fields
+- `"nullify"` -- treats unmapped fields as null values
+- `"load"` -- loads unmapped fields dynamically. **Limitation:** `"load"` is incompatible with subqueries and views. Use
+ `"nullify"` when composing subqueries or querying views.
**`time_zone`** (Serverless GA; self-managed planned) -- sets the default timezone for the query, overriding UTC
-default.
+default. Accepts any IANA timezone string or UTC offset. Applies to all date/time operations: `DATE_TRUNC`,
+`DATE_FORMAT`, `NOW()`, etc.
**Examples:**
@@ -79,6 +82,12 @@ FROM employees
| SORT emp_no
| LIMIT 1
+SET time_zone = "America/Los_Angeles";
+FROM error_triage
+| EVAL hour = DATE_TRUNC(1 hour, @timestamp)
+| STATS errors = COUNT(*) BY hour, service
+| SORT hour DESC
+
SET time_zone = "+05:00";
TS k8s
| WHERE @timestamp == "2024-05-10T00:04:49.000Z"
@@ -86,7 +95,11 @@ TS k8s
```
> **When to use:** `unmapped_fields` is useful when querying across multiple indices where some indices may not have all
-> fields mapped. `time_zone` shifts date functions and display to a non-UTC zone.
+> fields mapped. `time_zone` shifts date functions and display to a non-UTC zone. There is no per-function timezone
+> argument — `DATE_TRUNC(1 hour, @timestamp, "America/Los_Angeles")` does **not** work.
+>
+> **Restriction:** `SET` directives cannot be used inside view definitions. The caller must apply `SET` when querying
+> the view.
---
@@ -123,6 +136,33 @@ FROM
FROM cluster_one:logs-*, cluster_two:logs-*
```
+**Subqueries (Serverless tech preview):** `FROM` supports parenthesized subqueries with UNION ALL semantics. Each branch
+is a complete ES|QL pipeline. Columns present in one branch but not another are filled with `null`.
+
+```esql
+// Combine logs from different indices with independent pipelines
+FROM
+ (FROM web_logs
+ | WHERE status_code >= 500
+ | KEEP @timestamp, message, service.name),
+ (FROM app_logs
+ | WHERE level == "error"
+ | KEEP @timestamp, message, service.name)
+| STATS errors = COUNT(*) BY service.name
+
+// Mix bare index patterns and subqueries
+FROM raw_index, (FROM other_index | WHERE active == true | KEEP id, name)
+```
+
+**Subquery constraints:**
+
+- Non-correlated only — branches cannot reference columns from the outer query
+- Columns with the same name must have compatible types across branches
+- `FORK` cannot be used inside or after subqueries
+- `SET unmapped_fields="load"` is incompatible with subqueries
+
+**Subqueries vs FORK:** Different data sources → subqueries. Same data, different analyses → FORK.
+
**Note:** Without explicit `LIMIT`, queries default to 1000 rows (or whatever the cluster setting
esql.query.result_truncation_default_size is set to).
@@ -147,7 +187,7 @@ ROW greeting = "hello", pi = 3.14159
### TS
Retrieves data from time series data streams (TSDS). Similar to `FROM` but enables time series aggregation functions in
-`STATS` and targets only time series indices. Available since 9.2.
+`STATS` and targets only time series indices. **Preview from 9.2 to 9.3, GA since 9.4**; GA on Elastic Cloud Serverless.
**Syntax:**
@@ -161,6 +201,11 @@ TS index_pattern [METADATA fields]
- Time series functions are evaluated per time series first, then aggregated by group using an outer function
- If no inner time series function is specified, `LAST_OVER_TIME()` is assumed implicitly
- Cannot be combined with `FORK` before `STATS` is applied
+- When the query has no `STATS`, `TS` returns rows sorted by `@timestamp` descending by default
+- When the first `STATS` after `TS` uses a **bare** time series function (not wrapped in an outer aggregation like
+ `AVG()` / `SUM()`), results are implicitly grouped by every dimension and include a `_timeseries` JSON column. Use
+ `BY WITHOUT(dim, ...)` (GA in 9.4) to narrow this grouping. Bare dimension columns in `BY` are rejected; only grouping
+ functions (`TBUCKET`, `WITHOUT`) are allowed alongside a bare time series function.
**Examples:**
@@ -177,6 +222,11 @@ TS metrics
// Average of per-time-series averages (explicit inner function)
TS metrics
| STATS AVG(AVG_OVER_TIME(memory_usage))
+
+// Bare time series function — group by every dimension except `pod` (9.4+ GA)
+TS k8s
+| STATS total_cost = SUM(network.cost) BY WITHOUT(pod)
+| SORT total_cost
```
**Best practices:**
@@ -185,6 +235,69 @@ TS metrics
- Use `TS` instead of `FROM` for aggregations on time series data
- Avoid aggregating metrics with different dimensional cardinalities in the same query
+### PROMQL
+
+Queries time series data streams (TSDS) using **Prometheus Query Language (PromQL)** instead of ES|QL syntax. Like `TS`,
+it produces a table that the rest of the ES|QL pipeline can process. Available since **9.4 (preview)** and on Elastic
+Cloud Serverless. See [promql-command.md](promql-command.md) for the full reference.
+
+**Syntax:**
+
+```esql
+PROMQL [ ... ] [ = ] ( )
+```
+
+**Options:**
+
+- `index` — indices/streams/aliases (default `metrics-*`)
+- `step` — query resolution step width
+- `buckets` — target bucket count for auto-step (default `100`, mutually exclusive with `step`)
+- `start`, `end` — explicit time range (defaults to Kibana date picker, otherwise unrestricted)
+- `scrape_interval` — expected metric collection interval (default `1m`); used for the implicit range selector window
+- `=( ... )` — name the metric output column
+
+**Output columns:**
+
+- The PromQL expression (or ``) as `double` — the metric value
+- `step` (`date`) — timestamp for each evaluation step
+- One `keyword` column per `by`/`without` grouping label, or a single `_timeseries` JSON column when there is no
+ cross-series aggregation
+
+**Examples:**
+
+```esql
+// Fully adaptive Kibana query — date picker drives time range and step
+PROMQL index=metrics-* sum by (instance) (rate(http_requests_total))
+
+// Explicit range query with a named result column
+PROMQL index=k8s step=1h cost=(max by (cluster) (network.total_bytes_in{cluster!="prod"}))
+| SORT cluster
+
+// Post-process with ES|QL after the PROMQL stage
+PROMQL index=k8s step=1h bytes=(max by (cluster) (network.bytes_in))
+| STATS max_bytes = MAX(bytes) BY cluster
+| SORT cluster
+
+// Enrich PromQL results with a lookup index
+PROMQL index=metrics-*
+ http_rate=(sum by (instance) (rate(http_requests_total)))
+| LOOKUP JOIN instance_metadata ON instance
+```
+
+**Implicit range selectors:** Range vector functions can omit the range selector (`rate(http_requests_total)` instead of
+`rate(http_requests_total[5m])`); the engine uses `max(step, scrape_interval)` as the window. This makes the query scale
+with the date picker.
+
+**Limitations (9.4 preview):**
+
+- Group modifiers (`on(...) group_left(...)`) are not supported
+- Set operators (`or`, `and`, `unless`) are not supported
+- Some PromQL functions are not available, including `histogram_quantile`, `predict_linear`, and `label_join`
+- Time buckets align to fixed calendar boundaries rather than the query start time, which can cause slight differences
+ from native Prometheus for short ranges or large step sizes
+
+When any of these are required, use the [`TS` command](#ts) and express the equivalent computation in ES|QL.
+
### SHOW
Returns information about the deployment.
@@ -464,12 +577,13 @@ FROM data
### LIMIT
-Limits the number of rows returned.
+Limits the number of rows returned. Supports optional grouped top-N with `BY` (Serverless).
**Syntax:**
```esql
LIMIT number
+LIMIT number BY field
```
**Examples:**
@@ -478,8 +592,16 @@ LIMIT number
FROM logs-*
| SORT @timestamp DESC
| LIMIT 100
+
+// Grouped top-N: keep top 3 rows per service after sorting
+FROM app_logs
+| STATS cnt = COUNT(*) BY service, level
+| SORT cnt DESC
+| LIMIT 3 BY service
```
+> **Note:** In `LIMIT n BY field`, the number comes **before** `BY`. `LIMIT BY field n` does not parse.
+
### DISSECT
Extracts structured fields from a string using a pattern.
@@ -857,65 +979,242 @@ FROM data
| STATS count = COUNT(*) BY tags
```
-### URI_PARTS (Planned)
+### METRICS_INFO
+
+Returns one row per distinct metric available in the targeted time series data stream(s), with applicable dimensions and
+metadata. Use it to discover the metric catalogue without inspecting index mappings or calling the field capabilities
+API. **GA since 9.4** (and on Elastic Cloud Serverless).
+
+**Syntax:**
+
+```esql
+METRICS_INFO
+```
+
+Takes no parameters.
+
+**Output columns** (all `keyword`):
+
+- `metric_name` — the metric field name (single-valued)
+- `data_stream` — data stream(s) containing this metric (multi-valued when several streams align on
+ unit/metric_type/field_type)
+- `unit` — declared unit from field mapping (e.g., `bytes`, `packets`); may be `null` or multi-valued
+- `metric_type` — `counter`, `gauge`, etc. (multi-valued when definitions differ across backing indices)
+- `field_type` — Elasticsearch field type (e.g., `long`, `double`, `integer`)
+- `dimension_fields` — union of dimension field names across all time series for that metric
+
+**Restrictions:**
+
+- Can only be used after a `TS` source command — `FROM | METRICS_INFO` is rejected.
+- Must appear before pipeline-breaking commands (`STATS`, `SORT`, `LIMIT`).
+- The output replaces the original table — downstream commands operate on the metadata rows, not the raw documents.
+
+**Examples:**
+
+```esql
+// List every metric in a TSDS, alphabetically
+TS k8s
+| METRICS_INFO
+| SORT metric_name
+
+// Narrow to metrics that have data matching a filter, then keep only key columns
+TS k8s
+| WHERE cluster == "prod"
+| METRICS_INFO
+| KEEP metric_name, metric_type
+| SORT metric_name
+
+// Count metrics by type
+TS k8s
+| METRICS_INFO
+| STATS metric_count = COUNT(*) BY metric_type
+| SORT metric_type
+
+// Find metrics matching a name pattern
+TS k8s
+| METRICS_INFO
+| WHERE metric_name LIKE "network.eth0*"
+| SORT metric_name
+```
+
+### TS_INFO
+
+Returns one row per (metric, time series) combination in the targeted TSDS, including the dimension key/value pairs that
+identify each series. Use it to enumerate the actual time series — and their labels — that exist for each metric. **GA
+since 9.4** (and on Elastic Cloud Serverless).
+
+**Syntax:**
+
+```esql
+TS_INFO
+```
+
+Takes no parameters.
+
+**Output columns** (all `keyword`):
+
+- All columns from `METRICS_INFO` (`metric_name`, `data_stream`, `unit`, `metric_type`, `field_type`,
+ `dimension_fields`)
+- `dimensions` — JSON-encoded object with the dimension key/value pairs identifying the time series, e.g.
+ `{"job":"elasticsearch","instance":"instance_1"}`. Single-valued.
+
+**Restrictions:**
+
+- Can only be used after a `TS` source command — `FROM | TS_INFO` is rejected.
+- Must appear before pipeline-breaking commands (`STATS`, `SORT`, `LIMIT`).
+- The output replaces the original table — downstream commands operate on the metadata rows, not the raw documents.
+
+**Examples:**
+
+```esql
+// Every (metric, time series) pair in a TSDS
+TS k8s
+| TS_INFO
+| SORT metric_name, dimensions
+
+// Restrict to series with data matching a filter, keep only key columns
+TS k8s
+| WHERE cluster == "prod"
+| TS_INFO
+| KEEP metric_name, dimensions
+| SORT metric_name, dimensions
+
+// Filter by metadata after TS_INFO
+TS k8s
+| TS_INFO
+| WHERE metric_type == "gauge"
+| SORT metric_name, dimensions
+
+// Count distinct time series per metric
+TS k8s
+| TS_INFO
+| STATS series_count = COUNT(*) BY metric_name
+| SORT metric_name
+
+// Count distinct metrics per time series — useful to spot under- or over-reporting series
+TS k8s
+| TS_INFO
+| STATS metric_count = COUNT_DISTINCT(metric_name) BY dimensions
+| SORT dimensions
+```
+
+> **`METRICS_INFO` vs `TS_INFO`:** `METRICS_INFO` returns one row **per distinct metric**; `TS_INFO` returns one row
+> **per (metric, time series) combination** and adds a `dimensions` column with the labels identifying each series. Use
+> `METRICS_INFO` to enumerate _what_ is being measured, and `TS_INFO` to enumerate _which_ time series exist.
+
+### URI_PARTS (Serverless)
+
+Pipe command that parses a URI string into structured columns. A target prefix is **required**.
+
+**Syntax:**
+
+```esql
+URI_PARTS target = field
+```
+
+**Output columns:** `target.domain`, `target.path`, `target.scheme`, `target.extension`, `target.port`, `target.query`,
+`target.fragment`, `target.user_info`, `target.username`, `target.password`.
-Parses a URI string and extracts its components (domain, path, port, query, scheme, etc.) into new columns. Not yet
-released.
+**Example:**
+
+```esql
+FROM web_logs
+| WHERE http.response.status_code >= 400
+| URI_PARTS parts = url.full
+| STATS errors = COUNT(*) BY parts.domain, parts.path
+| SORT errors DESC
+```
+
+### USER_AGENT (Serverless)
+
+Pipe command that parses a user agent string into structured columns. A target prefix is **required**.
**Syntax:**
```esql
-URI_PARTS prefix = expression
+USER_AGENT target = field
```
+**Output columns:** `target.name`, `target.version`, `target.os.name`, `target.os.version`, `target.os.full`,
+`target.device.name`.
+
**Example:**
```esql
FROM web_logs
-| URI_PARTS url_parts = request_url
-| KEEP url_parts.domain, url_parts.path, url_parts.query
+| USER_AGENT ua = user_agent.original
+| STATS cnt = COUNT(*) BY ua.name, ua.version
+```
+
+### REGISTERED_DOMAIN (Serverless)
+
+Pipe command that extracts the registered domain, top-level domain, and subdomain from a hostname. A target prefix is
+**required**.
+
+**Syntax:**
+
+```esql
+REGISTERED_DOMAIN target = field
```
+**Output columns:** `target.domain` (full input), `target.registered_domain`, `target.top_level_domain`,
+`target.subdomain`.
+
+**Example:**
+
+```esql
+FROM dns_logs
+| REGISTERED_DOMAIN rd = dns.question.name
+| STATS queries = COUNT(*) BY rd.registered_domain
+| SORT queries DESC
+```
+
+> **Note:** `URI_PARTS`, `USER_AGENT`, and `REGISTERED_DOMAIN` are **pipe commands** (like `DISSECT`/`GROK`), not scalar
+> functions. The syntax `URI_PARTS(field)` does not work — use `| URI_PARTS target = field`.
+
---
## Aggregate Functions
Used with STATS command.
-| Function | Description | Example |
-| ---------------------------------- | ------------------------------------------------- | ------------------------------------------------ |
-| `COUNT(*)` | Count all rows | `STATS n = COUNT(*)` |
-| `COUNT(field)` | Count non-null values | `STATS n = COUNT(status)` |
-| `COUNT_DISTINCT(field)` | Count unique values | `STATS unique = COUNT_DISTINCT(user_id)` |
-| `SUM(field)` | Sum of values | `STATS total = SUM(amount)` |
-| `AVG(field)` | Average | `STATS avg_price = AVG(price)` |
-| `MIN(field)` | Minimum value | `STATS min_temp = MIN(temperature)` |
-| `MAX(field)` | Maximum value | `STATS max_score = MAX(score)` |
-| `MEDIAN(field)` | Median value | `STATS med = MEDIAN(response_time)` |
-| `PERCENTILE(field, p)` | Percentile | `STATS p95 = PERCENTILE(latency, 95)` |
-| `STD_DEV(field)` | Standard deviation | `STATS sd = STD_DEV(values)` |
-| `VARIANCE(field)` | Variance | `STATS var = VARIANCE(values)` |
-| `VALUES(field)` | Collect all values | `STATS all_tags = VALUES(tag)` |
-| `TOP(field, n, order)` | Top N values | `STATS top3 = TOP(score, 3, "desc")` |
-| `WEIGHTED_AVG(val, weight)` | Weighted average | `STATS wavg = WEIGHTED_AVG(score, weight)` |
-| `MEDIAN_ABSOLUTE_DEVIATION(field)` | Robust variability measure | `STATS mad = MEDIAN_ABSOLUTE_DEVIATION(latency)` |
-| `ABSENT(field)` | True if no non-null values (9.2+) | `STATS is_absent = ABSENT(error_code)` |
-| `PRESENT(field)` | True if any non-null values (9.2+) | `STATS has_data = PRESENT(metric)` |
-| `SAMPLE(field, n)` | Collect n sample values (8.19/9.1+) | `STATS examples = SAMPLE(message, 5)` |
-| `FIRST(field, sort_field)` | Earliest value by sort field (Serverless preview) | `STATS earliest = FIRST(message, @timestamp)` |
-| `LAST(field, sort_field)` | Latest value by sort field (Serverless preview) | `STATS latest = LAST(message, @timestamp)` |
-| `ST_CENTROID_AGG(field)` | Spatial centroid of points | `STATS center = ST_CENTROID_AGG(location)` |
-| `ST_EXTENT_AGG(field)` | Bounding box of geometries (8.18/9.0+, preview) | `STATS bbox = ST_EXTENT_AGG(location)` |
+| Function | Description | Example |
+| ---------------------------------- | ----------------------------------------------- | ------------------------------------------------ |
+| `COUNT(*)` | Count all rows | `STATS n = COUNT(*)` |
+| `COUNT(field)` | Count non-null values | `STATS n = COUNT(status)` |
+| `COUNT_DISTINCT(field)` | Count unique values | `STATS unique = COUNT_DISTINCT(user_id)` |
+| `SUM(field)` | Sum of values | `STATS total = SUM(amount)` |
+| `AVG(field)` | Average | `STATS avg_price = AVG(price)` |
+| `MIN(field)` | Minimum value | `STATS min_temp = MIN(temperature)` |
+| `MAX(field)` | Maximum value | `STATS max_score = MAX(score)` |
+| `MEDIAN(field)` | Median value | `STATS med = MEDIAN(response_time)` |
+| `PERCENTILE(field, p)` | Percentile | `STATS p95 = PERCENTILE(latency, 95)` |
+| `STD_DEV(field)` | Standard deviation | `STATS sd = STD_DEV(values)` |
+| `VARIANCE(field)` | Variance | `STATS var = VARIANCE(values)` |
+| `VALUES(field)` | Collect all values (GA) | `STATS all_tags = VALUES(tag)` |
+| `TOP(field, n, order)` | Top N values | `STATS top3 = TOP(score, 3, "desc")` |
+| `WEIGHTED_AVG(val, weight)` | Weighted average | `STATS wavg = WEIGHTED_AVG(score, weight)` |
+| `MEDIAN_ABSOLUTE_DEVIATION(field)` | Robust variability measure | `STATS mad = MEDIAN_ABSOLUTE_DEVIATION(latency)` |
+| `ABSENT(field)` | True if no non-null values (9.2+) | `STATS is_absent = ABSENT(error_code)` |
+| `PRESENT(field)` | True if any non-null values (9.2+) | `STATS has_data = PRESENT(metric)` |
+| `SAMPLE(field, n)` | Collect n sample values (8.19/9.1+) | `STATS examples = SAMPLE(message, 5)` |
+| `FIRST(field, sort_field)` | Earliest value by sort field (Serverless GA) | `STATS earliest = FIRST(message, @timestamp)` |
+| `LAST(field, sort_field)` | Latest value by sort field (Serverless GA) | `STATS latest = LAST(message, @timestamp)` |
+| `EARLIEST(field)` | Earliest value (single-arg; Serverless GA) | `STATS e = EARLIEST(@timestamp)` |
+| `LATEST(field)` | Latest value (single-arg; Serverless GA) | `STATS l = LATEST(@timestamp)` |
+| `ST_CENTROID_AGG(field)` | Spatial centroid of points | `STATS center = ST_CENTROID_AGG(location)` |
+| `ST_EXTENT_AGG(field)` | Bounding box of geometries (8.18/9.0+, preview) | `STATS bbox = ST_EXTENT_AGG(location)` |
### Grouping Functions
Used in the `BY` clause of `STATS` and `INLINE STATS` to create dynamic groups.
-| Function | Description | Example |
-| --------------------- | ------------------------------------------------- | ----------------------------------------------------- |
-| `BUCKET(field, size)` | Create fixed-size buckets for numbers or dates | `STATS count = COUNT(*) BY b = BUCKET(price, 10)` |
-| `TBUCKET(interval)` | Time-based bucketing (9.2+, for use with `TS`) | `STATS SUM(RATE(reqs)) BY TBUCKET(1 hour)` |
-| `CATEGORIZE(field)` | Auto-categorize text values (8.18/9.0+, Platinum) | `STATS count = COUNT(*) BY cat = CATEGORIZE(message)` |
+| Function | Description | Example |
+| --------------------- | -------------------------------------------------------------------- | ----------------------------------------------------- |
+| `BUCKET(field, size)` | Create fixed-size buckets for numbers or dates | `STATS count = COUNT(*) BY b = BUCKET(price, 10)` |
+| `TBUCKET(interval)` | Time-based bucketing (preview 9.2-9.3, GA in 9.4) | `STATS SUM(RATE(reqs)) BY TBUCKET(1 hour)` |
+| `WITHOUT(dim, ...)` | Group time series by every dimension except those listed (GA in 9.4) | `STATS total = SUM(network.cost) BY WITHOUT(pod)` |
+| `CATEGORIZE(field)` | Auto-categorize text values (8.18/9.0+, Platinum) | `STATS count = COUNT(*) BY cat = CATEGORIZE(message)` |
**CATEGORIZE options (9.2+):**
@@ -951,7 +1250,18 @@ FROM logs-*
Used with the `STATS` command after a `TS` source command. These functions evaluate per time series first, then
aggregate by group using an outer function (e.g., `SUM`, `AVG`). An optional second argument specifies a sliding time
-window. Available since 9.2.
+window.
+
+**Availability:** **All** time series aggregation functions are **GA since 9.4** — both the 9.2-introduced set (`RATE`,
+`IRATE`, `INCREASE`, `DELTA`, `IDELTA`, `AVG_OVER_TIME`, `SUM_OVER_TIME`, `MIN_OVER_TIME`, `MAX_OVER_TIME`,
+`FIRST_OVER_TIME`, `LAST_OVER_TIME`, `COUNT_OVER_TIME`, `COUNT_DISTINCT_OVER_TIME`, `PRESENT_OVER_TIME`,
+`ABSENT_OVER_TIME`) and the 9.3-introduced set (`DERIV`, `PERCENTILE_OVER_TIME`, `STDDEV_OVER_TIME`,
+`VARIANCE_OVER_TIME`). On clusters in 9.2-9.3 these functions are still in tech preview.
+
+**Sliding window parameter (second argument):** in 9.2-9.3 (preview) the window must be a multiple of the `TBUCKET`
+interval; **9.4+ (GA)** accepts arbitrary durations, with performance optimizations when the window is a multiple of the
+bucket interval. Within a single query, you cannot mix windows smaller than the bucket interval for one metric with
+windows larger than the bucket interval for another metric.
| Function | Description | Metric Types |
| ----------------------------------- | ------------------------------ | -------------- |
@@ -998,40 +1308,53 @@ TS metrics
## String Functions
-| Function | Description | Example |
-| --------------------------- | --------------------------------------------------- | -------------------------------------------------------------------- |
-| `LENGTH(s)` | String length | `EVAL len = LENGTH(name)` |
-| `CONCAT(s1, s2, ...)` | Concatenate strings | `EVAL full = CONCAT(first, " ", last)` |
-| `SUBSTRING(s, start, len)` | Extract substring | `EVAL sub = SUBSTRING(text, 1, 10)` |
-| `LEFT(s, n)` | Left n characters | `EVAL l = LEFT(text, 5)` |
-| `RIGHT(s, n)` | Right n characters | `EVAL r = RIGHT(text, 5)` |
-| `TRIM(s)` | Remove whitespace | `EVAL clean = TRIM(input)` |
-| `LTRIM(s)` | Trim left | `EVAL clean = LTRIM(input)` |
-| `RTRIM(s)` | Trim right | `EVAL clean = RTRIM(input)` |
-| `TO_UPPER(s)` | Uppercase | `EVAL upper = TO_UPPER(name)` |
-| `TO_LOWER(s)` | Lowercase | `EVAL lower = TO_LOWER(name)` |
-| `REPLACE(s, old, new)` | Replace text | `EVAL fixed = REPLACE(msg, "err", "error")` |
-| `SPLIT(s, delim)` | Split into array | `EVAL parts = SPLIT(path, "/")` |
-| `STARTS_WITH(s, prefix)` | Check prefix | `WHERE STARTS_WITH(url, "https")` |
-| `ENDS_WITH(s, suffix)` | Check suffix | `WHERE ENDS_WITH(file, ".log")` |
-| `CONTAINS(s, substr)` | Check contains | `WHERE CONTAINS(message, "error")` |
-| `LOCATE(substr, s)` | Find position | `EVAL pos = LOCATE("@", email)` |
-| `REVERSE(s)` | Reverse string | `EVAL rev = REVERSE(text)` |
-| `REPEAT(s, n)` | Repeat string | `EVAL sep = REPEAT("-", 10)` |
-| `SPACE(n)` | N spaces | `EVAL spaces = SPACE(5)` |
-| `BIT_LENGTH(s)` | Bit length (8.17+) | `EVAL bits = BIT_LENGTH(name)` |
-| `BYTE_LENGTH(s)` | Byte length (8.17+) | `EVAL bytes = BYTE_LENGTH(name)` |
-| `CHUNK(field, settings)` | Split text into chunks (9.3+, preview) | `EVAL chunks = CHUNK(body, {"strategy":"word","max_chunk_size":50})` |
-| `HASH(alg, s)` | Hash string (8.18/9.0+) | `EVAL h = HASH("SHA-256", msg)` |
-| `MD5(s)` | MD5 hash (8.18/9.0+) | `EVAL h = MD5(content)` |
-| `SHA1(s)` | SHA-1 hash (8.18/9.0+) | `EVAL h = SHA1(content)` |
-| `SHA256(s)` | SHA-256 hash (8.18/9.0+) | `EVAL h = SHA256(content)` |
-| `FROM_BASE64(s)` | Decode base64 | `EVAL decoded = FROM_BASE64(encoded)` |
-| `TO_BASE64(s)` | Encode to base64 | `EVAL encoded = TO_BASE64(data)` |
-| `URL_DECODE(s)` | URL-decode (9.2+) | `EVAL decoded = URL_DECODE(url)` |
-| `URL_ENCODE(s)` | URL-encode (9.2+) | `EVAL encoded = URL_ENCODE(text)` |
-| `URL_ENCODE_COMPONENT(s)` | URL-encode for URI components (9.2+) | `EVAL encoded = URL_ENCODE_COMPONENT(text)` |
-| `JSON_EXTRACT(field, path)` | Extract value from JSON string (Serverless preview) | `EVAL name = JSON_EXTRACT(raw, "$.user.name")` |
+| Function | Description | Example |
+| --------------------------- | ---------------------------------------------- | -------------------------------------------------------------------- |
+| `LENGTH(s)` | String length | `EVAL len = LENGTH(name)` |
+| `CONCAT(s1, s2, ...)` | Concatenate strings | `EVAL full = CONCAT(first, " ", last)` |
+| `SUBSTRING(s, start, len)` | Extract substring | `EVAL sub = SUBSTRING(text, 1, 10)` |
+| `LEFT(s, n)` | Left n characters | `EVAL l = LEFT(text, 5)` |
+| `RIGHT(s, n)` | Right n characters | `EVAL r = RIGHT(text, 5)` |
+| `TRIM(s)` | Remove whitespace | `EVAL clean = TRIM(input)` |
+| `LTRIM(s)` | Trim left | `EVAL clean = LTRIM(input)` |
+| `RTRIM(s)` | Trim right | `EVAL clean = RTRIM(input)` |
+| `TO_UPPER(s)` | Uppercase | `EVAL upper = TO_UPPER(name)` |
+| `TO_LOWER(s)` | Lowercase | `EVAL lower = TO_LOWER(name)` |
+| `REPLACE(s, old, new)` | Replace text | `EVAL fixed = REPLACE(msg, "err", "error")` |
+| `SPLIT(s, delim)` | Split into array | `EVAL parts = SPLIT(path, "/")` |
+| `STARTS_WITH(s, prefix)` | Check prefix | `WHERE STARTS_WITH(url, "https")` |
+| `ENDS_WITH(s, suffix)` | Check suffix | `WHERE ENDS_WITH(file, ".log")` |
+| `CONTAINS(s, substr)` | Check contains | `WHERE CONTAINS(message, "error")` |
+| `LOCATE(substr, s)` | Find position | `EVAL pos = LOCATE("@", email)` |
+| `REVERSE(s)` | Reverse string | `EVAL rev = REVERSE(text)` |
+| `REPEAT(s, n)` | Repeat string | `EVAL sep = REPEAT("-", 10)` |
+| `SPACE(n)` | N spaces | `EVAL spaces = SPACE(5)` |
+| `BIT_LENGTH(s)` | Bit length (8.17+) | `EVAL bits = BIT_LENGTH(name)` |
+| `BYTE_LENGTH(s)` | Byte length (8.17+) | `EVAL bytes = BYTE_LENGTH(name)` |
+| `CHUNK(field, settings)` | Split text into chunks (9.3+, preview) | `EVAL chunks = CHUNK(body, {"strategy":"word","max_chunk_size":50})` |
+| `HASH(alg, s)` | Hash string (8.18/9.0+) | `EVAL h = HASH("SHA-256", msg)` |
+| `MD5(s)` | MD5 hash (8.18/9.0+) | `EVAL h = MD5(content)` |
+| `SHA1(s)` | SHA-1 hash (8.18/9.0+) | `EVAL h = SHA1(content)` |
+| `SHA256(s)` | SHA-256 hash (8.18/9.0+) | `EVAL h = SHA256(content)` |
+| `FROM_BASE64(s)` | Decode base64 | `EVAL decoded = FROM_BASE64(encoded)` |
+| `TO_BASE64(s)` | Encode to base64 | `EVAL encoded = TO_BASE64(data)` |
+| `URL_DECODE(s)` | URL-decode (9.2+) | `EVAL decoded = URL_DECODE(url)` |
+| `URL_ENCODE(s)` | URL-encode (9.2+) | `EVAL encoded = URL_ENCODE(text)` |
+| `URL_ENCODE_COMPONENT(s)` | URL-encode for URI components (9.2+) | `EVAL encoded = URL_ENCODE_COMPONENT(text)` |
+| `JSON_EXTRACT(field, path)` | Extract value from JSON string (Serverless GA) | `EVAL name = JSON_EXTRACT(raw, "$.user.name")` |
+
+**JSON_EXTRACT with \_source — flattened field workaround:**
+
+ES|QL does not natively access `flattened` field sub-keys. Use `METADATA _source` with `JSON_EXTRACT` to reach inside
+flattened objects. `_source` can be passed directly to `JSON_EXTRACT` — do not wrap it with `TO_STRING()`.
+
+```esql
+FROM logs-* METADATA _source
+| EVAL provider = JSON_EXTRACT(_source, "$.cloud.provider")
+| STATS count = COUNT(*) BY provider
+```
+
+This also works for any field that exists in the raw document but has no explicit mapping.
---
@@ -1423,15 +1746,15 @@ FROM logs-*
Access document metadata with the `METADATA` directive on the `FROM` command. Once enabled, metadata fields behave like
regular index fields.
-| Field | Type | Description |
-| ------------- | ------- | --------------------------------------------------------------- |
-| `_id` | keyword | Unique document ID |
-| `_index` | keyword | Index name |
-| `_version` | long | Document version number |
-| `_score` | float | Query relevance score (updated by full-text search functions) |
-| `_ignored` | keyword | Fields that were ignored when the document was indexed |
-| `_index_mode` | keyword | Index mode (`standard`, `lookup`, `logsdb`, `time_series` etc.) |
-| `_source` | special | Original JSON document body (not supported by functions) |
+| Field | Type | Description |
+| ------------- | ------- | -------------------------------------------------------------------------------------- |
+| `_id` | keyword | Unique document ID |
+| `_index` | keyword | Index name |
+| `_version` | long | Document version number |
+| `_score` | float | Query relevance score (updated by full-text search functions) |
+| `_ignored` | keyword | Fields that were ignored when the document was indexed |
+| `_index_mode` | keyword | Index mode (`standard`, `lookup`, `logsdb`, `time_series` etc.) |
+| `_source` | special | Original JSON document body. Use `JSON_EXTRACT` to access flattened or unmapped fields |
```esql
FROM logs METADATA _id, _index, _version
diff --git a/skills/elasticsearch/elasticsearch-esql/references/esql-version-history.md b/skills/elasticsearch/elasticsearch-esql/references/esql-version-history.md
index 19ceab0..13ef9a1 100644
--- a/skills/elasticsearch/elasticsearch-esql/references/esql-version-history.md
+++ b/skills/elasticsearch/elasticsearch-esql/references/esql-version-history.md
@@ -27,51 +27,58 @@ determine compatibility when writing queries for specific Elasticsearch deployme
## Version Timeline Overview
-| Version | Release | Status | Key Additions |
-| ------- | -------- | ------------ | ------------------------------------------------------------------------------------------------ |
-| 8.11 | Nov 2023 | Tech Preview | Initial ES\|QL release |
-| 8.12 | Jan 2024 | Tech Preview | Spatial types, PROFILE |
-| 8.13 | Mar 2024 | Tech Preview | Async queries, cross-cluster ENRICH |
-| 8.14 | May 2024 | **GA** | Spatial functions, regex optimization |
-| 8.15 | Aug 2024 | GA | Type casting (`::`), Arrow output |
-| 8.16 | Oct 2024 | GA | Per-aggregation WHERE, new math/string functions |
-| 8.17 | Dec 2024 | GA | MATCH, QSTR full-text functions |
-| 8.18 | Feb 2025 | GA | LOOKUP JOIN (preview), scoring, KQL |
-| 8.19 | Apr 2025 | GA | MATCH_PHRASE, FORK, CHANGE_POINT (preview) |
-| 9.0 | Feb 2025 | GA | Released with 8.18 features |
-| 9.1 | Jun 2025 | GA | Full-text functions GA, FORK (preview) |
-| 9.2 | Oct 2025 | GA | Multi-field joins, TS, INLINE STATS (preview), CHANGE_POINT GA, FUSE (preview), RERANK (preview) |
-| 9.3 | Jan 2026 | GA | INLINE STATS GA, SET directive (preview), Lucene-pushable JOIN predicates |
+| Version | Release | Status | Key Additions |
+| ------- | -------- | ------------ | ------------------------------------------------------------------------------------------------------- |
+| 8.11 | Nov 2023 | Tech Preview | Initial ES\|QL release |
+| 8.12 | Jan 2024 | Tech Preview | Spatial types, PROFILE |
+| 8.13 | Mar 2024 | Tech Preview | Async queries, cross-cluster ENRICH |
+| 8.14 | May 2024 | **GA** | Spatial functions, regex optimization |
+| 8.15 | Aug 2024 | GA | Type casting (`::`), Arrow output |
+| 8.16 | Oct 2024 | GA | Per-aggregation WHERE, new math/string functions |
+| 8.17 | Dec 2024 | GA | MATCH, QSTR full-text functions |
+| 8.18 | Feb 2025 | GA | LOOKUP JOIN (preview), scoring, KQL |
+| 8.19 | Apr 2025 | GA | MATCH_PHRASE, FORK, CHANGE_POINT (preview) |
+| 9.0 | Feb 2025 | GA | Released with 8.18 features |
+| 9.1 | Jun 2025 | GA | Full-text functions GA, FORK (preview) |
+| 9.2 | Oct 2025 | GA | Multi-field joins, TS, INLINE STATS (preview), CHANGE_POINT GA, FUSE (preview), RERANK (preview) |
+| 9.3 | Jan 2026 | GA | INLINE STATS GA, SET directive (preview), Lucene-pushable JOIN predicates |
+| 9.4 | May 2026 | GA | TS GA, time series functions GA, WITHOUT/METRICS_INFO/TS_INFO GA, PROMQL (preview), MV_EXPAND/VALUES GA |
## Feature Availability by Version
### Commands
-| Command | Introduced | GA | Notes |
-| -------------- | ---------- | -------- | ------------------------------------------- |
-| `FROM` | 8.11 | 8.14 | Source command |
-| `WHERE` | 8.11 | 8.14 | Filtering |
-| `EVAL` | 8.11 | 8.14 | Computed columns |
-| `STATS ... BY` | 8.11 | 8.14 | Aggregations with grouping |
-| `SORT` | 8.11 | 8.14 | Ordering results |
-| `LIMIT` | 8.11 | 8.14 | Result set size |
-| `KEEP` | 8.11 | 8.14 | Column selection |
-| `DROP` | 8.11 | 8.14 | Column removal |
-| `RENAME` | 8.11 | 8.14 | Column renaming |
-| `DISSECT` | 8.11 | 8.14 | Pattern extraction |
-| `GROK` | 8.11 | 8.14 | Log parsing |
-| `ENRICH` | 8.11 | 8.14 | Data enrichment |
-| `MV_EXPAND` | 8.11 | 8.14 | Multi-value expansion |
-| `SHOW` | 8.11 | 8.14 | Metadata display |
-| `ROW` | 8.11 | 8.14 | Literal row creation |
-| `LOOKUP JOIN` | 8.18/9.0 | 8.19/9.1 | SQL-style LEFT JOIN with lookup indices |
-| `INLINE STATS` | 9.2 | 9.3 | Inline aggregations (like window functions) |
-| `FORK` | 8.19/9.1 | Preview | Multiple execution branches |
-| `FUSE` | 9.2 | Preview | Combine results from FORK branches |
-| `TS` | 9.2 | 9.2 | Time series mode |
-| `RERANK` | 9.2 | Preview | Re-score results with inference |
-| `COMPLETION` | 9.2 | 9.2 | LLM text generation |
-| `SAMPLE` | 8.19/9.1 | Preview | Random sampling |
+| Command | Introduced | GA | Notes |
+| -------------- | ---------- | -------- | ----------------------------------------------- |
+| `FROM` | 8.11 | 8.14 | Source command |
+| `WHERE` | 8.11 | 8.14 | Filtering |
+| `EVAL` | 8.11 | 8.14 | Computed columns |
+| `STATS ... BY` | 8.11 | 8.14 | Aggregations with grouping |
+| `SORT` | 8.11 | 8.14 | Ordering results |
+| `LIMIT` | 8.11 | 8.14 | Result set size |
+| `KEEP` | 8.11 | 8.14 | Column selection |
+| `DROP` | 8.11 | 8.14 | Column removal |
+| `RENAME` | 8.11 | 8.14 | Column renaming |
+| `DISSECT` | 8.11 | 8.14 | Pattern extraction |
+| `GROK` | 8.11 | 8.14 | Log parsing |
+| `ENRICH` | 8.11 | 8.14 | Data enrichment |
+| `MV_EXPAND` | 8.11 | 9.4 | Multi-value expansion (GA) |
+| `SHOW` | 8.11 | 8.14 | Metadata display |
+| `ROW` | 8.11 | 8.14 | Literal row creation |
+| `LOOKUP JOIN` | 8.18/9.0 | 8.19/9.1 | SQL-style LEFT JOIN with lookup indices |
+| `INLINE STATS` | 9.2 | 9.3 | Inline aggregations (like window functions) |
+| `FORK` | 8.19/9.1 | Preview | Multiple execution branches |
+| `FUSE` | 9.2 | Preview | Combine results from FORK branches |
+| `TS` | 9.2 | 9.4 | Time series source command |
+| `PROMQL` | 9.4 | Preview | Source command using PromQL syntax on TSDS |
+| `METRICS_INFO` | 9.4 | 9.4 | TSDS metric catalogue (after `TS`) |
+| `TS_INFO` | 9.4 | 9.4 | Per-(metric, time series) metadata (after `TS`) |
+| `RERANK` | 9.2 | Preview | Re-score results with inference |
+| `COMPLETION` | 9.2 | 9.2 | LLM text generation |
+| `SAMPLE` | 8.19/9.1 | Preview | Random sampling |
+| `URI_PARTS` | Srvless | Srvless | Parse URI into structured columns |
+| `USER_AGENT` | Srvless | Srvless | Parse user agent into structured columns |
+| `REG_DOMAIN` | Srvless | Srvless | `REGISTERED_DOMAIN`: extract from hostname |
### Full-Text Search Functions
@@ -157,19 +164,22 @@ determine compatibility when writing queries for specific Elasticsearch deployme
| `MEDIAN`, `MEDIAN_ABSOLUTE_DEVIATION` | 8.11 | Statistical |
| `PERCENTILE` | 8.11 | Percentile calculation |
| `TOP` | 8.15 | Top N values |
-| `VALUES` | 8.14 | Collect unique values |
+| `VALUES` | 8.14 | Unique values (GA in 9.4) |
| `ST_EXTENT_AGG` | 8.18/9.0 | Spatial bounding box |
| `WEIGHTED_AVG` | 8.16 | Weighted average |
| `STD_DEV` | 8.18/9.0 | Standard deviation |
| `VARIANCE` | 8.18/9.0 | Variance |
+| `FIRST` / `EARLIEST` | Serverless | Earliest value by sort field |
+| `LAST` / `LATEST` | Serverless | Latest value by sort field |
### Grouping Functions
-| Function | Introduced | Notes |
-| ------------ | ------------- | ------------------------------------------------- |
-| `BUCKET` | 8.11 | Numeric/date bucketing in `BY` clause |
-| `CATEGORIZE` | 8.18/9.0 | Auto-categorization of text in `BY` clause |
-| `TBUCKET` | 9.2 (preview) | Time bucketing from `@timestamp`; preferred in TS |
+| Function | Introduced | Notes |
+| ------------ | ---------- | ---------------------------------------------------------------- |
+| `BUCKET` | 8.11 | Numeric/date bucketing in `BY` clause |
+| `CATEGORIZE` | 8.18/9.0 | Auto-categorization of text in `BY` clause |
+| `TBUCKET` | 9.2 | Time bucketing from `@timestamp`; preferred in TS (GA in 9.4) |
+| `WITHOUT` | 9.4 | Group time series by every dimension except the listed ones (GA) |
### Per-Aggregation WHERE
@@ -189,31 +199,39 @@ Available since 8.16. Allows filtering individual aggregations without affecting
### Time Series Aggregation Functions
-Available under `TS ... | STATS`. See [time-series-queries.md](time-series-queries.md) for full reference.
-
-| Function | Introduced | Notes |
-| -------------------------- | ------------- | ----------------------------------------------- |
-| `RATE` | 9.2 (preview) | Per-second rate of counter increase |
-| `IRATE` | 9.2 (preview) | Instant rate (last two data points) |
-| `INCREASE` | 9.2 (preview) | Absolute counter increase in window |
-| `DELTA` | 9.2 (preview) | Absolute change of a gauge |
-| `IDELTA` | 9.2 (preview) | Change between last two data points |
-| `AVG_OVER_TIME` | 9.2 (preview) | Average value over time |
-| `SUM_OVER_TIME` | 9.2 (preview) | Sum of values over time |
-| `MIN_OVER_TIME` | 9.2 (preview) | Minimum value over time |
-| `MAX_OVER_TIME` | 9.2 (preview) | Maximum value over time |
-| `FIRST_OVER_TIME` | 9.2 (preview) | Earliest value by `@timestamp` |
-| `LAST_OVER_TIME` | 9.2 (preview) | Latest value by `@timestamp` (implicit default) |
-| `COUNT_OVER_TIME` | 9.2 (preview) | Count of values over time |
-| `COUNT_DISTINCT_OVER_TIME` | 9.2 (preview) | Count of distinct values over time |
-| `PRESENT_OVER_TIME` | 9.2 (preview) | `true` if field has values in window |
-| `ABSENT_OVER_TIME` | 9.2 (preview) | `true` if field has no values in window |
-| `DERIV` | 9.3 (preview) | Derivative via linear regression |
-| `PERCENTILE_OVER_TIME` | 9.3 (preview) | Percentile of values over time |
-| `STDDEV_OVER_TIME` | 9.3 (preview) | Population standard deviation over time |
-| `VARIANCE_OVER_TIME` | 9.3 (preview) | Population variance over time |
-
-Sliding window parameter (second argument) available since 9.3 preview.
+Available under `TS ... | STATS`. See [time-series-queries.md](time-series-queries.md) for full reference. All time
+series aggregation functions in this table — both the 9.2-introduced set and the 9.3-introduced set (`DERIV`,
+`PERCENTILE_OVER_TIME`, `STDDEV_OVER_TIME`, `VARIANCE_OVER_TIME`) — are **GA since 9.4**.
+
+| Function | Introduced | Status | Notes |
+| -------------------------- | ------------- | -------- | ----------------------------------------------- |
+| `RATE` | 9.2 (preview) | GA (9.4) | Per-second rate of counter increase |
+| `IRATE` | 9.2 (preview) | GA (9.4) | Instant rate (last two data points) |
+| `INCREASE` | 9.2 (preview) | GA (9.4) | Absolute counter increase in window |
+| `DELTA` | 9.2 (preview) | GA (9.4) | Absolute change of a gauge |
+| `IDELTA` | 9.2 (preview) | GA (9.4) | Change between last two data points |
+| `AVG_OVER_TIME` | 9.2 (preview) | GA (9.4) | Average value over time |
+| `SUM_OVER_TIME` | 9.2 (preview) | GA (9.4) | Sum of values over time |
+| `MIN_OVER_TIME` | 9.2 (preview) | GA (9.4) | Minimum value over time |
+| `MAX_OVER_TIME` | 9.2 (preview) | GA (9.4) | Maximum value over time |
+| `FIRST_OVER_TIME` | 9.2 (preview) | GA (9.4) | Earliest value by `@timestamp` |
+| `LAST_OVER_TIME` | 9.2 (preview) | GA (9.4) | Latest value by `@timestamp` (implicit default) |
+| `COUNT_OVER_TIME` | 9.2 (preview) | GA (9.4) | Count of values over time |
+| `COUNT_DISTINCT_OVER_TIME` | 9.2 (preview) | GA (9.4) | Count of distinct values over time |
+| `PRESENT_OVER_TIME` | 9.2 (preview) | GA (9.4) | `true` if field has values in window |
+| `ABSENT_OVER_TIME` | 9.2 (preview) | GA (9.4) | `true` if field has no values in window |
+| `DERIV` | 9.3 (preview) | GA (9.4) | Derivative via linear regression |
+| `PERCENTILE_OVER_TIME` | 9.3 (preview) | GA (9.4) | Percentile of values over time |
+| `STDDEV_OVER_TIME` | 9.3 (preview) | GA (9.4) | Population standard deviation over time |
+| `VARIANCE_OVER_TIME` | 9.3 (preview) | GA (9.4) | Population variance over time |
+
+**Sliding window parameter (second argument):**
+
+- 9.2-9.3 (preview) — accepted window values are limited to multiples of the `TBUCKET` interval in the `BY` clause; if
+ no window is specified, the bucket interval is used implicitly.
+- 9.4+ (GA) — all window values are accepted, with performance optimizations when the window is a multiple of the
+ `TBUCKET` interval. Mixing windows that are smaller than the time bucket for one metric with windows larger than the
+ time bucket for another metric in the same query is not allowed.
### Conditional Functions
@@ -252,22 +270,32 @@ ES|QL **does not support cursor-based pagination** like the Search API's `search
- Use `STATS` to aggregate at query time
- For exports, use Search API with `search_after` instead
-### Time Zone Support (Limited)
+### Time Zone Support (Limited before Serverless / 9.4)
-ES|QL has **limited timezone support**.
+ES|QL has **limited timezone support** on self-managed clusters prior to 9.4. All dates are processed in UTC internally
+and there is no per-function timezone argument.
-**Current limitations:**
+On **Serverless**, ES|QL supports query-wide timezone via the `SET time_zone` directive (GA on Serverless). This accepts
+IANA timezone strings and UTC offsets, and applies to all date/time operations including `DATE_TRUNC`, `DATE_FORMAT`,
+`NOW()`, bucketing, and display.
-- `DATE_FORMAT` and `DATE_PARSE` do not support timezone parameters
-- All dates processed in UTC internally
-- Kibana charts may show timezone inconsistencies
+```esql
+SET time_zone = "America/New_York";
+FROM logs-*
+| STATS errors = COUNT(*) BY hour = DATE_TRUNC(1 hour, @timestamp)
+| SORT hour DESC
+```
+
+**Remaining limitations (all versions):**
+
+- No per-function timezone argument — `DATE_TRUNC(1 hour, @timestamp, "America/New_York")` does **not** work
+- `DATE_FORMAT` and `DATE_PARSE` do not accept timezone parameters directly; use `SET time_zone` instead
- GitHub tracking issue: [#107560](https://github.com/elastic/elasticsearch/issues/107560)
-**Workarounds:**
+**Self-managed before 9.4:**
-- Store timezone offset in a separate field
-- Convert to UTC before querying
-- Use `EVAL` to add/subtract hours manually:
+- `SET time_zone` only accepts UTC offsets (`"+05:00"`), not IANA timezone strings
+- Workaround: use `EVAL` to add/subtract hours manually:
```esql
| EVAL local_time = timestamp + 1 hour
@@ -285,16 +313,16 @@ returned at all** — they are silently omitted from results.
These field types are not supported or have limitations:
-| Type | Status |
-| -------------- | ---------------------------- |
-| `nested` | Not supported - returns null |
-| `flattened` | Not supported |
-| `join` | Not supported |
-| `date_range` | Not supported |
-| `binary` | Not supported |
-| `completion` | Not supported |
-| `rank_feature` | Not supported |
-| `histogram` | Not supported |
+| Type | Status |
+| -------------- | ---------------------------------------------------------------------------------- |
+| `nested` | Not supported - returns null |
+| `flattened` | Not natively supported; use `METADATA _source` + `JSON_EXTRACT` for sub-key access |
+| `join` | Not supported |
+| `date_range` | Not supported |
+| `binary` | Not supported |
+| `completion` | Not supported |
+| `rank_feature` | Not supported |
+| `histogram` | Not supported |
### JOIN Limitations
@@ -317,15 +345,25 @@ These field types are not supported or have limitations:
- Lucene-pushable predicates: `MATCH`, `QSTR`, `KQL`, `CIDR_MATCH` in join conditions
- Further performance gains for filtered joins
-### No Subqueries
+### Subqueries (Limited)
-ES|QL does not support:
+ES|QL supports **subqueries in `FROM`** (Serverless tech preview) for combining results from multiple pipelines (UNION
+ALL semantics). These are non-correlated — each branch is independent.
-- Subqueries in WHERE clauses
-- Nested SELECT statements
-- CTEs (Common Table Expressions)
+```esql
+FROM
+ (FROM web_logs | WHERE status >= 500 | KEEP @timestamp, message, service.name),
+ (FROM app_logs | WHERE level == "error" | KEEP @timestamp, message, service.name)
+| SORT @timestamp DESC
+```
+
+**Not supported:**
-Use `INLINE STATS` (9.2+) for some subquery-like patterns.
+- Subqueries in `WHERE` clauses (no `WHERE field IN (FROM ...)`)
+- Correlated subqueries (branches cannot reference outer columns)
+- Nested SELECT / CTEs (Common Table Expressions)
+
+Use `INLINE STATS` (9.2+) for per-row vs. aggregate comparison patterns.
## Cross-Cluster Query Support
@@ -378,17 +416,48 @@ Use `INLINE STATS` (9.2+) for some subquery-like patterns.
### 9.2+
-- Use `TS` with `RATE`, `AVG_OVER_TIME`, etc. for time series metrics aggregations
-- Use `TBUCKET` for time bucketing in TS queries
+- Use `TS` with `RATE`, `AVG_OVER_TIME`, etc. for time series metrics aggregations (preview in 9.2-9.3, GA in 9.4)
+- Use `TBUCKET` for time bucketing in TS queries (GA in 9.4)
- Multi-field `LOOKUP JOIN` for complex correlations
- `FUSE` for hybrid search scoring
### 9.3+
- Use `TRANGE` instead of manual `WHERE @timestamp` filters
-- Sliding window parameter for time series functions (e.g. `RATE(field, 10m)`)
+- Sliding window parameter for time series functions (e.g. `RATE(field, 10m)`); in 9.2-9.3 the window must be a multiple
+ of the `TBUCKET` interval, this restriction is lifted in 9.4
- `CLAMP`, `CLAMP_MIN`, `CLAMP_MAX` for bounding metric values
+### 9.4+
+
+- `TS` source command and **all** time series aggregation functions are now **GA** — both the 9.2-introduced set
+ (`RATE`, `IRATE`, `INCREASE`, `DELTA`, `IDELTA`, `*_OVER_TIME`, `PRESENT_OVER_TIME`, `ABSENT_OVER_TIME`) and the
+ 9.3-introduced set (`DERIV`, `PERCENTILE_OVER_TIME`, `STDDEV_OVER_TIME`, `VARIANCE_OVER_TIME`).
+- `TBUCKET` grouping function is **GA**.
+- New `WITHOUT(...)` grouping function (GA) for time series queries: `BY WITHOUT(dim1, ...)` groups by every dimension
+ except the listed ones; `BY WITHOUT()` (no args) is equivalent to the implicit "group by all dimensions" behavior.
+- New `METRICS_INFO` and `TS_INFO` processing commands (both **GA**) for discovering the metric catalogue and dimension
+ labels of TSDS data without inspecting index mappings. Both must come after a `TS` source command and must appear
+ before pipeline-breaking commands (`STATS`/`SORT`/`LIMIT`). `METRICS_INFO` returns one row per distinct metric
+ signature; `TS_INFO` returns one row per (metric, time series) combination with the identifying dimension labels.
+- Sliding window parameter (`RATE(field, 10m)`) accepts arbitrary durations — no longer limited to multiples of the
+ `TBUCKET` interval. Note: a single query cannot mix windows smaller than the bucket for one metric with windows larger
+ than the bucket for another metric.
+- New `PROMQL` source command (preview) to run Prometheus Query Language directly against TSDS indices, with implicit
+ range selectors and a Kibana-aware `step`/`buckets` model. See [promql-command.md](promql-command.md). Prefer `PROMQL`
+ only when the user explicitly thinks in PromQL or is migrating Prometheus dashboards/alerts; otherwise prefer `TS`.
+- `MV_EXPAND` is GA
+- `VALUES` aggregation is GA
+
+### Serverless (latest)
+
+- `SET time_zone` with IANA timezone strings for query-wide timezone support (GA)
+- `LIMIT n BY field` for grouped top-N queries
+- `URI_PARTS`, `USER_AGENT`, `REGISTERED_DOMAIN` pipe commands for parsing structured strings
+- `FROM` subqueries for combining results from multiple pipelines (tech preview)
+- `EARLIEST`/`LATEST` aliases for `FIRST`/`LAST` aggregations
+- `JSON_EXTRACT` on `METADATA _source` for accessing flattened field sub-keys
+
## Version Detection
To check ES|QL availability and version:
diff --git a/skills/elasticsearch/elasticsearch-esql/references/generation-tips.md b/skills/elasticsearch/elasticsearch-esql/references/generation-tips.md
index dea929c..e4192a0 100644
--- a/skills/elasticsearch/elasticsearch-esql/references/generation-tips.md
+++ b/skills/elasticsearch/elasticsearch-esql/references/generation-tips.md
@@ -147,7 +147,7 @@ FROM my-index-2024.* // Dated indices
```
For time series data streams (TSDS), use `TS` instead of `FROM` to enable time series aggregation functions like `RATE`,
-`AVG_OVER_TIME`, etc. (9.2+):
+`AVG_OVER_TIME`, etc. (preview from 9.2 to 9.3, **GA since 9.4**):
```esql
TS metrics-* // Time series source — enables RATE, AVG_OVER_TIME, etc.
@@ -503,6 +503,12 @@ TS metrics-tsds
See [Time Series Queries](time-series-queries.md) for the full inner/outer aggregation model.
+**Version status:** `TS`, `TBUCKET`, the new `WITHOUT(...)` grouping function, the new `METRICS_INFO` / `TS_INFO`
+discovery commands, and **all** time series aggregation functions are **GA since 9.4** — including the 9.2-introduced
+set (`RATE`, `IRATE`, `INCREASE`, `DELTA`, `IDELTA`, all `*_OVER_TIME`, `PRESENT_OVER_TIME`, `ABSENT_OVER_TIME`) and the
+9.3-introduced set (`DERIV`, `PERCENTILE_OVER_TIME`, `STDDEV_OVER_TIME`, `VARIANCE_OVER_TIME`). On clusters in 9.2-9.3
+these features are tech preview. `TRANGE` remains in preview.
+
**Pre-9.2 limitation:** The `TS` command, `RATE()`, `TBUCKET()`, and `AVG_OVER_TIME()` all require Elasticsearch
**9.2+**. On older clusters, counter fields (`counter_long`, `counter_double`) cannot be aggregated meaningfully —
standard aggregation functions like `MAX()`, `SUM()`, and `AVG()` reject counter field types. There is no workaround.
@@ -512,6 +518,10 @@ the `TS` command and `RATE()` are required (9.2+) and the query cannot be expres
For **gauge** fields in time-series indices on pre-9.2 clusters, `FROM` with standard aggregations (`AVG`, `MAX`, `MIN`)
still works — only counter fields are affected.
+**Sliding window restriction (9.2-9.3):** When the user wants a per-time-series aggregation window different from the
+`TBUCKET` interval (`RATE(field, 10m) BY TBUCKET(1m)`), the window must be a multiple of the bucket interval on preview
+clusters. **9.4+** (GA) accepts arbitrary windows.
+
### INLINE STATS (9.2+)
`INLINE STATS` is available in **9.2+** only. It computes an aggregation and appends the result as a new column to every
@@ -522,6 +532,70 @@ ES|QL before 9.2**. There is no fallback.
When the cluster is pre-9.2 and the question requires per-row vs. aggregate comparison, explain that `INLINE STATS` is
needed and suggest the user either upgrade or perform the comparison client-side.
+### Pipe Commands: URI_PARTS, USER_AGENT, REGISTERED_DOMAIN (Serverless)
+
+These are **pipe commands** (like `DISSECT`/`GROK`), not scalar functions. They must appear on their own pipeline stage
+with `target = expression` syntax. A target prefix is mandatory.
+
+```esql
+// WRONG — function-call syntax does not work
+| EVAL parts = URI_PARTS(url.full)
+
+// CORRECT — pipe command syntax with target prefix
+| URI_PARTS parts = url.full
+| KEEP parts.domain, parts.path, parts.scheme
+```
+
+When the user asks to "parse URLs", "extract domains", or "parse user agents", reach for these commands instead of
+`DISSECT`/`GROK`:
+
+| User Request | Command |
+| ------------------------- | ------------------- |
+| Parse/decompose a URL | `URI_PARTS` |
+| Parse a user agent string | `USER_AGENT` |
+| Extract registered domain | `REGISTERED_DOMAIN` |
+
+### Grouped Top-N with LIMIT BY (Serverless)
+
+`LIMIT n BY field` keeps the top N rows per group after sorting. The number comes **before** `BY`.
+
+```esql
+// Top 3 error-producing hosts per service
+FROM logs-*
+| WHERE level == "error"
+| STATS cnt = COUNT(*) BY service.name, host.name
+| SORT cnt DESC
+| LIMIT 3 BY service.name
+```
+
+This replaces the common `INLINE STATS` + rank-and-filter pattern for simple grouped top-N.
+
+### Subqueries in FROM vs FORK
+
+**Subqueries** (Serverless tech preview) combine results from **different** data sources (UNION ALL semantics). **FORK**
+runs **different analyses** on the **same** data source.
+
+| Scenario | Use |
+| ------------------------------------- | ---------- |
+| Combine errors from two index sets | Subqueries |
+| Run multiple aggregations on one set | FORK |
+| Compare time windows of the same data | FORK |
+| Union independent pipelines | Subqueries |
+
+```esql
+// Subqueries — different sources
+FROM
+ (FROM web_logs | WHERE status >= 500 | KEEP @timestamp, message, service.name),
+ (FROM app_logs | WHERE level == "error" | KEEP @timestamp, message, service.name)
+| SORT @timestamp DESC
+
+// FORK — same source, different analyses
+FROM logs-*
+| FORK
+ ( WHERE level == "error" | STATS errors = COUNT(*) BY service.name )
+ ( WHERE level == "warning" | STATS warnings = COUNT(*) BY service.name )
+```
+
### External IPs — CIDR_MATCH with RFC 1918
When the user asks about "external IPs" or "public IPs", exclude private (RFC 1918) ranges with `NOT CIDR_MATCH`:
diff --git a/skills/elasticsearch/elasticsearch-esql/references/promql-command.md b/skills/elasticsearch/elasticsearch-esql/references/promql-command.md
new file mode 100644
index 0000000..31b01c9
--- /dev/null
+++ b/skills/elasticsearch/elasticsearch-esql/references/promql-command.md
@@ -0,0 +1,323 @@
+# ES|QL PROMQL Command
+
+Query time series indices using **Prometheus Query Language (PromQL)** as a source command in ES|QL. The `PROMQL`
+command is the bridge for users who already know PromQL or are migrating Prometheus dashboards and alerts onto an
+Elasticsearch backend, while still letting them post-process results with regular ES|QL pipes.
+
+> **Version:** `PROMQL` is a **preview** feature available since Elastic Stack **9.4** and on Elastic Cloud Serverless.
+> Treat it as preview — syntax, options, and supported PromQL functions may change in future releases. See
+> [esql-version-history.md](esql-version-history.md) for version availability.
+
+## Table of Contents
+
+- [When to Use PROMQL](#when-to-use-promql)
+- [Syntax](#syntax)
+- [Options](#options)
+- [Output Columns](#output-columns)
+- [Implicit Range Selectors](#implicit-range-selectors)
+- [Examples](#examples)
+- [Post-Processing with ES|QL](#post-processing-with-esql)
+- [PROMQL vs TS](#promql-vs-ts)
+- [Limitations](#limitations)
+- [Kibana Time Filtering](#kibana-time-filtering)
+- [Guidelines](#guidelines)
+- [References](#references)
+
+---
+
+## When to Use PROMQL
+
+Prefer `PROMQL` when **any** of the following apply:
+
+- The user explicitly asks for a PromQL query, references Prometheus syntax (`sum by (instance) (...)`, label matchers
+ like `{cluster="prod"}`, etc), or is migrating a Prometheus dashboard or alert. If the user explicitly requests for
+ PromQL but the query is not supported yet (check [Limitations](#limitations) below), state the issue.
+- Compatibility with Prometheus tooling is required (Grafana panels, alerting rules, scripts that already speak PromQL).
+
+Prefer the [`TS` command](time-series-queries.md) when:
+
+- The user wrote ES|QL (or is asking in natural language without PromQL terms) and the query is naturally expressed in
+ the inner/outer aggregation paradigm (`SUM(RATE(...))`, `AVG(AVG_OVER_TIME(...))`).
+- The query mixes time series with non-time-series data sources or uses ES|QL features like `LOOKUP JOIN`,
+ `CHANGE_POINT`, or `INLINE STATS` _before_ the metrics aggregation.
+
+`PROMQL` and `TS` target the same TSDS indices — choose based on the syntax that best matches the user's intent.
+
+---
+
+## Syntax
+
+```esql
+PROMQL [ ... ] [ = ] ( )
+```
+
+- Zero or more space-separated `key=value` options.
+- A PromQL expression, optionally wrapped in parentheses and assigned a ``.
+- The expression follows standard
+ [Prometheus query language](https://prometheus.io/docs/prometheus/latest/querying/basics/) syntax (label matchers,
+ range selectors, aggregations, binary operations) within the [Limitations](#limitations) below.
+
+### Minimal example
+
+```esql
+PROMQL sum by (instance) (rate(http_requests_total))
+```
+
+### Named result
+
+```esql
+PROMQL http_rate = (sum by (instance) (rate(http_requests_total)))
+```
+
+When a `` is provided, the metric column is named `` instead of the raw PromQL expression. In
+the example above, the column would be named `http_rate`.
+
+---
+
+## Options
+
+The options mirror the Prometheus [HTTP API](https://prometheus.io/docs/prometheus/latest/querying/api/#range-queries)
+with ES|QL-specific additions.
+
+| Option | Default | Description |
+| ----------------- | ----------- | --------------------------------------------------------------------------------------------------------------------- |
+| `index` | `metrics-*` | Indices, data streams, or aliases. Supports wildcards and date math. |
+| `step` | inferred | Query resolution step width. Auto-derived from `buckets` and the time range when omitted. |
+| `buckets` | `100` | Target bucket count for auto-step derivation. Mutually exclusive with `step`. Requires a known time range. |
+| `start` | inferred | Inclusive start of the time range. Falls back to Kibana's date picker, or unrestricted if missing. |
+| `end` | inferred | Inclusive end of the time range. Falls back to Kibana's date picker, or unrestricted if missing. |
+| `scrape_interval` | `1m` | Expected metric collection interval. Used as the implicit range selector window: `max(step, scrape_interval)`. |
+| `=` | _none_ | Optional name for the metric output column. Defaults to the PromQL expression text. Wrap the expression in `( ... )`. |
+
+**Time format for `start` / `end`:** ISO-8601 strings (e.g., `"2026-04-01T00:00:00Z"`). The same formats accepted by
+`TRANGE` work here.
+
+**`step` vs `buckets`:** Pass exactly one. `step` fixes the resolution (`step=5m`); `buckets` lets the engine pick a
+step that produces around N buckets across the time range (`buckets=50`).
+
+---
+
+## Output Columns
+
+The result table has these columns:
+
+| Column | Type | Description |
+| ------------------------------------------------------- | --------- | --------------------------------------------------------------- |
+| The PromQL expression (or `` if specified) | `double` | The computed metric value |
+| `step` | `date` | Timestamp for each evaluation step |
+| Grouping labels (when `by (...)` or `without (...)`) | `keyword` | One column per grouping label |
+| `_timeseries` | `keyword` | JSON-encoded labels when there is no `by`/`without` aggregation |
+
+When the PromQL expression includes a cross-series aggregation like `sum by (instance) (...)`, each grouping label
+becomes its own column (`instance:keyword`). Without a cross-series aggregation, all labels collapse into a single
+`_timeseries` column as a JSON string.
+
+---
+
+## Implicit Range Selectors
+
+Standard PromQL requires range vector functions to specify a range selector: `rate(http_requests_total[5m])`. The
+`PROMQL` command **allows omitting the range selector** entirely:
+
+```esql
+PROMQL scrape_interval=15s sum(rate(http_requests_total))
+```
+
+When the range selector is absent, the window is computed automatically as `max(step, scrape_interval)`. This is
+particularly useful for Kibana dashboards where `step` is determined by the date picker and you want the range vector to
+scale with it.
+
+You can still pass an explicit range selector when you need a fixed window: `rate(http_requests_total[5m])`.
+
+---
+
+## Examples
+
+### Fully adaptive query (recommended for Kibana)
+
+Let Kibana's date picker drive the time range, and let `step` and the range selector be inferred:
+
+```esql
+PROMQL index=metrics-* sum by (instance) (rate(http_requests_total))
+```
+
+The query responds to the date picker, adjusts the step size to the selected range, and sizes the implicit range
+selector window accordingly. This is the recommended pattern for dashboard panels.
+
+### Range query with explicit parameters
+
+```esql
+PROMQL index=k8s step=5m start="2024-05-10T00:20:00.000Z" end="2024-05-10T00:25:00.000Z" (
+ sum(avg_over_time(network.cost[5m]))
+)
+```
+
+| sum(avg_over_time(network.cost[5m])):double | step:date |
+| ------------------------------------------- | ------------------------ |
+| 50.25 | 2024-05-10T00:20:00.000Z |
+
+### Cross-series aggregation by label
+
+```esql
+PROMQL index=k8s step=1h result=(sum by (cluster) (network.cost))
+| SORT result
+```
+
+| result:double | step:datetime | cluster:keyword |
+| ------------- | ------------------------ | --------------- |
+| 15.875 | 2024-05-10T00:00:00.000Z | staging |
+| 18.625 | 2024-05-10T00:00:00.000Z | prod |
+| 26.5 | 2024-05-10T00:00:00.000Z | qa |
+
+### Label filtering with named result
+
+```esql
+PROMQL index=k8s step=1h cost=(max by (cluster) (network.total_bytes_in{cluster!="prod"}))
+| SORT cluster
+```
+
+| cost:double | step:datetime | cluster:keyword |
+| ----------- | ------------------------ | --------------- |
+| 10797.0 | 2024-05-10T00:00:00.000Z | qa |
+| 7403.0 | 2024-05-10T00:00:00.000Z | staging |
+
+### Ad-hoc query with inferred step
+
+For queries outside Kibana, set `start` and `end` explicitly. The step and range selector window are still inferred from
+the time range and the default `buckets` value:
+
+```esql
+PROMQL index=metrics-*
+ start="2026-04-01T00:00:00Z"
+ end="2026-04-01T01:00:00Z"
+ sum by (instance) (rate(http_requests_total))
+```
+
+### Bucket count instead of fixed step
+
+```esql
+PROMQL index=metrics-*
+ buckets=50
+ start="2026-04-01T00:00:00Z"
+ end="2026-04-01T01:00:00Z"
+ sum(rate(http_requests_total))
+```
+
+---
+
+## Post-Processing with ES|QL
+
+Because `PROMQL` is a source command, its output flows into the rest of the pipeline. Use ES|QL commands after the
+PROMQL stage for further aggregation, filtering, ordering, and enrichment:
+
+```esql
+PROMQL index=k8s step=1h bytes=(max by (cluster) (network.bytes_in))
+| STATS max_bytes = MAX(bytes) BY cluster
+| SORT cluster
+```
+
+| max_bytes:double | cluster:keyword |
+| ---------------- | --------------- |
+| 931.0 | prod |
+| 972.0 | qa |
+| 238.0 | staging |
+
+### Enrich with LOOKUP JOIN
+
+Join PromQL results with a lookup index using a grouping label as the join key:
+
+```esql
+PROMQL index=metrics-*
+ http_rate=(sum by (instance) (rate(http_requests_total)))
+| LOOKUP JOIN instance_metadata ON instance
+```
+
+This pattern combines PromQL's expressiveness for time series math with ES|QL's strengths for joining external metadata,
+filtering, and shaping output.
+
+---
+
+## PROMQL vs TS
+
+| Aspect | `PROMQL` | `TS` |
+| ------------------- | ------------------------------------------- | -------------------------------------------------- |
+| Syntax | Prometheus Query Language | ES\|QL inner/outer aggregation |
+| Default index | `metrics-*` | None — caller must specify |
+| Time filtering | `start`/`end` options or Kibana date picker | `WHERE TRANGE(...)` or `WHERE @timestamp ...` |
+| Bucketing | `step` / `buckets` options | `BY TBUCKET(interval)` |
+| Range vector window | Implicit (`max(step, scrape_interval)`) | Bucket interval, or sliding window arg (9.3+) |
+| Counter aggregation | `sum(rate(metric))` | `STATS SUM(RATE(metric)) BY TBUCKET(...)` |
+| Gauge aggregation | `avg_over_time(metric[5m])` | `STATS AVG(AVG_OVER_TIME(metric)) BY TBUCKET(...)` |
+| Label filtering | `metric{cluster="prod"}` | `WHERE cluster == "prod"` |
+| Available since | 9.4 (preview) | 9.2 (preview) |
+
+Both commands target TSDS indices and can be followed by the same set of ES|QL processing commands (`WHERE`, `EVAL`,
+`STATS`, `SORT`, `LIMIT`, `LOOKUP JOIN`, etc.).
+
+---
+
+## Limitations
+
+In 9.4 preview, `PROMQL` has the following limitations:
+
+- **Group modifiers are not supported.** Constructs like `on(chip) group_left(chip_name)` will fail. Use `LOOKUP JOIN`
+ in ES|QL after the PROMQL stage to attach extra labels.
+- **Set operators are not supported.** `or`, `and`, and `unless` between PromQL expressions are unavailable. Express set
+ logic in ES|QL after the PROMQL stage instead.
+- **Some PromQL functions are unavailable.** Notably `histogram_quantile`, `predict_linear`, and `label_join` are not
+ supported. Use `TS` with `PERCENTILE_OVER_TIME` for percentile-style metrics, or compute equivalents in ES|QL.
+- **Time bucket alignment differs.** Buckets align to fixed calendar boundaries rather than the query start time. This
+ can cause slight differences from native Prometheus, especially for short ranges or large step sizes.
+- **Index defaults to `metrics-*`.** If your TSDS data lives elsewhere, always set `index` explicitly to avoid scanning
+ unrelated indices.
+- **Preview status.** Behavior, supported PromQL surface, and option names may evolve before GA.
+
+When a question requires a feature in this list, fall back to the [`TS` command](time-series-queries.md) and express the
+equivalent computation in ES|QL.
+
+---
+
+## Kibana Time Filtering
+
+When writing `PROMQL` queries for Kibana (Discover, dashboards, alerts), **do not set `start` and `end` manually**.
+Kibana injects the date picker's range automatically and the engine derives `step` from it. Setting `start`/`end`
+explicitly overrides the date picker.
+
+```esql
+// Kibana — let the date picker drive start/end and step
+PROMQL index=metrics-* sum by (instance) (rate(http_requests_total))
+```
+
+For ad-hoc queries outside Kibana (HTTP API, `node scripts/esql.js raw "..."`), set `start` and `end` explicitly.
+
+---
+
+## Guidelines
+
+- **Prefer `PROMQL` only when the user explicitly thinks in PromQL** or is porting a Prometheus query/dashboard.
+ Otherwise, prefer `TS` — it integrates more naturally with the rest of ES|QL and is GA in 9.4.
+- **Always set `index`** in production queries instead of relying on the `metrics-*` default — narrower patterns reduce
+ scan volume and prevent accidental matches against unrelated indices.
+- **Use named results** (`http_rate=(...)`) when chaining further ES|QL commands. Named columns are easier to reference
+ than the raw PromQL expression text.
+- **Omit range selectors for adaptive dashboards.** Implicit range selectors (`rate(http_requests_total)` without
+ `[5m]`) make the query scale with the date picker.
+- **Pick `step` or `buckets`, not both.** Use `buckets` when you want a target panel resolution; use `step` when you
+ need a fixed grain (e.g., to align with downstream aggregation).
+- **Fall back to `TS` for unsupported features.** Histograms (`histogram_quantile`), set logic (`or`/`and`/`unless`),
+ group modifiers, and `label_join` are not available — express the computation with ES|QL primitives instead.
+- **Do not mix `WHERE @timestamp` filters with `start`/`end`.** Time filtering belongs in the PROMQL options or via
+ Kibana's date picker; standard ES|QL `WHERE` clauses run _after_ the PromQL stage and don't bound the metric scan.
+
+---
+
+## References
+
+- [ES|QL PROMQL command](https://www.elastic.co/docs/reference/query-languages/esql/commands/promql) — official
+ documentation
+- [Prometheus Query Language](https://prometheus.io/docs/prometheus/latest/querying/basics/) — PromQL fundamentals
+- [Prometheus HTTP API](https://prometheus.io/docs/prometheus/latest/querying/api/#range-queries) — origin of the option
+ semantics
+- [Time series data streams (TSDS)](https://www.elastic.co/docs/manage-data/data-store/data-streams/time-series-data-stream-tsds)
+- [time-series-queries.md](time-series-queries.md) — `TS` command and ES|QL native time series functions
+- [esql-version-history.md](esql-version-history.md) — feature availability by Elasticsearch version
diff --git a/skills/elasticsearch/elasticsearch-esql/references/query-patterns.md b/skills/elasticsearch/elasticsearch-esql/references/query-patterns.md
index bd47dd8..50ff4bd 100644
--- a/skills/elasticsearch/elasticsearch-esql/references/query-patterns.md
+++ b/skills/elasticsearch/elasticsearch-esql/references/query-patterns.md
@@ -437,6 +437,60 @@ FROM flights
| KEEP flight_id, destination, distance, avg_dist
```
+### Grouped Top-N with LIMIT BY (Serverless)
+
+```text
+"top 3 error types per service"
+→
+FROM logs-*
+| WHERE @timestamp > NOW() - 24 hours AND level == "error"
+| STATS cnt = COUNT(*) BY service.name, error.type
+| SORT cnt DESC
+| LIMIT 3 BY service.name
+```
+
+```text
+"most recent event per user"
+→
+// Serverless: use LATEST to get the most recent value per group
+FROM events-*
+| STATS last_action = LATEST(event.action), last_ts = LATEST(@timestamp) BY user.name
+
+// Alternative with LIMIT BY
+FROM events-*
+| SORT @timestamp DESC
+| LIMIT 1 BY user.name
+| KEEP user.name, @timestamp, event.action
+```
+
+### Subquery Composition (Serverless tech preview)
+
+```text
+"combine web server errors and application errors into one view"
+→
+FROM
+ (FROM web_logs
+ | WHERE @timestamp > NOW() - 1 hour AND status_code >= 500
+ | EVAL source = "web"
+ | KEEP @timestamp, message, service.name, source),
+ (FROM app_logs
+ | WHERE @timestamp > NOW() - 1 hour AND level == "error"
+ | EVAL source = "app"
+ | KEEP @timestamp, message, service.name, source)
+| SORT @timestamp DESC
+| LIMIT 100
+```
+
+```text
+"count errors from different log sources by service"
+→
+FROM
+ (FROM web_logs | WHERE status_code >= 500 | KEEP @timestamp, service.name),
+ (FROM app_logs | WHERE level == "error" | KEEP @timestamp, service.name)
+| STATS errors = COUNT(*) BY service.name
+| SORT errors DESC
+```
+
### MATCH_PHRASE (8.19/9.1+)
```text
diff --git a/skills/elasticsearch/elasticsearch-esql/references/time-series-queries.md b/skills/elasticsearch/elasticsearch-esql/references/time-series-queries.md
index d4afc01..a4d5f08 100644
--- a/skills/elasticsearch/elasticsearch-esql/references/time-series-queries.md
+++ b/skills/elasticsearch/elasticsearch-esql/references/time-series-queries.md
@@ -3,12 +3,25 @@
Query metrics data in Elasticsearch using the `TS` source command and time series aggregation functions. Requires
Elasticsearch 9.2+.
+> **Status:** `TS`, **all** time series aggregation functions (the 9.2-introduced set — `RATE`, `IRATE`, `INCREASE`,
+> `DELTA`, `IDELTA`, `AVG_OVER_TIME`, `SUM_OVER_TIME`, `MIN_OVER_TIME`, `MAX_OVER_TIME`, `FIRST_OVER_TIME`,
+> `LAST_OVER_TIME`, `COUNT_OVER_TIME`, `COUNT_DISTINCT_OVER_TIME`, `PRESENT_OVER_TIME`, `ABSENT_OVER_TIME` — and the
+> 9.3-introduced set — `DERIV`, `PERCENTILE_OVER_TIME`, `STDDEV_OVER_TIME`, `VARIANCE_OVER_TIME`), the `TBUCKET`
+> grouping function, the new `WITHOUT(...)` grouping function, and the new `METRICS_INFO` and `TS_INFO` discovery
+> commands are **GA since 9.4** (preview from 9.2 to 9.3 for the 9.2/9.3 features; new in 9.4 for `WITHOUT`,
+> `METRICS_INFO`, and `TS_INFO`). `TRANGE` remains in preview. **Looking for PromQL?** Elasticsearch 9.4+ also exposes a
+> `PROMQL` source command for running Prometheus Query Language directly against TSDS indices. See
+> [promql-command.md](promql-command.md). Prefer `PROMQL` only when the user explicitly thinks in PromQL or is migrating
+> Prometheus dashboards/alerts; otherwise prefer `TS` and the inner/outer aggregation paradigm described below.
+
## Table of Contents
- [TS Source Command](#ts-source-command)
- [Inner/Outer Aggregation Paradigm](#innerouter-aggregation-paradigm)
- [Time Series Aggregation Functions](#time-series-aggregation-functions)
- [TBUCKET Grouping Function](#tbucket-grouping-function)
+- [WITHOUT Grouping Function](#without-grouping-function)
+- [Metric and Time Series Discovery](#metric-and-time-series-discovery)
- [TRANGE Time Filter](#trange-time-filter)
- [CLAMP Functions](#clamp-functions)
- [Kibana Time Filtering](#kibana-time-filtering)
@@ -24,6 +37,8 @@ Elasticsearch 9.2+.
[time series data streams (TSDS)](https://www.elastic.co/docs/manage-data/data-store/data-streams/time-series-data-stream-tsds).
It enables time series aggregation functions (`RATE`, `AVG_OVER_TIME`, etc.) inside `STATS`.
+**Availability:** Preview from 9.2 to 9.3, **GA since 9.4**. GA on Elastic Cloud Serverless.
+
**Syntax:**
```esql
@@ -36,6 +51,8 @@ TS index_pattern [METADATA fields]
- Enables inner/outer aggregation paradigm in `STATS`
- Cannot combine with `FORK` before `STATS` is applied
- Optimized for processing time series data; `FROM` may produce unexpected results on TSDS indices
+- When there is **no** `STATS` command in the query, `TS` returns rows sorted by `@timestamp` descending by default —
+ useful for listing recent values across many time series.
**Best practices:**
@@ -72,7 +89,9 @@ TS metrics | STATS AVG(LAST_OVER_TIME(memory_usage))
```
Since 9.3 (preview), use a time series function directly without an outer aggregation to get one value per time series
-per bucket:
+per bucket. The result is implicitly grouped by all dimensions of each time series and includes a `_timeseries` column
+with the dimension key/value pairs — see [WITHOUT Grouping Function](#without-grouping-function) for narrowing this
+grouping (`BY WITHOUT(dim, ...)`, GA since 9.4).
```esql
TS metrics
@@ -99,11 +118,11 @@ is used as the window.
For fields with `time_series_metric: counter` (`counter_double`, `counter_integer`, `counter_long`).
-| Function | Description | Since |
-| ---------- | ---------------------------------------------------------------------- | ----- |
-| `RATE` | Per-second average rate of increase; handles counter resets | 9.2 |
-| `IRATE` | Per-second rate between the last two data points; responsive to spikes | 9.2 |
-| `INCREASE` | Absolute increase of the counter in the time window; handles resets | 9.2 |
+| Function | Description | Since | Status |
+| ---------- | ---------------------------------------------------------------------- | ------------- | -------- |
+| `RATE` | Per-second average rate of increase; handles counter resets | 9.2 (preview) | GA (9.4) |
+| `IRATE` | Per-second rate between the last two data points; responsive to spikes | 9.2 (preview) | GA (9.4) |
+| `INCREASE` | Absolute increase of the counter in the time window; handles resets | 9.2 (preview) | GA (9.4) |
```esql
// Average rate per host per hour
@@ -127,22 +146,22 @@ TS k8s
For gauge metrics and general numeric fields (`double`, `integer`, `long`, `aggregate_metric_double`).
-| Function | Description | Since |
-| -------------------------- | -------------------------------------------------- | ----- |
-| `AVG_OVER_TIME` | Average value over the time window | 9.2 |
-| `SUM_OVER_TIME` | Sum of values over the time window | 9.2 |
-| `MIN_OVER_TIME` | Minimum value over the time window | 9.2 |
-| `MAX_OVER_TIME` | Maximum value over the time window | 9.2 |
-| `FIRST_OVER_TIME` | Earliest value by `@timestamp` | 9.2 |
-| `LAST_OVER_TIME` | Latest value by `@timestamp` (implicit default) | 9.2 |
-| `COUNT_OVER_TIME` | Count of values over the time window | 9.2 |
-| `COUNT_DISTINCT_OVER_TIME` | Count of distinct values over the time window | 9.2 |
-| `PERCENTILE_OVER_TIME` | Percentile of values; takes `(field, percentile)` | 9.3 |
-| `STDDEV_OVER_TIME` | Population standard deviation over the time window | 9.3 |
-| `VARIANCE_OVER_TIME` | Population variance over the time window | 9.3 |
-| `DELTA` | Absolute change of a gauge in the time window | 9.2 |
-| `IDELTA` | Change between the last two data points only | 9.2 |
-| `DERIV` | Derivative over time using linear regression | 9.3 |
+| Function | Description | Since | Status |
+| -------------------------- | -------------------------------------------------- | ------------- | -------- |
+| `AVG_OVER_TIME` | Average value over the time window | 9.2 (preview) | GA (9.4) |
+| `SUM_OVER_TIME` | Sum of values over the time window | 9.2 (preview) | GA (9.4) |
+| `MIN_OVER_TIME` | Minimum value over the time window | 9.2 (preview) | GA (9.4) |
+| `MAX_OVER_TIME` | Maximum value over the time window | 9.2 (preview) | GA (9.4) |
+| `FIRST_OVER_TIME` | Earliest value by `@timestamp` | 9.2 (preview) | GA (9.4) |
+| `LAST_OVER_TIME` | Latest value by `@timestamp` (implicit default) | 9.2 (preview) | GA (9.4) |
+| `COUNT_OVER_TIME` | Count of values over the time window | 9.2 (preview) | GA (9.4) |
+| `COUNT_DISTINCT_OVER_TIME` | Count of distinct values over the time window | 9.2 (preview) | GA (9.4) |
+| `PERCENTILE_OVER_TIME` | Percentile of values; takes `(field, percentile)` | 9.3 (preview) | GA (9.4) |
+| `STDDEV_OVER_TIME` | Population standard deviation over the time window | 9.3 (preview) | GA (9.4) |
+| `VARIANCE_OVER_TIME` | Population variance over the time window | 9.3 (preview) | GA (9.4) |
+| `DELTA` | Absolute change of a gauge in the time window | 9.2 (preview) | GA (9.4) |
+| `IDELTA` | Change between the last two data points only | 9.2 (preview) | GA (9.4) |
+| `DERIV` | Derivative over time using linear regression | 9.3 (preview) | GA (9.4) |
```esql
// Average memory per cluster per 5 minutes
@@ -171,10 +190,10 @@ TS k8s
Detect whether a field has data in a given time window. Return `boolean`.
-| Function | Description | Since |
-| ------------------- | ----------------------------------------------- | ----- |
-| `PRESENT_OVER_TIME` | `true` if field has values in the window | 9.2 |
-| `ABSENT_OVER_TIME` | `true` if field has **no** values in the window | 9.2 |
+| Function | Description | Since | Status |
+| ------------------- | ----------------------------------------------- | ------------- | -------- |
+| `PRESENT_OVER_TIME` | `true` if field has values in the window | 9.2 (preview) | GA (9.4) |
+| `ABSENT_OVER_TIME` | `true` if field has **no** values in the window | 9.2 (preview) | GA (9.4) |
```esql
// Detect pods with missing data
@@ -182,10 +201,11 @@ TS k8s
| STATS missing = MAX(ABSENT_OVER_TIME(events_received)) BY pod, TBUCKET(2 minute)
```
-### Sliding Window (9.3+)
+### Sliding Window
-Pass a `time_duration` as the second argument to any time series function to use a sliding window larger than the bucket
-interval. The window must be a multiple of the `TBUCKET` interval.
+Pass a `time_duration` as the second argument to any time series function to use a sliding window for the
+per-time-series aggregation. The window is orthogonal to time bucketing of output results (`TBUCKET`). If the window is
+omitted, the `TBUCKET` interval is used implicitly.
```esql
// Average rate per host over a 10-minute sliding window, bucketed by 1 minute
@@ -195,6 +215,15 @@ TS metrics
| STATS AVG(RATE(requests, 10m)) BY TBUCKET(1m), host
```
+**Version behavior:**
+
+- **9.2-9.3 (preview):** the window must be a **multiple of the `TBUCKET` interval** in the `BY` clause (for example,
+ with `TBUCKET(1m)` you may use `1m`, `2m`, `10m`; `7m` is rejected). If no window is specified, the `TBUCKET` interval
+ is used implicitly.
+- **9.4+ (GA):** all window values are accepted, with performance optimizations when the window is a multiple of the
+ `TBUCKET` interval. **Restriction:** within a single query you cannot mix windows that are **smaller** than the bucket
+ interval for one metric with windows that are **larger** than the bucket interval for another metric.
+
---
## TBUCKET Grouping Function
@@ -214,7 +243,7 @@ The interval is a time duration (`1 hour`, `5 minute`, `30s`) or date period (`1
`TBUCKET` is the preferred bucketing function for `TS` queries. It has a simpler signature than
`DATE_TRUNC(interval, @timestamp)` and is aware of time series semantics.
-**Availability:** Preview since 9.2.
+**Availability:** Preview from 9.2 to 9.3, **GA since 9.4**.
```esql
// 1-hour buckets
@@ -229,6 +258,161 @@ TS metrics
---
+## WITHOUT Grouping Function
+
+When the first `STATS` after `TS` uses a **bare** time series aggregation function (one not wrapped in an outer
+aggregation such as `AVG()` or `SUM()`), rows are implicitly grouped by **all** dimensions of each time series. The
+output includes a `_timeseries` `keyword` column containing a JSON-encoded object with the dimension key/value pairs
+identifying each group. Only the dimensions that actually exist for a given time series appear in `_timeseries` — not
+every dimension declared in the index mappings — so different rows in the result may carry different dimension keys.
+
+`WITHOUT(...)` lets you make this grouping explicit, or narrow it to a subset of dimensions:
+
+- `BY WITHOUT(dim1, dim2, ...)` groups by **all** dimensions **except** those listed.
+- `BY WITHOUT()` (no arguments) explicitly groups by every dimension; it is equivalent to the implicit "group by all"
+ behavior.
+
+When combining a bare time series function with other groupings, **only grouping functions** (`TBUCKET`, `WITHOUT`) are
+allowed in the `BY` clause — bare dimension columns are rejected. For example:
+
+```esql
+// INVALID -- bare time series function with a bare dimension column in BY
+TS k8s | STATS rate(network.total_bytes_in) BY host
+```
+
+Use `BY TBUCKET(...)` and/or `BY WITHOUT(...)`, or wrap the time series function with an outer aggregation.
+
+**Availability:** **GA since 9.4**. Can only be used in the **first** `STATS` command under a `TS` source — using it in
+a `FROM | STATS ... BY WITHOUT(...)` query is rejected.
+
+**Examples:**
+
+```esql
+// Group by every dimension implicitly — _timeseries column carries the dimension labels
+TS k8s
+| STATS avg = AVG_OVER_TIME(network.cost)
+| SORT avg DESC
+
+// Group by every dimension EXCEPT pod
+TS k8s
+| STATS total_cost = SUM(network.cost) BY WITHOUT(pod)
+| SORT total_cost
+
+// Combine WITHOUT with TBUCKET to add a time bucket to the surviving dimensions
+TS k8s
+| STATS total_cost = SUM(network.cost) BY WITHOUT(pod), tbucket = TBUCKET(1 hour)
+| SORT total_cost
+
+// Equivalent to implicit grouping (group by all dimensions)
+TS k8s
+| STATS avg = AVG_OVER_TIME(network.cost) BY WITHOUT()
+```
+
+---
+
+## Metric and Time Series Discovery
+
+Two processing commands introduced in 9.4 expose the metric catalogue of a TSDS so you can discover what to query
+without inspecting index mappings or calling the field capabilities API. Both must follow a `TS` source command and must
+appear before pipeline-breaking commands (`STATS`, `SORT`, `LIMIT`).
+
+| Command | Granularity | Status | Adds beyond `METRICS_INFO` |
+| -------------- | ------------------------------------- | ------------ | ---------------------------------------------------------------- |
+| `METRICS_INFO` | One row per **metric** | GA since 9.4 | — |
+| `TS_INFO` | One row per **(metric, time series)** | GA since 9.4 | `dimensions` JSON column with the labels identifying each series |
+
+### METRICS_INFO
+
+Returns one row per distinct metric in the targeted TSDS, with applicable dimensions and metadata. Useful for "what
+metrics exist in this stream?" and "what dimensions apply to metric X?".
+
+**Output columns** (all `keyword`):
+
+- `metric_name` — single-valued
+- `data_stream` — multi-valued when several streams align on unit/metric_type/field_type
+- `unit` — declared unit; may be `null` or multi-valued
+- `metric_type` — `counter`, `gauge`, etc.
+- `field_type` — `long`, `double`, `integer`, etc.
+- `dimension_fields` — union of dimension field names across the series for the metric
+
+```esql
+// List every metric, alphabetically
+TS k8s
+| METRICS_INFO
+| SORT metric_name
+
+// Restrict to metrics that have data matching a filter
+TS k8s
+| WHERE cluster == "prod"
+| METRICS_INFO
+| SORT metric_name
+
+// Filter by metric type, count by it
+TS k8s
+| METRICS_INFO
+| STATS metric_count = COUNT(*) BY metric_type
+| SORT metric_type
+
+// Find metrics whose name matches a pattern
+TS k8s
+| METRICS_INFO
+| WHERE metric_name LIKE "network.eth0*"
+| SORT metric_name
+```
+
+### TS_INFO
+
+Returns one row per (metric, time series) combination, including the dimension key/value pairs that identify each
+series. Useful for "which time series report this metric?" and "what label combinations exist?".
+
+`TS_INFO` includes **all `METRICS_INFO` columns** plus a `dimensions` column — a JSON-encoded object such as
+`{"job":"elasticsearch","instance":"instance_1"}` (single-valued).
+
+```esql
+// Every (metric, time series) pair in the data stream
+TS k8s
+| TS_INFO
+| SORT metric_name, dimensions
+
+// Filter the underlying series before discovery
+TS k8s
+| WHERE cluster == "prod"
+| TS_INFO
+| KEEP metric_name, dimensions
+| SORT metric_name, dimensions
+
+// Filter by metadata after TS_INFO (gauges only)
+TS k8s
+| TS_INFO
+| WHERE metric_type == "gauge"
+| SORT metric_name, dimensions
+
+// Count distinct time series per metric
+TS k8s
+| TS_INFO
+| STATS series_count = COUNT(*) BY metric_name
+| SORT metric_name
+
+// Count distinct metrics per time series — spot under- or over-reporting series
+TS k8s
+| TS_INFO
+| STATS metric_count = COUNT_DISTINCT(metric_name) BY dimensions
+| SORT dimensions
+```
+
+### Guidelines
+
+- **Use these for TSDS schema discovery** before writing `RATE`/`AVG_OVER_TIME` queries — they replace the older
+ workflow of inspecting `_settings`, `_mapping`, or field capabilities for time series indices.
+- **Reach for `METRICS_INFO` first** to enumerate metrics; reach for `TS_INFO` only when you need the exact dimension
+ combinations (label sets) of individual series.
+- **Filter before discovery** with `WHERE` on dimension fields to scope the catalogue to a relevant subset of series.
+- **The output replaces the original table.** Anything you `STATS` / `SORT` / `LIMIT` afterwards operates on metadata
+ rows, not raw documents — there's no way to re-attach the data points after `METRICS_INFO` or `TS_INFO`.
+- **Both commands are TSDS-only.** `FROM | METRICS_INFO` and `FROM | TS_INFO` are rejected.
+
+---
+
## TRANGE Time Filter
Filter data by time range using `@timestamp`. Prefer `TRANGE` over manual `WHERE @timestamp > NOW() - ...` filters.
@@ -376,7 +560,7 @@ TS k8s
| STATS SUM(IRATE(network.total_bytes_in)) BY cluster, TBUCKET(10 minute)
```
-### Sliding Window Rate (9.3+)
+### Sliding Window Rate
```esql
TS metrics
@@ -384,6 +568,9 @@ TS metrics
| STATS AVG(RATE(requests, 10m)) BY TBUCKET(1m), host
```
+In 9.2-9.3 the window must be a multiple of the `TBUCKET` interval. In 9.4+ any window value is accepted (mixing windows
+smaller and larger than the bucket interval for different metrics in the same query is not supported).
+
---
## Guidelines
@@ -395,9 +582,15 @@ TS metrics
- **Always add a time range filter** with `TRANGE` (or `WHERE @timestamp`) to limit scan volume, except in Kibana where
the date picker handles this automatically. Don't add a range filter if the user explicitly asks not to add it.
- **Version requirements:**
- - `TS`, `TBUCKET`, time series functions: 9.2+ (preview)
- - `TRANGE`, sliding windows, `DERIV`, `STDDEV_OVER_TIME`, `VARIANCE_OVER_TIME`, `PERCENTILE_OVER_TIME`: 9.3+ (preview)
- - `CLAMP`, `CLAMP_MIN`, `CLAMP_MAX`: 9.3+ (preview)
+ - `TS`, `TBUCKET`, `WITHOUT`, `METRICS_INFO`, `TS_INFO`, and **all** time series aggregation functions (the
+ 9.2-introduced set — `RATE`, `IRATE`, `INCREASE`, `DELTA`, `IDELTA`, all `*_OVER_TIME` from 9.2,
+ `PRESENT_OVER_TIME`, `ABSENT_OVER_TIME` — and the 9.3-introduced set — `DERIV`, `PERCENTILE_OVER_TIME`,
+ `STDDEV_OVER_TIME`, `VARIANCE_OVER_TIME`): **GA since 9.4** (preview from 9.2-9.3 for the 9.2/9.3 functions; new in
+ 9.4 for `WITHOUT`, `METRICS_INFO`, `TS_INFO`).
+ - `TRANGE`: 9.3+ (preview).
+ - Sliding window parameter (second argument to time series functions): introduced in 9.2-9.3 (preview, restricted to
+ multiples of the `TBUCKET` interval); **GA in 9.4** with arbitrary durations.
+ - `CLAMP`, `CLAMP_MIN`, `CLAMP_MAX`: 9.3+ (preview).
- **Do not nest time series functions.** `AVG_OVER_TIME(RATE(field))` is invalid. Use a standard aggregation as the
outer function.
- **Avoid mixing metrics with different dimensions** in one query. If `foo` and `bar` have different dimension values,
@@ -406,6 +599,9 @@ TS metrics
## References
- [ES|QL TS Command](https://www.elastic.co/docs/reference/query-languages/esql/commands/ts)
+- [ES|QL PROMQL Command](https://www.elastic.co/docs/reference/query-languages/esql/commands/promql) — alternative
+ source command using PromQL syntax (9.4+ preview)
+- [promql-command.md](promql-command.md) — PROMQL command reference in this skill
- [Time Series Aggregation Functions](https://www.elastic.co/docs/reference/query-languages/esql/functions-operators/time-series-aggregation-functions)
- [TBUCKET Function](https://www.elastic.co/docs/reference/query-languages/esql/functions-operators/grouping-functions/tbucket)
- [TRANGE Function](https://www.elastic.co/docs/reference/query-languages/esql/functions-operators/date-time-functions/trange)
diff --git a/skills/elasticsearch/elasticsearch-onboarding/SKILL.md b/skills/elasticsearch/elasticsearch-onboarding/SKILL.md
index 35a39f6..979d3be 100644
--- a/skills/elasticsearch/elasticsearch-onboarding/SKILL.md
+++ b/skills/elasticsearch/elasticsearch-onboarding/SKILL.md
@@ -3,8 +3,9 @@ name: elasticsearch-onboarding
description: >
Help developers new to Elasticsearch get from zero to a working search experience.
Guide them through understanding their intent, mapping their data, and building
- a search experience with best practices baked in. Use this when developers are new
- to Elasticsearch and need help getting started with their search use case.
+ a search experience with best practices baked in. Use this when the user shows intent
+ to build search-related functionality, asks about Elasticsearch-related concepts
+ for their use case, or expresses the need for help getting started with Elasticsearch.
compatibility: Elasticsearch 9.x
metadata:
author: elastic
@@ -29,6 +30,16 @@ Example user intents that should trigger this skill:
- "What are the best practices for building a search experience?"
- "Can you help me understand how to model my data for search?"
- "How do I build a vector database?"
+- "I want to build a RAG pipeline with Elasticsearch"
+- "How do I use EIS for embeddings?"
+- "How do I connect an LLM to Elasticsearch?"
+- "How do I do kNN search in Elasticsearch?"
+- "How do I use ELSER for semantic search?"
+- "How do I set up the Elasticsearch MCP?"
+- "How do I combine keyword and vector results with RRF?"
+- "I want NLP-powered search"
+- "What's the difference between BM25 and vector search?"
+- "Can I use ES|QL to query my data?"
## Guidelines
diff --git a/skills/elasticsearch/elasticsearch-onboarding/references/elasticsearch-onboarding-playbook.md b/skills/elasticsearch/elasticsearch-onboarding/references/elasticsearch-onboarding-playbook.md
index 80f840e..c5366f5 100644
--- a/skills/elasticsearch/elasticsearch-onboarding/references/elasticsearch-onboarding-playbook.md
+++ b/skills/elasticsearch/elasticsearch-onboarding/references/elasticsearch-onboarding-playbook.md
@@ -218,8 +218,8 @@ explanations:
> Does this look right, or would you add/remove anything?
**Surface the hybrid option when it adds value.** If the use case involves descriptive or natural-language queries,
-recommend semantic search alongside keyword. Explain the tradeoff: requires an embedding model (ELSER built-in, or
-OpenAI/Cohere), slightly slower indexing — but catches meaning-based queries keywords miss.
+recommend semantic search alongside keyword. Explain the tradeoff: requires an embedding model served via EIS (managed)
+or a user-provided inference endpoint, slightly slower indexing — but catches meaning-based queries keywords miss.
**For RAG retrieval**, recommend hybrid if documents contain specific terms or codes users will search for exactly
(policy names, product IDs, error codes).
diff --git a/skills/kibana/kibana-anomaly-detection/SKILL.md b/skills/kibana/kibana-anomaly-detection/SKILL.md
new file mode 100644
index 0000000..5f03663
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/SKILL.md
@@ -0,0 +1,416 @@
+---
+name: kibana-anomaly-detection
+description: Elastic ML anomaly detection skill — investigation/RCA, score explanation,
+ job operations (create, datafeed, start/stop, results), and troubleshooting (missing
+ docs, memory limits, datafeed health, lifecycle). Operates against Kibana Agent
+ Builder MCP tools (`ad_*`) on `.ml-anomalies-*`, `.ml-config`, `.ml-notifications-*`,
+ `.ml-annotations-*`. Use when answering "what broke?"/"which entity?"/RCA, "why
+ is score high/low?"/renormalization, "datafeed stopped"/"memory limit", or any request
+ to set up or configure an ML anomaly detection job.
+metadata:
+ author: elastic
+ version: 0.2.0
+compatibility: Kibana 8.x–9.x with Agent Builder and Workflows; Elasticsearch 8.x–9.x
+ with machine learning
+---
+
+# Elastic ML Anomaly Detection
+
+Single skill covering all anomaly detection work against **Kibana Agent Builder** MCP at
+`{KIBANA_URL}/api/agent_builder/mcp`. Use the **Mode Selector** below to pick the right approach for the user's question
+— modes share the same tool surface and concepts.
+
+## Platform
+
+- Read path: ES|QL against `.ml-anomalies-*`, `.ml-config`, `.ml-notifications-*`, `.ml-annotations-*`
+- Always-available: `platform.core.execute_esql` (plus additional platform tools for search, index mapping, and
+ documentation — see `scripts/agent_builder_constants.json`)
+- ML API spec (if available): `.kibana_ai_openapi_spec_elasticsearch` — see
+ [references/anomaly-detection-openapi-spec-discover.md](references/anomaly-detection-openapi-spec-discover.md) for
+ discovery pattern.
+- **Run `ad_validate_ml_tool_permissions` first** when tools return empty/misleading results — missing privileges are
+ the most common cause of false negatives. Full permissions matrix:
+ [references/permissions-matrix.md](references/permissions-matrix.md).
+
+## Mode Selector
+
+| User intent | Mode |
+| ----------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------ |
+| "What broke?" / RCA / cross-job / blast radius / influencers / log categories | **Investigate** |
+| "Why score high/low?" / renormalization / model bounds / forecasts | **Explain** |
+| Missing docs / memory limit / datafeed stopped / CCS / lifecycle / calendars | **Troubleshoot** |
+| Create a job / configure a datafeed / start analysis / retrieve results | **Manage** |
+| Security framing (attack chains, MITRE, exfil) | Investigate + [references/security-anomaly-expert.md](references/security-anomaly-expert.md) |
+| Observability/SRE framing (degradation, capacity, deployment regression) | Investigate + [references/observability-anomaly-expert.md](references/observability-anomaly-expert.md) |
+
+When a question spans modes: **Investigate → Explain → Troubleshoot**. Don't blend mode logic — finish one before moving
+on.
+
+---
+
+## Score Quick Reference
+
+- `record_score` bands: **>75** critical · **50–75** warning · **25–50** minor · **<25** informational
+- `multi_bucket_impact ≥ 3` → sustained shift (not a transient spike)
+- `initial_record_score >> record_score` → renormalization (model saw worse anomalies later)
+- `actual << typical` with `count`/`low_count`/`low_mean` → absence/outage, not just low value
+- Low scores across many jobs > one high score — composite cross-job signal often beats single-detector severity
+
+> Full score definitions, renormalization mechanics, and `anomaly_score_explanation` components:
+> [references/score-reference.md](references/score-reference.md).
+
+## Core concepts
+
+Treat `.ml-anomalies-*` as three layers, accessed via `result_type`:
+
+- **`bucket`** — bucket-level unusualness per `bucket_span`. `anomaly_score` is the aggregate across all detectors.
+- **`record`** — finest-grained rows with `actual` vs `typical`, `probability`, `record_score`,
+ `anomaly_score_explanation`.
+- **`influencer`** — entity contributions ranked within a bucket (`influencer_score`).
+
+Read scores this way:
+
+- `anomaly_score` / `record_score` = **current normalized** values (move as the model sees new extremes).
+- `initial_anomaly_score` / `initial_record_score` = **immutable snapshots** from detection time.
+- Compare `actual` to `typical`; use `probability` for raw likelihood.
+- Map entities via `partition_field_value` / `by_field_value` / `over_field_value`.
+- Read `multi_bucket_impact` (-5 to +5) to separate single-bucket spikes from sustained trends.
+
+---
+
+## Mode: Investigate — RCA
+
+**When:** "what broke?", "which entity caused this?", cross-job correlation, blast radius, attack/cascade chains.
+
+### Tool chain
+
+| Phase | Tools |
+| --------------------- | -------------------------------------------------------------------------------------------------------------- |
+| Discovery | `ad_get_available_metadata`, `ad_get_jobs`, `ad_discover_related_jobs`, `ad_discover_jobs_by_datafeed_index` |
+| Timeline / scope | `ad_query_anomaly_timeline` |
+| Cross-job / entities | `ad_rca_cross_job_entity_match`, `ad_rca_multi_job_entities`, `ad_rca_entity_profile` |
+| Records / influencers | `ad_query_anomaly_records`, `ad_query_influencers` |
+| RCA depth | `ad_rca_detector_fingerprint`, `ad_rca_correlation`, `ad_rca_blast_radius`, `ad_rca_score_reassessment` |
+| Evidence / categories | `ad_get_job_datafeed_config`, `ad_rca_source_evidence`, `ad_get_categories`, `ad_search_log_category_examples` |
+
+### Protocol
+
+Follow the 14-step sequence in [references/protocols/investigation.md](references/protocols/investigation.md). High
+level: `ad_get_available_metadata` → pair `ad_discover_jobs_by_datafeed_index` with `ad_discover_related_jobs` →
+`ad_query_anomaly_timeline` → rank with `ad_rca_multi_job_entities` (`min_job_count=2`) → `ad_rca_detector_fingerprint`
+→ drill with `ad_query_anomaly_records` + `ad_query_influencers` (low `min_score=25`) → profile with
+`ad_rca_entity_profile` → order with `ad_rca_correlation` → confirm with `ad_rca_source_evidence`. When
+`by_field_name == "mlcategory"`, compare with `ad_get_categories` + paired `ad_search_log_category_examples` (baseline
+vs. anomaly window).
+
+Finish with a written RCA: **root cause entity · affected jobs · temporal progression · fault class
+(resource/network/application) · severity · recommended actions**. Worked example:
+[references/worked-example.md](references/worked-example.md). Full ES|QL templates and parameters:
+[references/investigate-anomaly-esql-tools.md](references/investigate-anomaly-esql-tools.md).
+
+### Rules
+
+1. **Multi-job entities are prime suspects; single-job entities are usually victims.** Use `min_job_count=2`.
+2. **Earliest anomaly timestamp wins** — sort `ad_rca_correlation` by timestamp; first-appearing entity = origin.
+3. **`multi_bucket_impact ≥ 3` = sustained behavioral shift**, weight higher than transient spikes.
+4. **Never close an RCA without `ad_rca_source_evidence`** — raw source documents are ground truth.
+5. **Use low `min_score` (25 or lower) for influencer queries** — high thresholds miss correlated entities.
+
+---
+
+## Mode: Explain — Score / model behavior
+
+**When:** "why is my score 30/90?", "score dropped overnight", "what is renormalization?", "why wasn't this detected?".
+
+### Score types
+
+| Field | Scope | Meaning |
+| ---------------------- | --------------- | ----------------------------------------------------------------------- |
+| `record_score` | Single record | Normalized severity after renormalization. |
+| `initial_record_score` | Single record | Score at detection time. Gap vs `record_score` = renormalization drift. |
+| `anomaly_score` | Bucket | Aggregate severity across all detectors in a bucket. |
+| `influencer_score` | Entity × bucket | How anomalous a specific entity is in that bucket. |
+
+### `anomaly_score_explanation` components
+
+| Component | Effect | What it means |
+| -------------------------------- | ------- | ------------------------------------------------------------ |
+| `anomaly_length` | ↑ score | More consecutive anomalous buckets |
+| `single_bucket_impact` | ↑ score | Lower probability → higher impact |
+| `multi_bucket_impact` | ↑ score | Sustained pattern contribution |
+| `anomaly_characteristics_impact` | ↑ score | Mean shift vs. variance change |
+| `high_variance_penalty` | ↓ score | Noisy data → wide bounds → anomaly less surprising |
+| `incomplete_bucket_penalty` | ↓ score | Bucket has less data than expected (ingest lag, sparse data) |
+
+### Why a score looks wrong
+
+- **Unexpectedly low:** `high_variance_penalty`, renormalization, <3 weeks training for weekly seasonality,
+ `bucket_span` too large, wrong detector function (`mean` vs `high_mean`), `incomplete_bucket_penalty`, suppression by
+ `custom_rules`.
+- **Unexpectedly high:** insufficient history (early training over-flags), high-cardinality split (too few points per
+ entity), `use_null: true` on a sparse field.
+
+### Tool chain
+
+| Purpose | Tools |
+| ---------------------- | ---------------------------------------------------------------------------------- |
+| Records + explanation | `ad_query_anomaly_records` (exact `job_id_pattern`) |
+| Renormalization drift | `ad_rca_score_reassessment` (`score_drift = initial_record_score - record_score`) |
+| Model bounds (visual) | `ad_get_model_plot` — actual outside `model_lower`/`model_upper` = anomaly |
+| Forecast overlap | `ad_get_forecast_results` |
+| Influencer attribution | `ad_query_influencers` |
+| Config & detector | `ad_get_job_datafeed_config` — `bucket_span`, function, `custom_rules`, `use_null` |
+| Categorization | `ad_get_categories` |
+| Model snapshots | `ad_get_model_snapshots` |
+| Structured diagnostic | **`ad_wf_troubleshoot_anomaly_score`** (full decision tree) |
+
+### Decision tree (`ad_wf_troubleshoot_anomaly_score`)
+
+1. `ad_get_jobs` — ≥3 weeks data for weekly seasonality?
+2. `ad_ts_model_memory_health` — `memory_status` healthy?
+3. `ad_ts_delayed_data_annotations` — no incomplete buckets?
+4. `ad_query_anomaly_records` — compare `record_score` vs `initial_record_score`.
+5. `ad_get_job_datafeed_config` — `bucket_span`, detector function, `custom_rules`, `use_null`.
+6. `ad_get_model_plot` — wide bounds → `high_variance_penalty`.
+7. `ad_rca_score_reassessment` — renormalization drift across history.
+8. Explain `anomaly_score_explanation` factors.
+
+### Rules
+
+1. **Always show both `initial_record_score` and `record_score`** — the gap is the renormalization story.
+2. **Explain renormalization before diagnosing config** — score drift is the most common "score dropped" cause and needs
+ no config change.
+3. **`actual << typical` with `count`/`low_count` is an absence anomaly** — distinguish outages from value spikes.
+4. **`high_variance_penalty` and `incomplete_bucket_penalty` explain most "low score" surprises** without remediation.
+5. **Weekly seasonality needs ≥3 weeks of training data** — flag young jobs as the cause.
+
+For detector function selection details, see
+[references/anomaly-detection-functions.md](references/anomaly-detection-functions.md).
+
+---
+
+## Mode: Troubleshoot — Job ops
+
+**When:** "missing documents", "datafeed stopped", "hard_limit", "results look wrong", lifecycle changes, calendars,
+CCS.
+
+### Common issues → fast paths
+
+| Issue | Fast path | Full decision tree |
+| ------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------- |
+| Missing docs / `query_delay` warning | `ad_ts_delayed_data_annotations` → `ad_ts_bucket_event_gaps` → `ad_ts_ingest_latency_estimate` → `ad_update_datafeed_query_delay` | `ad_wf_troubleshoot_query_delay` |
+| Memory `soft_limit` / `hard_limit` | `ad_ts_model_memory_health` → `ad_wf_ts_field_cardinality` → `ad_estimate_memory_requirement` → `ad_update_model_memory_limit` | `ad_wf_troubleshoot_memory_limit` |
+| Datafeed not running / job state | `ad_get_jobs` (state) → `ad_get_job_messages` → `ad_manage_datafeed` | — |
+| CCS / `remote_cluster:` indices | `ad_ts_ccs_diagnostics` | — |
+| Score sanity check | — | `ad_wf_troubleshoot_anomaly_score` |
+
+> `hard_limit` corrupts model state and causes downstream missing-doc false alarms (categorizer silently skips events
+> for unknown categories). **Fix memory before fixing `query_delay`.**
+
+### Memory concepts
+
+| Field | Meaning |
+| ----------------------------------- | ------------------------------------------------------- |
+| `model_bytes` | Current memory used |
+| `peak_model_bytes` | High-water mark since job opened |
+| `model_bytes_memory_limit` | Configured `model_memory_limit` |
+| `memory_status` | `ok` / `soft_limit` (pruning) / `hard_limit` (critical) |
+| `total_by_field_count > 100k` | `by_field` cardinality too high — dominant driver |
+| `total_partition_field_count > 10k` | Partition explosion |
+| `total_category_count > 10k` | Too many distinct log patterns |
+
+Prefer **`ad_estimate_memory_requirement`** (samples cardinality from source, calls Estimate Model Memory API) over
+heuristics like `peak_model_bytes * 1.3` — the heuristic ignores pure influencer and categorization memory.
+
+### Datafeed & timing concepts
+
+- **`query_delay`** — how far behind real time the datafeed queries. Too small → missing docs; too large → slower
+ alerts. Set to **P95 ingest latency + buffer** (default `60s`–`120s`).
+- **`delayed_data_check_config`** — how aggressively the datafeed checks for late data.
+- **`bucket_span`** — analysis interval. Align with data granularity and detection window.
+- **`frequency`** — defaults to `min(query_delay, bucket_span / 2)`.
+
+### Lifecycle for config changes (memory limit, query_delay)
+
+1. Stop datafeed: `ad_manage_datafeed` (`action=_stop`)
+2. Close job
+3. Update config: `ad_update_model_memory_limit`, `ad_update_datafeed_query_delay`,
+ `ad_update_delayed_data_check_config`
+4. Open job: `ad_open_job`
+5. Start datafeed: `ad_manage_datafeed` (`action=_start`)
+
+Recover a corrupted period without resetting the whole model: `ad_revert_model_snapshot`.
+
+### Tool surface
+
+| Category | Tools |
+| ---------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| Permissions / metadata | `ad_validate_ml_tool_permissions`, `ad_get_available_metadata`, `ad_get_jobs` |
+| Job + datafeed state | `ad_get_job_datafeed_config`, `ad_get_job_messages`, `ad_manage_datafeed`, `ad_preview_datafeed_with_latency` |
+| Timing / missing docs | `ad_ts_delayed_data_annotations`, `ad_ts_bucket_event_gaps`, `ad_ts_ingest_latency_estimate`, `ad_update_datafeed_query_delay`, `ad_update_delayed_data_check_config`, `ad_wf_troubleshoot_query_delay` |
+| Memory | `ad_ts_model_memory_health`, `ad_wf_ts_field_cardinality`, `ad_estimate_memory_requirement`, `ad_update_model_memory_limit`, `ad_wf_troubleshoot_memory_limit` |
+| Model / lifecycle | `ad_get_model_snapshots`, `ad_revert_model_snapshot`, `ad_open_job`, `ad_create_job` |
+| CCS | `ad_ts_ccs_diagnostics` |
+| Calendars | `ad_get_calendar_events`, `ad_create_calendar_event` |
+
+Full parameter tables, ES|QL templates, and REST step lists:
+[references/troubleshoot-anomaly-tool-reference.md](references/troubleshoot-anomaly-tool-reference.md).
+
+### Rules
+
+1. **`ad_validate_ml_tool_permissions` first** — missing privileges produce misleading empty results.
+2. **Fix memory before `query_delay`** — `hard_limit` corrupts state; `query_delay` fixes on a memory-limited job are
+ wasted.
+3. **Stop the datafeed before updating it.** Updating a running datafeed is rejected.
+4. **Close the job before updating memory limit.** Sequence above.
+5. **Prefer workflow tools (`ad_wf_*`) over manually chaining diagnostics** for complex decisions.
+6. **`ad_preview_datafeed_with_latency` before starting** — confirm the datafeed returns data after config changes.
+
+---
+
+## Mode: Manage — Create / configure jobs
+
+**When:** "set up a job", "create an ML detector", "monitor X over time", "detect rare/unusual/anomalous values".
+
+### 4-step workflow
+
+```text
+PUT _ml/anomaly_detectors/ # 1. Define job (ad_create_job)
+PUT _ml/datafeeds/datafeed- # 2. Define datafeed (ad_create_datafeed)
+POST _ml/anomaly_detectors//_open # 3a. Open job (ad_open_job)
+POST _ml/datafeeds/datafeed-/_start # 3b. Start datafeed (ad_manage_datafeed action=_start)
+GET _ml/anomaly_detectors//results/records # 4. Read results
+```
+
+### Process
+
+1. **Build configs.** Parse the user request into job + datafeed JSON with no null fields.
+2. **Apply smart defaults:**
+
+ | Field | Default | Override when |
+ | ---------------- | --------------------------------------- | ------------------------------------------------- |
+ | `bucket_span` | `"15m"` | User specifies a different span |
+ | `time_field` | `"@timestamp"` | User names a different timestamp field |
+ | `index` | `"logs-*"` | User specifies an index or pattern |
+ | `datafeed_query` | `{"match_all": {}}` | User mentions filters, processes, or time windows |
+ | `influencers` | by/over/partition fields from detectors | User adds extra influencer fields |
+ | `job_id` | Generated from user description | User provides an explicit ID |
+ | `query_delay` | `"60s"` | P95 ingest latency is higher |
+
+3. **Choose detector function** from user intent — full table in
+ [references/anomaly-detection-functions.md](references/anomaly-detection-functions.md):
+ - "high CPU" / "unusually large" → `high_mean` or `high_sum`
+ - "rare logins" / "unusual values" → `rare` (variants below)
+ - "too many requests" / "spike in count" → `high_count`
+
+ `rare` variants:
+ - Infrequent globally → `rare by_field_name: X`
+ - Infrequent vs peers → `rare by_field_name: X over_field_name: Y`
+ - Infrequent per segment → `rare by_field_name: X partition_field_name: Y`
+ - Infrequent per segment vs peers → `rare by_field_name: X over_field_name: Y partition_field_name: Z`
+
+4. **Validate.** `platform.core.get_index_mapping` on the target index to verify field existence/types →
+ `ad_validate_job_spec`. If errors, fix and re-validate (max 3 attempts).
+
+5. **Present and confirm.** Show the **complete** job + datafeed bodies formatted as the exact API calls. Ask for
+ approval **once**. If feedback, incorporate and re-present (up to 3 rounds).
+
+6. **Deploy.** After confirmation: `ad_create_job` → `ad_create_datafeed` → `ad_open_job` → `ad_manage_datafeed`
+ (`action=_start`). Report final `job_id` and `datafeed_id`.
+
+For **batch analysis on historical data**, pass `start` and `end` to the datafeed start call.
+
+> Worked examples (rare-username, DNS exfil, large-downloads) with full JSON bodies and datafeed filters:
+> [references/job-creation-recipes.md](references/job-creation-recipes.md).
+
+### Rules
+
+1. **Create job before datafeed.** Datafeed references job by ID.
+2. **Open job before starting datafeed.** Start on a closed job is rejected.
+3. **`query_delay` = P95 ingest latency + buffer** (60s–120s safe default).
+4. **Forecasts require non-population jobs** — `over_field_name` jobs cannot be forecasted; warn before attempting.
+5. **`by_field_name` vs `over_field_name`:** `by` compares entity to its own history; `over` compares to peer group in
+ the same bucket. `partition_field_name` = fully independent sub-model with its own normalization.
+6. **`bucket_span` matches detection granularity** — 15m for high-frequency, 1h for operational metrics, 1d for daily
+ patterns. Larger smooths short spikes; smaller increases noise.
+
+---
+
+## Registration (Kibana Agent Builder)
+
+Requires Node.js 18+. Defaults to `elastic`/`changeme` when no credentials are supplied.
+
+```bash
+cd skills/kibana/kibana-anomaly-detection
+
+# tools → workflows → skills
+node scripts/kibana-agent-builder.mjs all register --kibana-url http://localhost:5601
+
+# HTTPS with self-signed cert
+node scripts/kibana-agent-builder.mjs all register --kibana-url https://localhost:5601 --insecure
+```
+
+`all register` runs `tools register`, then `workflows register`, then `skills register`. Kibana allows **at most five**
+`tool_ids` per skill; the script fills them by scanning `SKILL.md` for tool mentions (in document order), then appends
+ids from `references/kibana/tools/esql/*.json` until the cap (workflow-only tools omitted by default). If you run
+`skills register` alone, run `tools register` first so those ids exist.
+
+Workflow tool exclusions and prefixes live in `scripts/agent_builder_constants.json`.
+
+**MCP API key permissions:**
+
+- Kibana: `read_onechat`, `space_read`
+- Index: `read`, `view_index_metadata` on `.ml-anomalies-*`, `.ml-annotations-*`, `.ml-notifications-*`, `.ml-config`
+- For source evidence: `read` on source data indices
+
+---
+
+## Tool inventory
+
+ES|QL tool specs live under `references/kibana/tools/esql/*.json`; workflow definitions under
+`references/kibana/workflows/*.yaml`. Each Mode section above lists the tools it uses. Full surface:
+[references/tools.md](references/tools.md) (ES|QL) and [references/workflow-tools.md](references/workflow-tools.md)
+(workflows).
+
+### Key system indices
+
+| Index | Relevant content |
+| --------------------- | ----------------------------------------------------------------------------------------------------------------------------- |
+| `.ml-anomalies-*` | `record`, `bucket`, `influencer`, `model_plot`, `model_forecast`, `model_snapshot`, `category_definition`, `model_size_stats` |
+| `.ml-config` | job/datafeed documents (visible even for never-run jobs) |
+| `.ml-annotations-*` | delayed data (`event == "delayed_data"`) |
+| `.ml-notifications-*` | job messages (`level`: info/warning/error) |
+
+---
+
+## Examples
+
+**RCA:** "Something caused a spike in our error rate at 2pm — what broke?" → Investigate → `ad_get_available_metadata` →
+`ad_query_anomaly_timeline` → `ad_rca_cross_job_entity_match` → `ad_rca_multi_job_entities` → RCA report.
+
+**Score drop:** "My anomaly score went from 90 to 55 — did the model change?" → Explain → `ad_rca_score_reassessment`
+for drift → explain renormalization if `score_drift` is large.
+
+**Memory limit:** "Job status shows `hard_limit` and results look wrong." → Troubleshoot → `ad_ts_model_memory_health` →
+`ad_wf_ts_field_cardinality` → `ad_estimate_memory_requirement` → `ad_update_model_memory_limit` (lifecycle: stop
+datafeed → close → update → open → start).
+
+**New job:** "Detect unusual error rates per host on nginx access logs." → Manage → `high_count` detector with
+`by_field_name: "host.keyword"` → validate → present → deploy.
+
+**Multi-mode:** "We had an incident last night, scores were high but now low — is the job healthy?" → Investigate the
+incident → Explain the score drift → Troubleshoot if `hard_limit` or delayed data is suspected.
+
+---
+
+## Guidelines
+
+1. **Pick a mode first.** Don't blend RCA logic with score-explanation logic in one response.
+2. **`ad_validate_ml_tool_permissions` first** on empty results — privileges are the most common false-negative cause.
+3. **Score bands are absolute thresholds**: `>75` critical, `50–75` warning, `25–50` minor, `<25` informational.
+4. **Multi-job entities are prime suspects.** Use `min_job_count=2` in `ad_rca_multi_job_entities`.
+5. **Show `initial_record_score` alongside `record_score`** — the gap tells the renormalization story.
+6. **Fix memory before `query_delay`.** `hard_limit` invalidates downstream diagnostics.
+7. **Stop datafeed → close job → update config → open job → start datafeed** for any config change to memory or query
+ delay.
+8. **Confirm RCAs with `ad_rca_source_evidence`.** Raw source documents are ground truth.
diff --git a/skills/kibana/kibana-anomaly-detection/package-lock.json b/skills/kibana/kibana-anomaly-detection/package-lock.json
new file mode 100644
index 0000000..31dd026
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/package-lock.json
@@ -0,0 +1,12 @@
+{
+ "name": "kibana-anomaly-detection-skills",
+ "version": "0.1.0",
+ "lockfileVersion": 3,
+ "requires": true,
+ "packages": {
+ "": {
+ "name": "kibana-anomaly-detection-skills",
+ "version": "0.1.0"
+ }
+ }
+}
diff --git a/skills/kibana/kibana-anomaly-detection/package.json b/skills/kibana/kibana-anomaly-detection/package.json
new file mode 100644
index 0000000..afab6e9
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/package.json
@@ -0,0 +1,13 @@
+{
+ "name": "kibana-anomaly-detection-skills",
+ "version": "0.1.0",
+ "private": true,
+ "type": "module",
+ "scripts": {
+ "tools:register": "node scripts/kibana-agent-builder.mjs tools register",
+ "workflows:register": "node scripts/kibana-agent-builder.mjs workflows register",
+ "skills:register": "node scripts/kibana-agent-builder.mjs skills register",
+ "all:register": "node scripts/kibana-agent-builder.mjs all register"
+ },
+ "dependencies": {}
+}
diff --git a/skills/kibana/kibana-anomaly-detection/references/README.md b/skills/kibana/kibana-anomaly-detection/references/README.md
new file mode 100644
index 0000000..6fb5277
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/README.md
@@ -0,0 +1,114 @@
+# kibana-anomaly-detection
+
+Claude Code plugin for Elastic ML anomaly detection — investigation, score explanation, and job operations via Kibana
+Agent Builder MCP tools.
+
+## What's included
+
+A single Agent Skill (`SKILL.md` at the package root, frontmatter `name: kibana-anomaly-detection`) covering four modes:
+**Investigate**, **Explain**, **Troubleshoot**, **Manage**. Domain-specific framing for security and observability lives
+under `references/`.
+
+**Kibana version:** Registration scripts are written for **Kibana 9.4+** (Agent Builder + Workflows). They use PUT for
+idempotent updates when the API supports it and fall back to DELETE + POST on stacks without tool/skill PUT so repeated
+`all register` runs stay safe.
+
+**Kibana Agent Builder assets** (under `references/kibana/`):
+
+- **24** ES|QL tool specs (`references/kibana/tools/esql/*.json`)
+- **23** workflow YAML files (`references/kibana/workflows/*.yaml`)
+- `scripts/kibana-agent-builder.mjs` — registers tools, workflows, and skills (Node.js 18+)
+
+## Reference pages
+
+| File | Purpose |
+| ---------------------------------------------------------------------------------------- | ------------------------------------------------------------- |
+| [score-reference.md](score-reference.md) | Score field definitions, bands, renormalization, explanation |
+| [anomaly-detection-functions.md](anomaly-detection-functions.md) | Detector function selection guide |
+| [investigate-anomaly-esql-tools.md](investigate-anomaly-esql-tools.md) | Full ES\|QL templates for the Investigate mode |
+| [troubleshoot-anomaly-tool-reference.md](troubleshoot-anomaly-tool-reference.md) | ES\|QL + workflow tool detail for the Troubleshoot mode |
+| [tools.md](tools.md) / [workflow-tools.md](workflow-tools.md) | Complete tool surface (ES\|QL and workflow) |
+| [protocols/investigation.md](protocols/investigation.md) | 14-step RCA protocol |
+| [worked-example.md](worked-example.md) | End-to-end investigation walkthrough |
+| [job-creation-recipes.md](job-creation-recipes.md) | `rare`, `high_mean`, `high_sum` job + datafeed recipes |
+| [permissions-matrix.md](permissions-matrix.md) | Privileges by tool category |
+| [anomaly-detection-openapi-spec-discover.md](anomaly-detection-openapi-spec-discover.md) | Discover ML REST endpoints via `.kibana_ai_openapi_spec_*` |
+| [security-anomaly-expert.md](security-anomaly-expert.md) | Threat-first framing (MITRE mappings, attack-chain protocol) |
+| [observability-anomaly-expert.md](observability-anomaly-expert.md) | SRE / reliability framing (degradation, capacity, regression) |
+
+## Using this package from agent-skills-sandbox
+
+This directory is part of the **agent-skills-sandbox** monorepo. There is **no** `install.sh` here — consume the skill
+by pointing your Agent Skills configuration at `skills/kibana/kibana-anomaly-detection/` (single `SKILL.md` at the
+package root). Restart your agent runtime after adding paths.
+
+## Kibana Agent Builder setup
+
+Requires Node.js 18+. Defaults to `elastic`/`changeme` when no credentials are supplied.
+
+```bash
+cd skills/kibana/kibana-anomaly-detection
+
+# Local Kibana — credentials default to elastic/changeme (tools → workflows → skills)
+node scripts/kibana-agent-builder.mjs all register --kibana-url http://localhost:5601
+
+# HTTPS with self-signed cert (common for local deployments)
+node scripts/kibana-agent-builder.mjs all register --kibana-url https://localhost:5601 --insecure
+```
+
+Environment-only flow (recommended):
+
+```bash
+cd skills/kibana/kibana-anomaly-detection
+export KIBANA_URL=https://localhost:5601
+export KIBANA_INSECURE=true
+node scripts/kibana-agent-builder.mjs all register
+```
+
+Elastic Cloud (API key):
+
+```bash
+cd skills/kibana/kibana-anomaly-detection
+export KIBANA_CLOUD_ID=""
+export KIBANA_API_KEY=""
+node scripts/kibana-agent-builder.mjs all register
+```
+
+`all register` runs `tools register`, `workflows register`, and `skills register`. The skill is posted with at most
+**five** `tool_ids` (Kibana limit): tools detected from `SKILL.md` text first, then supplemental ids from the
+registration script until the cap. Workflow-backed tools require Elastic Workflows (preview) where applicable;
+exclusions live in **`scripts/agent_builder_constants.json`**.
+
+**MCP API key permissions required:**
+
+- Kibana: `read_onechat`, `space_read`
+- Index: `read`, `view_index_metadata` on `.ml-anomalies-*`, `.ml-annotations-*`, `.ml-notifications-*`, `.ml-config`
+- For source evidence: `read` on source data indices
+
+## Layout (this repo)
+
+```text
+kibana-anomaly-detection/
+├── SKILL.md # Single skill: Investigate / Explain / Troubleshoot / Manage modes
+├── package.json # Dev deps for the registration script
+├── scripts/
+│ ├── kibana-agent-builder.mjs
+│ └── agent_builder_constants.json
+└── references/
+ ├── README.md
+ ├── score-reference.md
+ ├── anomaly-detection-functions.md
+ ├── investigate-anomaly-esql-tools.md
+ ├── troubleshoot-anomaly-tool-reference.md
+ ├── job-creation-recipes.md
+ ├── worked-example.md
+ ├── permissions-matrix.md
+ ├── tools.md
+ ├── workflow-tools.md
+ ├── security-anomaly-expert.md # threat-first framing
+ ├── observability-anomaly-expert.md # SRE / reliability framing
+ ├── protocols/investigation.md
+ └── kibana/
+ ├── tools/esql/
+ └── workflows/
+```
diff --git a/skills/kibana/kibana-anomaly-detection/references/anomaly-detection-functions.md b/skills/kibana/kibana-anomaly-detection/references/anomaly-detection-functions.md
new file mode 100644
index 0000000..c564ddb
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/anomaly-detection-functions.md
@@ -0,0 +1,194 @@
+# Elastic ML Anomaly Detection Functions
+
+Functions marked with `*` support **high/low one-sided variants** (e.g., `high_count`, `low_mean`) to detect anomalies
+in only one direction.
+
+---
+
+## Count Functions
+
+Analyze the **occurrence rate** of events or documents over time.
+
+| Function | Description |
+| ------------------- | --------------------------------------------------------------------- |
+| `count` \* | Number of documents in a bucket |
+| `non_zero_count` \* | Like `count` but ignores zero-count buckets — use for **sparse data** |
+| `distinct_count` \* | Cardinality (uniqueness) of values for a specific field |
+
+### `count` vs `non_zero_count`
+
+| Scenario | Use |
+| ------------------------------------------------------------- | ----------------------------------------------------- |
+| Events arrive in every bucket (e.g., web traffic, heartbeats) | `count` — zero buckets are meaningful (outage signal) |
+| Events arrive intermittently (e.g., batch jobs, error logs) | `non_zero_count` — zeros are expected, not anomalous |
+| Detecting a **drop to zero** as an outage | `count` with `low_count` variant |
+| Detecting **bursts** in normally sparse traffic | `non_zero_count` with `high_non_zero_count` variant |
+
+**Example:** An intrusion detection log index gets events only when triggered. Using `count` means the model learns that
+zero-event buckets are normal overnight — making it impossible to distinguish a genuine quiet night from a monitoring
+gap. Use `non_zero_count` so the model only learns from buckets that had events, and flags when event volumes spike
+unexpectedly.
+
+### `distinct_count`
+
+Counts the number of unique values for a field within each bucket. Useful for detecting credential stuffing (unusually
+high distinct usernames), data exfiltration (high distinct destination IPs), or DGA activity (high distinct DNS query
+names).
+
+---
+
+## Metric Functions
+
+Operate on **numerical fields** within the data.
+
+| Function | Description |
+| ------------------------- | -------------------------------------------------------------- |
+| `min` / `max` | Minimum or maximum value in a bucket |
+| `mean` \* | Average value |
+| `median` \* | Median value |
+| `sum` / `non_null_sum` \* | Total sum of a field; `non_null_sum` is for **sparse data** |
+| `varp` \* | Variance / volatility of a metric |
+| `metric` | Shorthand that applies `min`, `max`, and `mean` simultaneously |
+
+### Choosing the right metric function
+
+| Goal | Function |
+| ---------------------------------------------- | ---------------------------------- |
+| Detect **average** latency spike | `mean` or `high_mean` |
+| Detect **worst-case** latency (tail) | `max` |
+| Detect **sustained volume** drop | `low_sum` |
+| Detect **unusual volatility** (erratic metric) | `high_varp` |
+| Detect both high and low deviations | `mean` (bidirectional, default) |
+| Detect only spikes, not drops | `high_mean` |
+| Monitor noisy metrics with outliers | `median` (more robust than `mean`) |
+
+### `sum` vs `non_null_sum`
+
+Use `non_null_sum` when the field is frequently absent from documents (sparse). Like `non_zero_count`, it skips empty
+buckets so the model learns only from active periods.
+
+### `metric` shorthand
+
+Creates three detectors in one: `min`, `max`, and `mean` on the same field. Convenient for a quick initial setup, but
+produces three anomaly records per detection event. Prefer explicit functions once you know which direction matters.
+
+---
+
+## Advanced & Specialized Functions
+
+Handle complex analysis types such as rarity or geographic data.
+
+| Function | Description |
+| ----------------- | -------------------------------------------------------------------------------- |
+| `rare` | Identifies values that occur at **low frequency** compared to the dataset |
+| `freq_rare` | Finds population members that **cause rare values to occur frequently** |
+| `info_content` \* | Entropy of text strings — useful for detecting **encrypted/obfuscated commands** |
+| `lat_long` | Detects unusual **geographic locations** from latitude/longitude coordinates |
+| `time_of_day` | Detects behavioral changes relative to **time of day** |
+| `time_of_week` | Detects behavioral changes relative to **day of week** |
+
+---
+
+### `rare` vs `freq_rare`
+
+These two are often confused but answer different questions:
+
+| Function | Question answered | Requires `over_field` |
+| ----------- | ----------------------------------------------------------- | --------------------- |
+| `rare` | "Which values of the `by_field` are unusual globally?" | No |
+| `freq_rare` | "Which entities (over_field) frequently cause rare values?" | Yes |
+
+**`rare` example:** Detect rare `process.name` values across all hosts. If `svchost.exe` with unusual arguments appears
+only once in 30 days, `rare` flags it. The focus is the rarity of the _value_.
+
+**`freq_rare` example:** Detect which _users_ (`over_field`) frequently trigger rare process executions (`by_field`).
+Most users run rare processes occasionally, but a user running rare processes consistently is a lateral movement signal.
+The focus is the entity's behavior pattern.
+
+**Practical guidance:**
+
+- Use `rare` for hunting unknown unknowns — values that shouldn't exist at all.
+- Use `freq_rare` for insider threat and lateral movement scenarios — who is repeatedly doing unusual things.
+- `rare` generates high false-positive rates in noisy environments; use custom rules to suppress known-good rare values.
+- `freq_rare` requires an `over_field` (population) — without it, use `rare`.
+
+### Types of rare analysis
+
+Translate business goals to the correct `rare` detector configuration:
+
+| Goal | Example | Detector config |
+| -------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------- |
+| Find infrequent values for a field | Detect hosts occurring infrequently | `rare` `by_field_name: host` |
+| Find infrequent values compared to peers | Detect hosts visited by few users vs. other hosts; detect users visiting rare hosts | `rare` `by_field_name: host` `over_field_name: user` |
+| Find infrequent values segmented by another field | Per location, detect infrequently seen hosts | `rare` `by_field_name: host` `partition_field_name: location` |
+| Find infrequent values by another field compared to peers, segmented | Per location, detect hosts visited by few users vs. peers; detect users visiting rare hosts in a location | `rare` `by_field_name: host` `over_field_name: user` `partition_field_name: location` |
+
+---
+
+### `info_content`
+
+Measures the **Shannon entropy** of text strings. High entropy = high randomness = potentially encoded, encrypted, or
+machine-generated content.
+
+| Entropy range | Interpretation | Example |
+| ----------------- | ------------------------------------ | -------------------------------------------------- |
+| Low (predictable) | Normal human-readable text | `GET /api/users HTTP/1.1` |
+| Medium | Mixed structured/variable content | Log messages with variable IDs |
+| High (random) | Encoded, encrypted, or DGA-generated | `aGVsbG8gd29ybGQ=` (base64), `zxq7v9abc.com` (DGA) |
+
+**Use cases:**
+
+- **DNS query names** with `by_field_name: "dns.question.name"` — DGA malware generates high-entropy domain names (e.g.,
+ `xk9p2mnqabcdef.ru`).
+- **User-agent strings** — malware C2 frameworks often use randomized or encoded user agents.
+- **URL paths / query strings** — webshell commands embedded in request parameters.
+- **Command-line arguments** — base64-encoded PowerShell payloads.
+
+**`high_info_content`** (one-sided): only alerts on unusually high entropy, which is almost always the right choice for
+security use cases.
+
+---
+
+### `time_of_day` vs `time_of_week`
+
+Both detect _when_ something happens relative to established patterns, not _how much_.
+
+| Function | Granularity | Best for |
+| -------------- | ------------------ | -------------------------------------------------------- |
+| `time_of_day` | Hour/minute of day | Detecting off-hours access (3am login for a 9-5 user) |
+| `time_of_week` | Day of week | Detecting weekend/holiday activity (batch job on Sunday) |
+
+**`time_of_day` example:** A database admin who always connects between 08:00–18:00 on weekdays. `time_of_day` learns
+this pattern. A connection at 02:30 scores anomalous — even if the volume of activity is normal.
+
+**`time_of_week` example:** A data pipeline that runs Monday–Friday. `time_of_week` flags it running on Saturday.
+`time_of_day` would not catch this (the time of day, e.g., 08:00, may be normal).
+
+**Choosing between them:**
+
+- Use `time_of_day` when the anomaly is "wrong hour of the day."
+- Use `time_of_week` when the anomaly is "wrong day of the week."
+- Use both together (two detectors) for comprehensive temporal coverage.
+- Neither function cares about _count_ or _metric values_ — use `count`/`mean` detectors in the same job for
+ volume-based detection.
+
+---
+
+## One-Sided Variants
+
+Functions marked with `*` support `high_` and `low_` prefixes:
+
+| Variant | Detects |
+| ------------------------ | ------------------------------------------------------ |
+| `high_` | Only values significantly **above** the expected range |
+| `low_` | Only values significantly **below** the expected range |
+| `` (no prefix) | Both directions (bidirectional) |
+
+**When to use one-sided:**
+
+- `high_mean(response_time)` — only alert on latency spikes, not drops (a faster response is never a problem).
+- `low_count(login_events)` — only alert on unusually low login volume (could indicate authentication system failure).
+- `high_distinct_count(destination.ip)` — only alert on abnormally high unique destination IPs (exfiltration signal).
+
+Bidirectional variants (`mean`, `count`) generate alerts for both directions, which can produce noise when only one
+direction is operationally relevant.
diff --git a/skills/kibana/kibana-anomaly-detection/references/anomaly-detection-openapi-spec-discover.md b/skills/kibana/kibana-anomaly-detection/references/anomaly-detection-openapi-spec-discover.md
new file mode 100644
index 0000000..4e89588
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/anomaly-detection-openapi-spec-discover.md
@@ -0,0 +1,74 @@
+## ML API Spec Discovery
+
+The `.kibana_ai_openapi_spec_elasticsearch` index, if present, contains one document per API endpoint. Document shape:
+
+```json
+{
+ "description": "Forecasts are not supported for jobs that perform population analysis; an\nerror occurs if you try to create a forecast for a job that has an\n`over_field_name` in its configuration. Forecasts predict future behavior\nbased on historical data.\n\n## Required authorization\n\n* Cluster privileges: `manage_ml`\n",
+ "endpoint": "POST /_ml/anomaly_detectors/{job_id}/_forecast",
+ "method": "post",
+ "operationId": "ml-forecast",
+ "path": "/_ml/anomaly_detectors/{job_id}/_forecast",
+ "path.keyword": "/_ml/anomaly_detectors/{job_id}/_forecast",
+ "summary": "Predict future behavior of a time series",
+ "tags": "ml anomaly"
+}
+```
+
+Use this to make informed API requests.
+
+`summary` and `description` are **semantic text fields** — use `MATCH()` for natural-language lookup.
+
+### Step 1 — Confirm index exists
+
+```text
+platform.core.list_indices → check for .kibana_ai_openapi_spec_elasticsearch
+```
+
+or use ES|QL query
+
+```esql
+FROM .kibana_ai_openapi_spec_elasticsearch
+```
+
+If index missing or error, fall back to researching the official documentation:
+[Elastic ML API docs](https://www.elastic.co/docs/api/doc/elasticsearch/group/endpoint-ml) and
+[Elastic ML Anomaly API docs](https://www.elastic.co/docs/api/doc/elasticsearch/group/endpoint-ml-anomaly).
+
+### Step 2 — List all ML endpoints
+
+```esql
+FROM .kibana_ai_openapi_spec_elasticsearch
+| WHERE tags == "ml"
+| SORT endpoint ASC
+| LIMIT 100
+```
+
+### Step 3 — Look up a specific endpoint by path
+
+```esql
+FROM .kibana_ai_openapi_spec_elasticsearch
+| WHERE path LIKE "/_ml/calendars*"
+| SORT endpoint ASC
+```
+
+### Step 4 — Semantic search when you know the intent, not the path
+
+```esql
+FROM .kibana_ai_openapi_spec_elasticsearch
+| WHERE tags == "ml" AND MATCH(summary, "model memory limit")
+| LIMIT 10
+```
+
+```esql
+FROM .kibana_ai_openapi_spec_elasticsearch
+| WHERE tags == "ml" AND MATCH(description, "revert snapshot")
+| KEEP method, endpoint, summary, description
+| LIMIT 5
+```
+
+**When to use this:**
+
+- A workflow tool returns 400/404 and you suspect a wrong path or missing required field
+- You want to discover optional parameters not covered by the current tool definition
+- You're adding a new `ad_*` workflow tool and need the exact endpoint and request body schema
diff --git a/skills/kibana/kibana-anomaly-detection/references/investigate-anomaly-esql-tools.md b/skills/kibana/kibana-anomaly-detection/references/investigate-anomaly-esql-tools.md
new file mode 100644
index 0000000..aa93499
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/investigate-anomaly-esql-tools.md
@@ -0,0 +1,371 @@
+# Investigate mode — ES|QL tool reference
+
+Supporting detail for the **Investigate** mode of the parent [SKILL.md](../SKILL.md). Use these templates with Kibana
+Agent Builder ES|QL tools.
+
+## ES|QL Tools (15)
+
+### Discovery & Metadata
+
+### `ad_get_available_metadata`
+
+Discover all jobs and their configured metadata. **Call first** when jobs are unknown.
+
+```esql
+FROM .ml-config
+| WHERE job_type == "anomaly_detector"
+| STATS job_count = COUNT(*),
+ job_ids = VALUES(job_id),
+ functions = VALUES(`analysis_config.detectors.function`),
+ fields = VALUES(`analysis_config.detectors.field_name`),
+ by_fields = VALUES(`analysis_config.detectors.by_field_name`),
+ over_fields = VALUES(`analysis_config.detectors.over_field_name`),
+ partition_fields = VALUES(`analysis_config.detectors.partition_field_name`),
+ influencers = VALUES(`analysis_config.influencers`),
+ bucket_spans = VALUES(`analysis_config.bucket_span`)
+```
+
+_No parameters._
+
+---
+
+### `ad_get_jobs`
+
+List all jobs with full config: bucket_span, detector functions, field names, memory limit.
+
+```esql
+FROM .ml-config
+| WHERE job_type == "anomaly_detector"
+| KEEP job_id, `analysis_config.bucket_span`, `analysis_config.detectors.function`,
+ `analysis_config.detectors.field_name`, `analysis_config.detectors.partition_field_name`,
+ `analysis_config.detectors.by_field_name`, `analysis_config.detectors.over_field_name`,
+ `analysis_config.influencers`, `analysis_limits.model_memory_limit`, groups, description
+| SORT job_id ASC | LIMIT 100
+```
+
+_No parameters._
+
+---
+
+### `ad_discover_related_jobs`
+
+Find jobs sharing the same entity field name (partition/by/over). Call early even when job names differ completely.
+
+| Parameter | Type | Description |
+| --------- | ---- | ------------------- |
+| `job_id` | text | The job of interest |
+
+```esql
+FROM .ml-config
+| WHERE job_type == "anomaly_detector"
+| EVAL entity_field = COALESCE(`analysis_config.detectors.partition_field_name`,
+ `analysis_config.detectors.by_field_name`,
+ `analysis_config.detectors.over_field_name`)
+| MV_EXPAND entity_field
+| WHERE entity_field IS NOT NULL
+| STATS job_count = COUNT_DISTINCT(job_id), jobs = VALUES(job_id),
+ influencers = VALUES(`analysis_config.influencers`) BY entity_field
+| WHERE MV_CONTAINS(jobs, ?job_id)
+| SORT job_count DESC | LIMIT 50
+```
+
+---
+
+### Anomaly Records & Influencers
+
+### `ad_query_anomaly_records`
+
+Primary anomaly search. Cross-job with `*` or single-job with exact ID.
+
+| Parameter | Type | Description |
+| ---------------- | ------ | ------------------------------------------------------------------- |
+| `job_id_pattern` | text | LIKE wildcards: `*` all, `rcaeval-*` group, exact ID for drill-down |
+| `min_score` | double | 50 significant · 25 broad · 75 critical |
+| `start_time` | text | ISO 8601 |
+| `end_time` | text | ISO 8601 |
+
+```esql
+FROM .ml-anomalies-*
+| WHERE result_type == "record"
+ AND job_id LIKE ?job_id_pattern
+ AND record_score >= ?min_score
+ AND timestamp >= ?start_time AND timestamp <= ?end_time
+| SORT record_score DESC | LIMIT 50
+| KEEP job_id, timestamp, record_score, function, field_name, actual, typical,
+ by_field_name, by_field_value, over_field_name, over_field_value,
+ partition_field_name, partition_field_value,
+ multi_bucket_impact, initial_record_score, detector_index, probability
+```
+
+---
+
+### `ad_query_anomaly_timeline`
+
+Cross-job bucket timeline. Composite scores reveal coordinated events (5 jobs × 30 = composite 150).
+
+| Parameter | Type | Description |
+| ---------------- | ------ | ----------------------------------- |
+| `job_id_pattern` | text | LIKE wildcards |
+| `min_score` | double | 25 for signal boosting, 50 standard |
+| `start_time` | text | ISO 8601 |
+| `end_time` | text | ISO 8601 |
+
+```esql
+FROM .ml-anomalies-*
+| WHERE result_type == "bucket"
+ AND job_id LIKE ?job_id_pattern
+ AND anomaly_score >= ?min_score
+ AND timestamp >= ?start_time AND timestamp <= ?end_time
+| STATS max_score = MAX(anomaly_score), job_count = COUNT_DISTINCT(job_id),
+ jobs = VALUES(job_id), composite_score = SUM(anomaly_score),
+ avg_score = AVG(anomaly_score) BY timestamp
+| SORT timestamp
+```
+
+---
+
+### `ad_query_influencers`
+
+Most anomalous entities. `job_count > 1` filter = cross-job shared influencers = strongest RCA signal.
+
+| Parameter | Type | Description |
+| ---------------- | ------ | ---------------------------------- |
+| `job_id_pattern` | text | LIKE wildcards |
+| `min_score` | double | 25 for shared influencer discovery |
+| `start_time` | text | ISO 8601 |
+| `end_time` | text | ISO 8601 |
+
+```esql
+FROM .ml-anomalies-*
+| WHERE result_type == "influencer"
+ AND job_id LIKE ?job_id_pattern
+ AND influencer_score >= ?min_score
+ AND timestamp >= ?start_time AND timestamp <= ?end_time
+| STATS total_score = SUM(influencer_score), job_count = COUNT_DISTINCT(job_id),
+ jobs = VALUES(job_id), max_score = MAX(influencer_score)
+ BY influencer_field_name, influencer_field_value
+| SORT total_score DESC | LIMIT 30
+```
+
+---
+
+### RCA Tools
+
+### `ad_rca_multi_job_entities`
+
+**Strongest root cause signal.** Entities anomalous in 2+ jobs simultaneously. Resource faults → multi-job; network
+faults → single-job.
+
+| Parameter | Type | Description |
+| --------------- | ------ | ------------------------- |
+| `min_score` | double | 25 broad · 50 significant |
+| `min_job_count` | double | Use 2 for cross-job RCA |
+| `start_time` | text | ISO 8601 |
+| `end_time` | text | ISO 8601 |
+
+```esql
+FROM .ml-anomalies-*
+| WHERE result_type == "record"
+ AND record_score >= ?min_score
+ AND timestamp >= ?start_time AND timestamp <= ?end_time
+| STATS job_count = COUNT_DISTINCT(job_id), jobs = VALUES(job_id),
+ max_score = MAX(record_score), total_records = COUNT(*),
+ functions = VALUES(function), fields = VALUES(field_name)
+ BY partition_field_value
+| WHERE job_count >= ?min_job_count
+| SORT job_count DESC, max_score DESC | LIMIT 20
+```
+
+---
+
+### `ad_rca_cross_job_entity_match`
+
+From an alert's entity value, find ALL jobs where it's anomalous. Returns `first_anomaly` per job for chronology
+reconstruction.
+
+| Parameter | Type | Description |
+| -------------- | ------ | ------------------------------------------ |
+| `entity_value` | text | From alert's partition/by/over field value |
+| `min_score` | double | 10–25 for comprehensive search |
+| `start_time` | text | Wider than alert window (alert minus 6h) |
+| `end_time` | text | Alert plus 1–2h to catch delayed effects |
+
+```esql
+FROM .ml-anomalies-*
+| WHERE result_type == "record"
+ AND record_score >= ?min_score
+ AND timestamp >= ?start_time AND timestamp <= ?end_time
+ AND (partition_field_value == ?entity_value
+ OR by_field_value == ?entity_value
+ OR over_field_value == ?entity_value)
+| STATS max_score = MAX(record_score), anomaly_count = COUNT(*),
+ functions = VALUES(function), fields = VALUES(field_name),
+ first_anomaly = MIN(timestamp), last_anomaly = MAX(timestamp)
+ BY job_id
+| SORT max_score DESC
+```
+
+---
+
+### `ad_rca_detector_fingerprint`
+
+Incident fingerprint — which system aspects are anomalous (CPU? Latency? Error rate?).
+
+| Parameter | Type | Description |
+| ---------------- | ------ | -------------------- |
+| `job_id_pattern` | text | LIKE wildcards |
+| `min_score` | double | Minimum record_score |
+| `start_time` | text | ISO 8601 |
+| `end_time` | text | ISO 8601 |
+
+```esql
+FROM .ml-anomalies-*
+| WHERE result_type == "record"
+ AND job_id LIKE ?job_id_pattern
+ AND record_score >= ?min_score
+ AND timestamp >= ?start_time AND timestamp <= ?end_time
+| STATS count = COUNT(*), max_score = MAX(record_score), avg_score = AVG(record_score)
+ BY job_id, function, field_name, detector_index
+| SORT max_score DESC
+```
+
+---
+
+### `ad_rca_correlation`
+
+Temporally ordered anomalies for cascade analysis. Earliest anomaly for an entity → root cause direction.
+
+| Parameter | Type | Description |
+| ---------------- | ------ | --------------------------------------- |
+| `job_id_pattern` | text | LIKE wildcards (scope to related group) |
+| `min_score` | double | Minimum record_score |
+| `start_time` | text | ISO 8601 |
+| `end_time` | text | ISO 8601 |
+
+```esql
+FROM .ml-anomalies-*
+| WHERE result_type == "record"
+ AND job_id LIKE ?job_id_pattern
+ AND record_score >= ?min_score
+ AND timestamp >= ?start_time AND timestamp <= ?end_time
+| SORT timestamp ASC
+| KEEP job_id, timestamp, record_score, function, field_name,
+ by_field_name, by_field_value, partition_field_name, partition_field_value,
+ over_field_name, over_field_value, multi_bucket_impact
+| LIMIT 100
+```
+
+---
+
+### `ad_rca_blast_radius`
+
+Scope of impact — how many partitions/jobs are affected by a specific anomalous value.
+
+| Parameter | Type | Description |
+| ----------------- | ------ | --------------------------------------- |
+| `anomalous_value` | text | e.g. `node-backdoor`, `payment-service` |
+| `min_score` | double | 10–25 for weak-signal aggregation |
+| `start_time` | text | ISO 8601 |
+| `end_time` | text | ISO 8601 |
+
+```esql
+FROM .ml-anomalies-*
+| WHERE result_type == "record"
+ AND record_score >= ?min_score
+ AND timestamp >= ?start_time AND timestamp <= ?end_time
+ AND (by_field_value == ?anomalous_value
+ OR over_field_value == ?anomalous_value
+ OR partition_field_value == ?anomalous_value)
+| STATS affected_partitions = COUNT_DISTINCT(partition_field_value),
+ affected_jobs = COUNT_DISTINCT(job_id), total_anomalies = COUNT(*),
+ max_score = MAX(record_score),
+ time_span_hours = DATE_DIFF("hour", MIN(timestamp), MAX(timestamp))
+ BY by_field_value
+| SORT affected_partitions DESC
+```
+
+---
+
+### `ad_rca_entity_profile`
+
+Complete anomaly dossier for a suspect entity across ALL jobs and field types.
+
+| Parameter | Type | Description |
+| -------------- | ---- | ----------------------------------- |
+| `entity_value` | text | e.g. `server-01`, `payment-service` |
+| `start_time` | text | ISO 8601 |
+| `end_time` | text | ISO 8601 |
+
+```esql
+FROM .ml-anomalies-*
+| WHERE result_type == "record"
+ AND timestamp >= ?start_time AND timestamp <= ?end_time
+ AND (by_field_value == ?entity_value
+ OR over_field_value == ?entity_value
+ OR partition_field_value == ?entity_value)
+| SORT timestamp ASC | LIMIT 100
+| KEEP job_id, timestamp, record_score, function, field_name,
+ by_field_name, by_field_value, over_field_name, over_field_value,
+ partition_field_name, partition_field_value, actual, typical, multi_bucket_impact
+```
+
+---
+
+### `ad_rca_source_evidence`
+
+Raw source documents from the original data index. Get source index from `ad_get_job_datafeed_config` first.
+
+| Parameter | Type | Description |
+| -------------- | ---- | ------------------------------------------------------------------------ |
+| `source_index` | text | LIKE pattern from datafeed config (e.g. `rcaeval-re1-ob`, `otel-flat-*`) |
+| `start_time` | text | ISO 8601 |
+| `end_time` | text | ISO 8601 |
+
+```esql
+FROM * METADATA _index
+| WHERE _index LIKE ?source_index
+ AND @timestamp >= ?start_time AND @timestamp <= ?end_time
+| SORT @timestamp DESC | LIMIT 50
+```
+
+---
+
+### Log Categorization (when `by_field_name == "mlcategory"`)
+
+### `ad_get_categories`
+
+Category definitions (terms, regex, examples) for jobs with `categorization_field_name`.
+
+| Parameter | Type | Description |
+| --------- | ---- | ---------------------------- |
+| `job_id` | text | The anomaly detection job ID |
+
+```esql
+FROM .ml-anomalies-*
+| WHERE result_type == "category_definition" AND job_id == ?job_id
+| SORT category_id ASC
+| KEEP job_id, category_id, terms, regex, max_matching_length, examples
+| LIMIT 100
+```
+
+---
+
+### `ad_search_log_category_examples`
+
+Raw log samples for two-window comparison (baseline vs anomaly window). Compare variable parts — IPs, hostnames, error
+codes — to find what changed.
+
+| Parameter | Type | Description |
+| -------------- | ---- | --------------------------------------- |
+| `source_index` | text | LIKE pattern from datafeed config |
+| `start_time` | text | ISO 8601 (baseline: 24h before anomaly) |
+| `end_time` | text | ISO 8601 |
+
+```esql
+FROM * METADATA _index
+| WHERE _index LIKE ?source_index
+ AND @timestamp >= ?start_time AND @timestamp <= ?end_time
+| SORT @timestamp DESC | LIMIT 50
+```
+
+---
diff --git a/skills/kibana/kibana-anomaly-detection/references/job-creation-recipes.md b/skills/kibana/kibana-anomaly-detection/references/job-creation-recipes.md
new file mode 100644
index 0000000..573d46e
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/job-creation-recipes.md
@@ -0,0 +1,186 @@
+# Job creation recipes
+
+Worked examples for creating Elastic ML anomaly detection jobs and datafeeds via Agent Builder workflow tools.
+
+> For the high-level process, see the **Manage** section of the parent `SKILL.md`. For detector function selection, see
+> [anomaly-detection-functions.md](anomaly-detection-functions.md).
+
+---
+
+## API call sequence
+
+```text
+PUT _ml/anomaly_detectors/ # 1. Define job + detectors (ad_create_job)
+PUT _ml/datafeeds/datafeed- # 2. Define datafeed (source + query) (ad_create_datafeed)
+POST _ml/anomaly_detectors//_open # 3a. Open job (ad_open_job)
+POST _ml/datafeeds/datafeed-/_start # 3b. Start datafeed (ad_manage_datafeed action=_start)
+GET _ml/anomaly_detectors//results/records # 4. Read results
+```
+
+To stop: `ad_manage_datafeed` (`action=_stop`) → `POST _ml/anomaly_detectors//_close`.
+
+For **batch analysis on historical data**, pass `start` and `end` to the datafeed start call:
+
+```json
+POST _ml/datafeeds/datafeed-revenue_over_users_api/_start
+{ "start": "2024-01-01T00:00:00Z", "end": "2024-03-01T00:00:00Z" }
+```
+
+---
+
+## Key `analysis_config` fields
+
+| Field | Description |
+| ---------------------------------- | -------------------------------------------------------------------------------------------------------------------------- |
+| `bucket_span` | Analysis interval (e.g. `15m`, `1h`). Align with data granularity and detection window. |
+| `detectors[].function` | Analysis function (`high_sum`, `rare`, `mean`, etc). See [anomaly-detection-functions.md](anomaly-detection-functions.md). |
+| `detectors[].field_name` | Numeric field to analyze. |
+| `detectors[].over_field_name` | Population analysis — each entity compared to its peers in the same bucket. |
+| `detectors[].by_field_name` | Per-entity modeling — each entity compared to its own history. |
+| `detectors[].partition_field_name` | Fully independent sub-model per entity with its own score normalization. |
+| `influencers` | Fields to track as anomaly contributors (shown as `influencer_score`). |
+| `data_description.time_field` | Timestamp field for time series ordering (e.g. `@timestamp`, `order_date`). |
+
+## Key datafeed fields
+
+| Field | Description |
+| ------------- | --------------------------------------------------------------------------------------------------------------------------------- |
+| `indices` | Array of index patterns containing source data. |
+| `query` | Elasticsearch DSL to filter source documents. Defaults to `match_all`. |
+| `query_delay` | How far behind real time the datafeed queries. Set to P95 ingest latency + buffer (default `60s`–`120s`). Too low → missing docs. |
+| `scroll_size` | Documents fetched per scroll request. Default `1000`. |
+| `frequency` | How often the datafeed polls. Defaults to `min(query_delay, bucket_span / 2)`. |
+
+---
+
+## Recipe 1 — `rare` detector: rare usernames
+
+**User query:** "Create an ML job to detect rare usernames in login events across logs-\*"
+
+Verify `user.name` (keyword) and `@timestamp` (date) exist via `platform.core.get_index_mapping`. Validate with
+`ad_validate_job_spec`. Then `ad_create_job` → `ad_create_datafeed` → `ad_open_job` → `ad_manage_datafeed`
+(`action=_start`).
+
+**Job body:**
+
+```json
+{
+ "description": "Detects rare values of user.name during login activity",
+ "analysis_config": {
+ "bucket_span": "15m",
+ "detectors": [
+ { "function": "rare", "by_field_name": "user.name", "detector_description": "Rare user.name values" }
+ ],
+ "influencers": ["user.name", "source.ip"]
+ },
+ "data_description": { "time_field": "@timestamp" }
+}
+```
+
+**Datafeed body:**
+
+```json
+{ "job_id": "rare-login-usernames", "indices": ["logs-*"], "query": { "match_all": {} } }
+```
+
+---
+
+## Recipe 2 — `high_mean` detector: DNS exfiltration with datafeed filter
+
+**User query:** "detect hosts with unusually high DNS query volume per domain, only for external DNS traffic on port 53"
+
+Verify `dns.question.count` (numeric), `host.name` (keyword), `dns.question.name` (keyword), `@timestamp` (date).
+
+**Job body:**
+
+```json
+{
+ "description": "Detects unusually high DNS query volume per host, partitioned by queried domain",
+ "analysis_config": {
+ "bucket_span": "15m",
+ "detectors": [
+ {
+ "function": "high_mean",
+ "field_name": "dns.question.count",
+ "over_field_name": "host.name",
+ "partition_field_name": "dns.question.name",
+ "detector_description": "High DNS query volume per host per domain"
+ }
+ ],
+ "influencers": ["host.name", "dns.question.name", "source.ip"]
+ },
+ "data_description": { "time_field": "@timestamp" }
+}
+```
+
+**Datafeed body:**
+
+```json
+{
+ "job_id": "high-dns-query-volume-per-host",
+ "indices": ["logs-*"],
+ "query": {
+ "bool": { "filter": [{ "term": { "network.transport": "udp" } }, { "term": { "destination.port": 53 } }] }
+ }
+}
+```
+
+---
+
+## Recipe 3 — `high_sum` detector: large downloads with time range
+
+**User query:** "detect users downloading unusually large amounts of data for sshd and sftp processes in the last 30
+days"
+
+Verify `destination.bytes` (numeric), `user.name` (keyword), `process.name` (keyword), `@timestamp` (date).
+
+**Job body:**
+
+```json
+{
+ "description": "Detects unusually high total bytes downloaded per user for specific processes",
+ "analysis_config": {
+ "bucket_span": "1h",
+ "detectors": [
+ {
+ "function": "high_sum",
+ "field_name": "destination.bytes",
+ "by_field_name": "user.name",
+ "detector_description": "High total bytes downloaded per user"
+ }
+ ],
+ "influencers": ["user.name", "process.name", "source.ip"]
+ },
+ "data_description": { "time_field": "@timestamp" }
+}
+```
+
+**Datafeed body:**
+
+```json
+{
+ "job_id": "high-download-volume-per-user",
+ "indices": ["logs-*"],
+ "query": {
+ "bool": {
+ "filter": [
+ { "terms": { "process.name": ["sshd", "sftp"] } },
+ { "range": { "@timestamp": { "gte": "now-30d", "lte": "now" } } }
+ ]
+ }
+ }
+}
+```
+
+---
+
+## Retrieving results
+
+| Result type | Endpoint | Description |
+| ----------- | -------------------------------------------------------- | ------------------------------------------------------------------- |
+| Buckets | `GET _ml/anomaly_detectors//results/buckets` | Aggregate anomaly score per time bucket |
+| Records | `GET _ml/anomaly_detectors//results/records` | Individual anomaly records with `actual`, `typical`, `record_score` |
+| Influencers | `GET _ml/anomaly_detectors//results/influencers` | Entity contribution scores |
+| Forecast | `POST _ml/anomaly_detectors//_forecast` | Predict future values; specify `duration` (e.g. `"10d"`) |
+
+> Forecasts are **not supported** for population analysis jobs (`over_field_name` set).
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/agent/anomaly_detective.json b/skills/kibana/kibana-anomaly-detection/references/kibana/agent/anomaly_detective.json
new file mode 100644
index 0000000..28878e3
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/agent/anomaly_detective.json
@@ -0,0 +1,21 @@
+{
+ "tools": [
+ "ad_get_available_metadata",
+ "ad_get_jobs",
+ "ad_discover_related_jobs",
+ "ad_rca_cross_job_entity_match",
+ "ad_query_anomaly_records",
+ "ad_query_anomaly_timeline",
+ "ad_query_influencers",
+ "ad_rca_entity_profile",
+ "ad_rca_detector_fingerprint",
+ "ad_rca_correlation",
+ "ad_rca_blast_radius",
+ "ad_rca_multi_job_entities",
+ "ad_rca_source_evidence",
+ "ad_discover_jobs_by_datafeed_index",
+ "ad_get_job_datafeed_config",
+ "ad_get_categories",
+ "ad_search_log_category_examples"
+ ]
+}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/agent/anomaly_explainer.json b/skills/kibana/kibana-anomaly-detection/references/kibana/agent/anomaly_explainer.json
new file mode 100644
index 0000000..d96fc9f
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/agent/anomaly_explainer.json
@@ -0,0 +1,14 @@
+{
+ "tools": [
+ "ad_get_available_metadata",
+ "ad_get_jobs",
+ "ad_query_anomaly_records",
+ "ad_query_influencers",
+ "ad_rca_score_reassessment",
+ "ad_get_model_plot",
+ "ad_get_categories",
+ "ad_get_forecast_results",
+ "ad_wf_troubleshoot_anomaly_score",
+ "ad_get_job_datafeed_config"
+ ]
+}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/agent/anomaly_maintainer.json b/skills/kibana/kibana-anomaly-detection/references/kibana/agent/anomaly_maintainer.json
new file mode 100644
index 0000000..4b0ea19
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/agent/anomaly_maintainer.json
@@ -0,0 +1,28 @@
+{
+ "tools": [
+ "ad_get_available_metadata",
+ "ad_get_jobs",
+ "ad_get_job_messages",
+ "ad_get_model_snapshots",
+ "ad_ts_bucket_event_gaps",
+ "ad_ts_delayed_data_annotations",
+ "ad_ts_ingest_latency_estimate",
+ "ad_ts_model_memory_health",
+ "ad_wf_ts_field_cardinality",
+ "ad_get_job_datafeed_config",
+ "ad_manage_datafeed",
+ "ad_preview_datafeed_with_latency",
+ "ad_create_job",
+ "ad_revert_model_snapshot",
+ "ad_create_calendar_event",
+ "ad_get_calendar_events",
+ "ad_update_datafeed_query_delay",
+ "ad_update_delayed_data_check_config",
+ "ad_update_model_memory_limit",
+ "ad_estimate_memory_requirement",
+ "ad_validate_ml_tool_permissions",
+ "ad_ts_ccs_diagnostics",
+ "ad_wf_troubleshoot_query_delay",
+ "ad_wf_troubleshoot_memory_limit"
+ ]
+}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_discover_related_jobs.json b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_discover_related_jobs.json
new file mode 100644
index 0000000..698c9b6
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_discover_related_jobs.json
@@ -0,0 +1,14 @@
+{
+ "name": "ad_discover_related_jobs",
+ "description": "Given a job of interest, find all other jobs that share the same entity field name (partition, by, or over field) by reading job configs from .ml-config. Returns the shared field name, the full list of sibling jobs in that group, their configured influencer fields, and the total sibling count. A job appears in its primary entity field group (partition > by > over). Call this early in investigation to find sibling jobs even when job names and prefixes differ completely. Complement with ad_discover_jobs_by_datafeed_index for source index overlap.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-config | WHERE job_type == \"anomaly_detector\" | EVAL entity_field = COALESCE(`analysis_config.detectors.partition_field_name`, `analysis_config.detectors.by_field_name`, `analysis_config.detectors.over_field_name`) | MV_EXPAND entity_field | WHERE entity_field IS NOT NULL | STATS job_count = COUNT_DISTINCT(job_id), jobs = VALUES(job_id), influencers = VALUES(`analysis_config.influencers`) BY entity_field | WHERE MV_CONTAINS(jobs, ?job_id) | SORT job_count DESC | LIMIT 50"
+ },
+ "parameters": {
+ "job_id": {
+ "type": "string",
+ "description": "The job ID of interest. Returns all other jobs that share the same entity split field (partition, by, or over field) as this job."
+ }
+ }
+}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_available_metadata.json b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_available_metadata.json
new file mode 100644
index 0000000..20362e8
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_available_metadata.json
@@ -0,0 +1,9 @@
+{
+ "name": "ad_get_available_metadata",
+ "description": "Discover all anomaly detection jobs and their configured metadata: job IDs, detector functions, monitored field names, entity split fields (by/over/partition), influencer field names, and bucket spans. Reads from .ml-config so all jobs are visible even if they have never produced an anomaly. Returns a single summary row. Call this FIRST before other tools to learn valid parameter values for job_id, function, field_name, and entity field filters.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-config | WHERE job_type == \"anomaly_detector\" | STATS job_count = COUNT(*), job_ids = VALUES(job_id), functions = VALUES(`analysis_config.detectors.function`), fields = VALUES(`analysis_config.detectors.field_name`), by_fields = VALUES(`analysis_config.detectors.by_field_name`), over_fields = VALUES(`analysis_config.detectors.over_field_name`), partition_fields = VALUES(`analysis_config.detectors.partition_field_name`), influencers = VALUES(`analysis_config.influencers`), bucket_spans = VALUES(`analysis_config.bucket_span`)"
+ },
+ "parameters": {}
+}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_categories.json b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_categories.json
new file mode 100644
index 0000000..8b2fb01
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_categories.json
@@ -0,0 +1,14 @@
+{
+ "name": "ad_get_categories",
+ "description": "Get log categories for categorization jobs: category definitions, regex patterns, and examples. Useful for understanding what types of log messages the model has learned and which categories are generating anomalies.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-anomalies-* | WHERE result_type == \"category_definition\" AND job_id == ?job_id | SORT category_id ASC | KEEP job_id, category_id, terms, regex, max_matching_length, examples | LIMIT 100"
+ },
+ "parameters": {
+ "job_id": {
+ "type": "string",
+ "description": "The anomaly detection job ID"
+ }
+ }
+}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_forecast_results.json b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_forecast_results.json
new file mode 100644
index 0000000..8738a67
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_forecast_results.json
@@ -0,0 +1,14 @@
+{
+ "name": "ad_get_forecast_results",
+ "description": "Retrieve forecast predictions with upper/lower bounds for capacity planning. Queries model_forecast results from .ml-anomalies-*.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-anomalies-* | WHERE result_type == \"model_forecast\" AND job_id == ?job_id | SORT timestamp ASC | KEEP job_id, timestamp, forecast_prediction, forecast_upper, forecast_lower, bucket_span | LIMIT 200"
+ },
+ "parameters": {
+ "job_id": {
+ "type": "string",
+ "description": "The anomaly detection job ID"
+ }
+ }
+}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_job_messages.json b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_job_messages.json
new file mode 100644
index 0000000..51975dc
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_job_messages.json
@@ -0,0 +1,14 @@
+{
+ "name": "ad_get_job_messages",
+ "description": "Retrieve all notifications for a job from the ML notifications index, including all severity levels (info, warning, error). Covers datafeed warnings, delayed data alerts, missing document warnings, query errors, timeout messages, CCS connectivity issues, memory limit warnings, and job lifecycle events. Returns all levels so the agent can filter by level in its reasoning. Use this for both general notification browsing and targeted datafeed warning investigation.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-notifications-* | WHERE job_id == ?job_id | SORT timestamp DESC | KEEP timestamp, level, message, node_name, job_id | LIMIT 50"
+ },
+ "parameters": {
+ "job_id": {
+ "type": "string",
+ "description": "The anomaly detection job ID"
+ }
+ }
+}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_jobs.json b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_jobs.json
new file mode 100644
index 0000000..fc5f769
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_jobs.json
@@ -0,0 +1,9 @@
+{
+ "name": "ad_get_jobs",
+ "description": "List all anomaly detection jobs with their full configuration: bucket_span, detector functions, monitored field names, entity split fields (partition/by/over), influencers, memory limit, groups, and description. Reads from .ml-config so ALL jobs appear regardless of whether they have run or produced any anomaly results. Use this early in any investigation to understand what jobs exist and how they are configured.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-config | WHERE job_type == \"anomaly_detector\" | KEEP job_id, `analysis_config.bucket_span`, `analysis_config.detectors.function`, `analysis_config.detectors.field_name`, `analysis_config.detectors.partition_field_name`, `analysis_config.detectors.by_field_name`, `analysis_config.detectors.over_field_name`, `analysis_config.influencers`, `analysis_limits.model_memory_limit`, groups, description | SORT job_id ASC | LIMIT 100"
+ },
+ "parameters": {}
+}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_model_plot.json b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_model_plot.json
new file mode 100644
index 0000000..2ac387d
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_model_plot.json
@@ -0,0 +1,22 @@
+{
+ "name": "ad_get_model_plot",
+ "description": "Get model bounds (upper/lower/median) to explain why something was or was not flagged as anomalous. Shows the model's confidence interval at each time point. If actual value is within bounds, no anomaly; if outside, anomaly score depends on distance from bounds.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-anomalies-* | WHERE result_type == \"model_plot\" AND job_id == ?job_id AND timestamp >= ?start_time AND timestamp <= ?end_time | SORT timestamp ASC | KEEP job_id, timestamp, model_lower, model_upper, model_median, actual, partition_field_value, by_field_value | LIMIT 500"
+ },
+ "parameters": {
+ "job_id": {
+ "type": "string",
+ "description": "The anomaly detection job ID"
+ },
+ "start_time": {
+ "type": "string",
+ "description": "Start of time range in ISO 8601 format"
+ },
+ "end_time": {
+ "type": "string",
+ "description": "End of time range in ISO 8601 format"
+ }
+ }
+}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_model_snapshots.json b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_model_snapshots.json
new file mode 100644
index 0000000..0f914ed
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_get_model_snapshots.json
@@ -0,0 +1,14 @@
+{
+ "name": "ad_get_model_snapshots",
+ "description": "List available model snapshots for a job, including timestamp, description, and size. Used for model revert operations when data quality issues have corrupted the model.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-anomalies-* | WHERE result_type == \"model_snapshot\" AND job_id == ?job_id | SORT timestamp DESC | KEEP job_id, timestamp, description, snapshot_doc_count | LIMIT 20"
+ },
+ "parameters": {
+ "job_id": {
+ "type": "string",
+ "description": "The anomaly detection job ID"
+ }
+ }
+}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_query_anomaly_records.json b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_query_anomaly_records.json
new file mode 100644
index 0000000..aeff156
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_query_anomaly_records.json
@@ -0,0 +1,26 @@
+{
+ "name": "ad_query_anomaly_records",
+ "description": "Search anomaly records by score threshold and time range. Use job_id_pattern='*' for cross-job search (default) or an exact job ID for single-job drill-down. Supports global cross-job search, job-scoped deep dive, absence detection (actual << typical), and value trend analysis. This is the primary tool for questions like 'find anomalies related to entity X' as well as per-job investigation.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-anomalies-* | WHERE result_type == \"record\" AND job_id LIKE ?job_id_pattern AND record_score >= ?min_score AND timestamp >= ?start_time AND timestamp <= ?end_time | SORT record_score DESC | LIMIT 50 | KEEP job_id, timestamp, record_score, function, field_name, actual, typical, by_field_name, by_field_value, over_field_name, over_field_value, partition_field_name, partition_field_value, multi_bucket_impact, initial_record_score, detector_index, probability"
+ },
+ "parameters": {
+ "job_id_pattern": {
+ "type": "string",
+ "description": "Job ID pattern using LIKE wildcards (* = multi-char). Use '*' for all jobs (cross-job search), or an exact job ID to scope to a single job (e.g. 'rcaeval-ob-cpu'). Use 'rcaeval-*' to scope to a prefix group."
+ },
+ "min_score": {
+ "type": "string",
+ "description": "Minimum record_score threshold (0-100). Use 50 for significant anomalies, 25 for broader search, 75 for critical only."
+ },
+ "start_time": {
+ "type": "string",
+ "description": "Start of time range in ISO 8601 format (e.g. 2024-01-01T00:00:00Z)"
+ },
+ "end_time": {
+ "type": "string",
+ "description": "End of time range in ISO 8601 format"
+ }
+ }
+}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_query_anomaly_timeline.json b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_query_anomaly_timeline.json
new file mode 100644
index 0000000..68bfd1e
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_query_anomaly_timeline.json
@@ -0,0 +1,26 @@
+{
+ "name": "ad_query_anomaly_timeline",
+ "description": "Get a time-series summary of anomaly severity across all or selected jobs, bucketed by time. Use for building cross-job swimlanes, detecting anomaly storms (multiple jobs firing together), and computing composite signal scores. When multiple jobs have anomaly_score > threshold in the same bucket, it indicates a common external trigger. Scope to a job group with job_id_pattern (e.g. 'rcaeval-*') or use '*' for all jobs.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-anomalies-* | WHERE result_type == \"bucket\" AND job_id LIKE ?job_id_pattern AND anomaly_score >= ?min_score AND timestamp >= ?start_time AND timestamp <= ?end_time | STATS max_score = MAX(anomaly_score), job_count = COUNT_DISTINCT(job_id), jobs = VALUES(job_id), composite_score = SUM(anomaly_score), avg_score = AVG(anomaly_score) BY timestamp | SORT timestamp"
+ },
+ "parameters": {
+ "job_id_pattern": {
+ "type": "string",
+ "description": "Job ID pattern using LIKE wildcards. Use '*' for all jobs, or a prefix pattern to scope to a related group (e.g. 'rcaeval-*')."
+ },
+ "min_score": {
+ "type": "string",
+ "description": "Minimum bucket anomaly_score. Use 25 for signal boosting (catch coordinated low-severity events), 50 for standard, 75 for critical."
+ },
+ "start_time": {
+ "type": "string",
+ "description": "Start of time range in ISO 8601 format"
+ },
+ "end_time": {
+ "type": "string",
+ "description": "End of time range in ISO 8601 format"
+ }
+ }
+}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_query_influencers.json b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_query_influencers.json
new file mode 100644
index 0000000..ba73a9a
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_query_influencers.json
@@ -0,0 +1,26 @@
+{
+ "name": "ad_query_influencers",
+ "description": "Find the most unusual entities across all or selected jobs for a time range. Answers 'What entities are most anomalous right now?' and 'Which entities appear as influencers in MULTIPLE jobs simultaneously?' When require_multi_job semantics are needed, filter results where job_count > 1 to find shared influencers for cross-job RCA. Scope to a job group with job_id_pattern (e.g. 'rcaeval-*') or use '*' for all jobs.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-anomalies-* | WHERE result_type == \"influencer\" AND job_id LIKE ?job_id_pattern AND influencer_score >= ?min_score AND timestamp >= ?start_time AND timestamp <= ?end_time | STATS total_score = SUM(influencer_score), job_count = COUNT_DISTINCT(job_id), jobs = VALUES(job_id), max_score = MAX(influencer_score) BY influencer_field_name, influencer_field_value | SORT total_score DESC | LIMIT 30"
+ },
+ "parameters": {
+ "job_id_pattern": {
+ "type": "string",
+ "description": "Job ID pattern using LIKE wildcards. Use '*' for all jobs, or a prefix pattern to scope to a related group (e.g. 'rcaeval-*')."
+ },
+ "min_score": {
+ "type": "string",
+ "description": "Minimum influencer_score threshold. Use 25 for broad search (shared influencer discovery), 50 for significant entities."
+ },
+ "start_time": {
+ "type": "string",
+ "description": "Start of time range in ISO 8601 format"
+ },
+ "end_time": {
+ "type": "string",
+ "description": "End of time range in ISO 8601 format"
+ }
+ }
+}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_blast_radius.json b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_blast_radius.json
new file mode 100644
index 0000000..ad63bd0
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_blast_radius.json
@@ -0,0 +1,26 @@
+{
+ "name": "ad_rca_blast_radius",
+ "description": "Measure how widespread a threat or issue is by counting how many hosts, partitions, or entities are affected by a specific anomalous value. Given a value (e.g. process.name: node-backdoor), counts distinct affected partitions across all jobs. Also aggregates weak signals: entities with multiple low-score anomalies across different jobs that individually are below alerting thresholds but collectively indicate a real problem.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-anomalies-* | WHERE result_type == \"record\" AND record_score >= ?min_score AND timestamp >= ?start_time AND timestamp <= ?end_time AND (by_field_value == ?anomalous_value OR over_field_value == ?anomalous_value OR partition_field_value == ?anomalous_value) | STATS affected_partitions = COUNT_DISTINCT(partition_field_value), affected_jobs = COUNT_DISTINCT(job_id), total_anomalies = COUNT(*), max_score = MAX(record_score), time_span_hours = DATE_DIFF(\"hour\", MIN(timestamp), MAX(timestamp)) BY by_field_value | SORT affected_partitions DESC"
+ },
+ "parameters": {
+ "anomalous_value": {
+ "type": "string",
+ "description": "The specific anomalous value to measure blast radius for (e.g. 'node-backdoor', 'suspicious-script.ps1')"
+ },
+ "min_score": {
+ "type": "string",
+ "description": "Minimum record_score. Use low values (10-25) for weak signal aggregation."
+ },
+ "start_time": {
+ "type": "string",
+ "description": "Start of time range in ISO 8601 format"
+ },
+ "end_time": {
+ "type": "string",
+ "description": "End of time range in ISO 8601 format"
+ }
+ }
+}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_correlation.json b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_correlation.json
new file mode 100644
index 0000000..1379832
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_correlation.json
@@ -0,0 +1,26 @@
+{
+ "name": "ad_rca_correlation",
+ "description": "Find temporally correlated anomalies across different jobs. Supports two modes: (1) co_occurrence — anomalies from different jobs in overlapping time windows regardless of shared influencers, essential for cross-domain correlation (K8s + APM + logs); (2) ordered_sequence — anomalies sorted by time to detect cascading failures (network -> app -> DB propagation). The agent examines temporal ordering and job types to infer causality. Use job_id_pattern to scope to a subset of jobs (e.g. 'rcaeval-ob-*') when many jobs exist.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-anomalies-* | WHERE result_type == \"record\" AND job_id LIKE ?job_id_pattern AND record_score >= ?min_score AND timestamp >= ?start_time AND timestamp <= ?end_time | SORT timestamp ASC | KEEP job_id, timestamp, record_score, function, field_name, by_field_name, by_field_value, partition_field_name, partition_field_value, over_field_name, over_field_value, multi_bucket_impact | LIMIT 100"
+ },
+ "parameters": {
+ "job_id_pattern": {
+ "type": "string",
+ "description": "Job ID pattern using LIKE wildcards (* = multi-char). Use '*' for all jobs, 'rcaeval-*' for all RCAEval jobs, 'rcaeval-ob-*' for Online Boutique only, 'nab-*' for NAB jobs, etc."
+ },
+ "min_score": {
+ "type": "string",
+ "description": "Minimum record_score threshold"
+ },
+ "start_time": {
+ "type": "string",
+ "description": "Start of time range in ISO 8601 format"
+ },
+ "end_time": {
+ "type": "string",
+ "description": "End of time range in ISO 8601 format"
+ }
+ }
+}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_cross_job_entity_match.json b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_cross_job_entity_match.json
new file mode 100644
index 0000000..9cef60c
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_cross_job_entity_match.json
@@ -0,0 +1,26 @@
+{
+ "name": "ad_rca_cross_job_entity_match",
+ "description": "Starting from a specific entity value (e.g. a service name extracted from an alert), find ALL other jobs where that entity appears as anomalous — across partition, by, and over fields. Returns per-job summary with max score, anomaly count, detector functions, field names, and the first/last anomaly timestamps. Use first_anomaly to reconstruct chronology: the job that detected the entity earliest is closest to the root cause. Unlike ad_rca_multi_job_entities (which groups by partition_field_value only), this matches the entity value across ALL split field types.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-anomalies-* | WHERE result_type == \"record\" AND record_score >= ?min_score AND timestamp >= ?start_time AND timestamp <= ?end_time AND (partition_field_value == ?entity_value OR by_field_value == ?entity_value OR over_field_value == ?entity_value) | STATS max_score = MAX(record_score), anomaly_count = COUNT(*), functions = VALUES(function), fields = VALUES(field_name), first_anomaly = MIN(timestamp), last_anomaly = MAX(timestamp) BY job_id | SORT max_score DESC"
+ },
+ "parameters": {
+ "entity_value": {
+ "type": "string",
+ "description": "The entity value to search for across all jobs (e.g. 'frontend', 'server-01', 'payment-service'). Extract this from the alerting anomaly's partition_field_value, by_field_value, or over_field_value."
+ },
+ "min_score": {
+ "type": "string",
+ "description": "Minimum record_score threshold. Use 10-25 for comprehensive search (catches weak cascade signals), 50 for significant anomalies only."
+ },
+ "start_time": {
+ "type": "string",
+ "description": "Start of time range in ISO 8601 format. Use a wider window than the alert (e.g. alert_time minus 6 hours) to catch the full cascade."
+ },
+ "end_time": {
+ "type": "string",
+ "description": "End of time range in ISO 8601 format. Use alert_time plus 1-2 hours to catch delayed effects."
+ }
+ }
+}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_detector_fingerprint.json b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_detector_fingerprint.json
new file mode 100644
index 0000000..726356b
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_detector_fingerprint.json
@@ -0,0 +1,26 @@
+{
+ "name": "ad_rca_detector_fingerprint",
+ "description": "For a specific incident time window, produce a fingerprint showing exactly which detectors fired across all or selected jobs and what they monitor. Shows which aspects of the system are anomalous (CPU? Network? Error rate? Latency?). Group by job_id + function + field_name to understand the incident signature. Scope to a job group with job_id_pattern (e.g. 'rcaeval-*') or use '*' for all jobs.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-anomalies-* | WHERE result_type == \"record\" AND job_id LIKE ?job_id_pattern AND record_score >= ?min_score AND timestamp >= ?start_time AND timestamp <= ?end_time | STATS count = COUNT(*), max_score = MAX(record_score), avg_score = AVG(record_score) BY job_id, function, field_name, detector_index | SORT max_score DESC"
+ },
+ "parameters": {
+ "job_id_pattern": {
+ "type": "string",
+ "description": "Job ID pattern using LIKE wildcards. Use '*' for all jobs, or a prefix pattern to scope to a related group (e.g. 'rcaeval-*')."
+ },
+ "min_score": {
+ "type": "string",
+ "description": "Minimum record_score threshold"
+ },
+ "start_time": {
+ "type": "string",
+ "description": "Start of time range in ISO 8601 format"
+ },
+ "end_time": {
+ "type": "string",
+ "description": "End of time range in ISO 8601 format"
+ }
+ }
+}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_entity_profile.json b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_entity_profile.json
new file mode 100644
index 0000000..98d0696
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_entity_profile.json
@@ -0,0 +1,22 @@
+{
+ "name": "ad_rca_entity_profile",
+ "description": "Build a complete anomaly dossier for a suspect entity across ALL jobs. Given an entity value (e.g. host.name: server-01), shows every anomaly where it appeared as an influencer, by_field, partition_field, or over_field value. Use after identifying a suspect via ad_query_influencers to build full context.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-anomalies-* | WHERE result_type == \"record\" AND timestamp >= ?start_time AND timestamp <= ?end_time AND (by_field_value == ?entity_value OR over_field_value == ?entity_value OR partition_field_value == ?entity_value) | SORT timestamp ASC | LIMIT 100 | KEEP job_id, timestamp, record_score, function, field_name, by_field_name, by_field_value, over_field_name, over_field_value, partition_field_name, partition_field_value, actual, typical, multi_bucket_impact"
+ },
+ "parameters": {
+ "entity_value": {
+ "type": "string",
+ "description": "The entity value to profile (e.g. 'server-01', 'payment-service', 'user@example.com')"
+ },
+ "start_time": {
+ "type": "string",
+ "description": "Start of time range in ISO 8601 format"
+ },
+ "end_time": {
+ "type": "string",
+ "description": "End of time range in ISO 8601 format"
+ }
+ }
+}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_multi_job_entities.json b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_multi_job_entities.json
new file mode 100644
index 0000000..1dc7283
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_multi_job_entities.json
@@ -0,0 +1,26 @@
+{
+ "name": "ad_rca_multi_job_entities",
+ "description": "Find entities that are anomalous in MULTIPLE jobs simultaneously — the strongest root cause signal. Returns entities ranked by the number of distinct jobs they appear in, with per-job max scores and detector functions. Entities in 2+ jobs are prime root cause candidates (e.g., a service with CPU AND latency anomalies); entities in only 1 job are likely secondary effects or victims. This is the key discriminator for RCA: resource faults produce multi-job anomalies, network faults produce single-job anomalies.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-anomalies-* | WHERE result_type == \"record\" AND record_score >= ?min_score AND timestamp >= ?start_time AND timestamp <= ?end_time | STATS job_count = COUNT_DISTINCT(job_id), jobs = VALUES(job_id), max_score = MAX(record_score), total_records = COUNT(*), functions = VALUES(function), fields = VALUES(field_name) BY partition_field_value | WHERE job_count >= ?min_job_count | SORT job_count DESC, max_score DESC | LIMIT 20"
+ },
+ "parameters": {
+ "min_score": {
+ "type": "string",
+ "description": "Minimum record_score threshold. Use 25 for broader search (catches weak cascade signals), 50 for significant anomalies."
+ },
+ "min_job_count": {
+ "type": "string",
+ "description": "Minimum number of distinct jobs the entity must appear in. Use 2 to find cross-job root cause candidates, 1 to include all entities."
+ },
+ "start_time": {
+ "type": "string",
+ "description": "Start of time range in ISO 8601 format"
+ },
+ "end_time": {
+ "type": "string",
+ "description": "End of time range in ISO 8601 format"
+ }
+ }
+}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_score_reassessment.json b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_score_reassessment.json
new file mode 100644
index 0000000..32fffd2
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_score_reassessment.json
@@ -0,0 +1,26 @@
+{
+ "name": "ad_rca_score_reassessment",
+ "description": "Find anomalies where the model has significantly changed its assessment over time due to renormalization. This tool defines score_drift = initial_record_score - record_score. When renormalization lowers the current score, initial_record_score stays higher → large positive drift (initial >> current). Large negative drift means the current score rose versus the initial snapshot (upward reconsideration). Records where scores stayed high indicate persistent anomalies the model never explained away. Scope to a specific job with job_id_pattern or use '*' for all jobs.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-anomalies-* | WHERE result_type == \"record\" AND job_id LIKE ?job_id_pattern AND timestamp >= ?start_time AND timestamp <= ?end_time | EVAL score_drift = initial_record_score - record_score | WHERE ABS(score_drift) >= ?min_drift | SORT ABS(score_drift) DESC | LIMIT 30 | KEEP job_id, timestamp, initial_record_score, record_score, score_drift, function, field_name, by_field_value, partition_field_value"
+ },
+ "parameters": {
+ "job_id_pattern": {
+ "type": "string",
+ "description": "Job ID pattern using LIKE wildcards. Use '*' for all jobs, or an exact job ID to scope to a single job (e.g. 'rcaeval-ob-cpu')."
+ },
+ "min_drift": {
+ "type": "string",
+ "description": "Minimum score point difference between initial and current score (default 20)"
+ },
+ "start_time": {
+ "type": "string",
+ "description": "Start of time range in ISO 8601 format"
+ },
+ "end_time": {
+ "type": "string",
+ "description": "End of time range in ISO 8601 format"
+ }
+ }
+}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_source_evidence.json b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_source_evidence.json
new file mode 100644
index 0000000..e4cab76
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_rca_source_evidence.json
@@ -0,0 +1,22 @@
+{
+ "name": "ad_rca_source_evidence",
+ "description": "After identifying an anomaly, retrieve raw source documents from the ORIGINAL data index for the anomaly's time window. This is the evidence drilldown — see the actual log lines, traces, or metrics that caused the statistical deviation. Without this, the agent can say 'something unusual happened' but cannot say 'here is what actually happened'.\n\nUsage: First find the source index from the job's datafeed configuration (via ad_get_job_datafeed_config). Pass the index name as source_index using LIKE wildcards (* for multi-char, ? for single-char). The tool returns raw documents with all original fields from that index.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM * METADATA _index | WHERE _index LIKE ?source_index AND @timestamp >= ?start_time AND @timestamp <= ?end_time | SORT @timestamp DESC | LIMIT 50"
+ },
+ "parameters": {
+ "source_index": {
+ "type": "string",
+ "description": "Source data index name or LIKE pattern (* = multi-char wildcard). Get this from the job's datafeed config. Examples: 'rcaeval-re1-ob', 'nab', 'otel-flat-*', 'smd'"
+ },
+ "start_time": {
+ "type": "string",
+ "description": "Start of time range in ISO 8601 format"
+ },
+ "end_time": {
+ "type": "string",
+ "description": "End of time range in ISO 8601 format"
+ }
+ }
+}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_search_log_category_examples.json b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_search_log_category_examples.json
new file mode 100644
index 0000000..4523ae9
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_search_log_category_examples.json
@@ -0,0 +1,22 @@
+{
+ "name": "ad_search_log_category_examples",
+ "description": "Search for log message examples in the source data for a specific time window. Used for two-window comparison in log categorization RCA: run once for a baseline window (e.g., 24h before anomaly) and once for the anomaly window. Compare the samples to identify what changed in the variable parts (IPs, hostnames, error codes, paths) that caused the category count anomaly. Returns raw log documents so you can inspect the categorization_field_name field specified in the job config.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM * METADATA _index | WHERE _index LIKE ?source_index AND @timestamp >= ?start_time AND @timestamp <= ?end_time | SORT @timestamp DESC | LIMIT 50"
+ },
+ "parameters": {
+ "source_index": {
+ "type": "string",
+ "description": "Source data index name or LIKE pattern from the job's datafeed config. Get this from ad_get_job_datafeed_config. Examples: 'logs-*', 'filebeat-*', 'otel-logs-*'"
+ },
+ "start_time": {
+ "type": "string",
+ "description": "Start of time range in ISO 8601 format. For baseline window, use a period before the anomaly (e.g., 24h before)."
+ },
+ "end_time": {
+ "type": "string",
+ "description": "End of time range in ISO 8601 format"
+ }
+ }
+}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_ts_bucket_event_gaps.json b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_ts_bucket_event_gaps.json
new file mode 100644
index 0000000..b437cf5
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_ts_bucket_event_gaps.json
@@ -0,0 +1,22 @@
+{
+ "name": "ad_ts_bucket_event_gaps",
+ "description": "Find buckets with zero or suspiciously low event counts for a specific job. These are the buckets actually affected by missing data. Compare event_count across buckets to identify time ranges where data was lost. Correlate with delayed data annotations and anomaly scores to confirm false positives from missing data.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-anomalies-* | WHERE result_type == \"bucket\" AND job_id == ?job_id AND timestamp >= ?start_time AND timestamp <= ?end_time | SORT timestamp ASC | KEEP timestamp, event_count, anomaly_score, bucket_span, is_interim | LIMIT 500"
+ },
+ "parameters": {
+ "job_id": {
+ "type": "string",
+ "description": "The anomaly detection job ID"
+ },
+ "start_time": {
+ "type": "string",
+ "description": "Start of time range in ISO 8601 format"
+ },
+ "end_time": {
+ "type": "string",
+ "description": "End of time range in ISO 8601 format"
+ }
+ }
+}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_ts_delayed_data_annotations.json b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_ts_delayed_data_annotations.json
new file mode 100644
index 0000000..142f55f
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_ts_delayed_data_annotations.json
@@ -0,0 +1,14 @@
+{
+ "name": "ad_ts_delayed_data_annotations",
+ "description": "Retrieve all delayed data annotations for a job, showing exactly when and how many documents were missed. The annotation field contains text like 'Datafeed has missed 30 documents due to ingest latency...' \u2014 frequent annotations indicate chronic ingest latency. This is the starting point for any 'missing documents' investigation.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-annotations-* | WHERE job_id == ?job_id AND event == \"delayed_data\" | SORT timestamp DESC | KEEP job_id, timestamp, end_timestamp, annotation | LIMIT 100"
+ },
+ "parameters": {
+ "job_id": {
+ "type": "string",
+ "description": "The anomaly detection job ID"
+ }
+ }
+}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_ts_ingest_latency_estimate.json b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_ts_ingest_latency_estimate.json
new file mode 100644
index 0000000..9895d1e
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_ts_ingest_latency_estimate.json
@@ -0,0 +1,22 @@
+{
+ "name": "ad_ts_ingest_latency_estimate",
+ "description": "Measure actual ingest latency by comparing event timestamps with ingestion timestamps in the SOURCE data. Determines whether the current query_delay is sufficient. If P95(event.ingested - @timestamp) > query_delay, data will be lost. Requires the source index to have an event.ingested or _ingest.timestamp field.\n\nUsage: Get the source index from the job's datafeed config (via ad_get_job_datafeed_config). Pass it as source_index using LIKE wildcards if needed.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM * METADATA _index | WHERE _index LIKE ?source_index AND @timestamp >= ?start_time AND @timestamp <= ?end_time AND event.ingested IS NOT NULL | EVAL latency_seconds = DATE_DIFF(\"second\", @timestamp, event.ingested) | STATS p50_latency = PERCENTILE(latency_seconds, 50), p95_latency = PERCENTILE(latency_seconds, 95), p99_latency = PERCENTILE(latency_seconds, 99), max_latency = MAX(latency_seconds), doc_count = COUNT(*) | LIMIT 1"
+ },
+ "parameters": {
+ "source_index": {
+ "type": "string",
+ "description": "Source data index name or LIKE pattern (* = multi-char wildcard). Get this from the job's datafeed config. Examples: 'rcaeval-re1-ob', 'nab', 'otel-flat-*', 'smd'"
+ },
+ "start_time": {
+ "type": "string",
+ "description": "Start of time range in ISO 8601 format"
+ },
+ "end_time": {
+ "type": "string",
+ "description": "End of time range in ISO 8601 format"
+ }
+ }
+}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_ts_model_memory_health.json b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_ts_model_memory_health.json
new file mode 100644
index 0000000..01b5459
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/tools/esql/ad_ts_model_memory_health.json
@@ -0,0 +1,18 @@
+{
+ "name": "ad_ts_model_memory_health",
+ "description": "Get memory health and growth trend for a job. Returns model_size_stats records in reverse-chronological order. Use limit=1 for a fast current-state snapshot (hard_limit/soft_limit check); use limit=500 for full memory growth trend analysis (stable plateau / linear / exponential). Interpretation: hard_limit = CRITICAL (job blind to new entities), soft_limit = WARNING (aggressive pruning), model_bytes/model_bytes_memory_limit > 0.8 = APPROACHING LIMIT.",
+ "type": "esql",
+ "configuration": {
+ "query": "FROM .ml-anomalies-* | WHERE result_type == \"model_size_stats\" AND job_id == ?job_id | SORT timestamp DESC | KEEP job_id, timestamp, model_bytes, peak_model_bytes, model_bytes_memory_limit, model_bytes_exceeded, memory_status, total_by_field_count, total_over_field_count, total_partition_field_count, bucket_allocation_failures_count | LIMIT ?limit"
+ },
+ "parameters": {
+ "job_id": {
+ "type": "string",
+ "description": "The anomaly detection job ID"
+ },
+ "limit": {
+ "type": "string",
+ "description": "Number of historical records to return. Use 1 for current memory status (fast snapshot), 500 for full memory growth trend analysis."
+ }
+ }
+}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_create_calendar_event.yaml b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_create_calendar_event.yaml
new file mode 100644
index 0000000..d2c4398
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_create_calendar_event.yaml
@@ -0,0 +1,42 @@
+name: ad_create_calendar_event
+description: >
+ Add a scheduled event to a calendar to suppress false positives during known downtime, maintenance windows, or
+ holidays.
+enabled: true
+tags: ["anomaly-detection", "remediation"]
+
+inputs:
+ - name: calendar_id
+ type: string
+ description: The calendar ID
+ - name: event_description
+ type: string
+ description: "Event description (e.g., 'Planned maintenance window')"
+ - name: start_time
+ type: string
+ description: Event start time in ISO 8601 format
+ - name: end_time
+ type: string
+ description: Event end time in ISO 8601 format
+
+triggers:
+ - type: manual
+
+steps:
+ - name: create_event
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/calendars/{{ inputs.calendar_id }}/events
+ body:
+ events:
+ - description: "{{ inputs.event_description }}"
+ start_time: "{{ inputs.start_time }}"
+ end_time: "{{ inputs.end_time }}"
+
+ - name: result
+ type: console
+ with:
+ message: |
+ Created event '{{ inputs.event_description }}' on calendar {{ inputs.calendar_id }}
+ from {{ inputs.start_time }} to {{ inputs.end_time }}.
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_create_datafeed.yaml b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_create_datafeed.yaml
new file mode 100644
index 0000000..5633649
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_create_datafeed.yaml
@@ -0,0 +1,34 @@
+name: ad_create_datafeed
+description: >
+ Create or replace a datafeed for an anomaly detection job via PUT _ml/datafeeds/{datafeed_id}. Use after job creation
+ and before opening the job / starting the datafeed.
+enabled: true
+tags: ["anomaly-detection", "management"]
+
+inputs:
+ - name: datafeed_id
+ type: string
+ description: Datafeed ID (typically datafeed-{job_id})
+ - name: datafeed_body
+ type: string
+ description: >
+ Full datafeed configuration as JSON text. Parsed with json_parse so the PUT body is a structured object for the ML
+ API, not a JSON-encoded string.
+
+triggers:
+ - type: manual
+
+steps:
+ - name: put_datafeed
+ type: elasticsearch.request
+ with:
+ method: PUT
+ path: /_ml/datafeeds/{{ inputs.datafeed_id }}
+ body: "${{ inputs.datafeed_body | json_parse }}"
+
+ - name: result
+ type: console
+ with:
+ message: |
+ Datafeed {{ inputs.datafeed_id }}:
+ {{ steps.put_datafeed.output | json:2 }}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_create_job.yaml b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_create_job.yaml
new file mode 100644
index 0000000..61ee9b5
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_create_job.yaml
@@ -0,0 +1,36 @@
+name: ad_create_job
+description: >
+ Create a new anomaly detection job from a configuration. Stretch goal — for advanced agent use cases where the agent
+ helps design and create jobs based on data exploration. job_body is JSON text; the step uses Liquid json_parse and typed
+ interpolation (${{ }}) so the PUT body is a structured JSON object for the ML API, not a JSON-encoded string.
+enabled: true
+tags: ["anomaly-detection", "management"]
+
+inputs:
+ - name: job_id
+ type: string
+ description: The new job ID to create
+ - name: job_body
+ type: string
+ description: >
+ Full job configuration as JSON text (same shape as PUT /_ml/anomaly_detectors/{job_id}). Parsed with json_parse
+ before the request so the HTTP body is an object. Runners that support a native object input may still pass JSON
+ text here.
+
+triggers:
+ - type: manual
+
+steps:
+ - name: create_job
+ type: elasticsearch.request
+ with:
+ method: PUT
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}
+ body: "${{ inputs.job_body | json_parse }}"
+
+ - name: result
+ type: console
+ with:
+ message: |
+ Created job {{ inputs.job_id }}:
+ {{ steps.create_job.output | json:2 }}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_discover_jobs_by_datafeed_index.yaml b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_discover_jobs_by_datafeed_index.yaml
new file mode 100644
index 0000000..538c00e
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_discover_jobs_by_datafeed_index.yaml
@@ -0,0 +1,63 @@
+name: ad_discover_jobs_by_datafeed_index
+description: >
+ Given a job of interest, find all other jobs whose datafeed reads from overlapping source indices. Step 1 retrieves
+ the target job config (including datafeed_config.indices). Step 2 logs the target job summary. Step 3 iterates over
+ each index in that list and queries .ml-config for other datafeed documents containing the same index, logging matches
+ per index pattern. Jobs reading from the same indices monitor the same system and are strong candidates for cross-job
+ correlation.
+enabled: true
+tags: ["anomaly-detection", "rca", "discovery"]
+
+inputs:
+ - name: job_id
+ type: string
+ description: >
+ The job ID of interest whose source indices you want to match against all other jobs.
+
+triggers:
+ - type: manual
+
+steps:
+ - name: get_target_job
+ type: elasticsearch.request
+ with:
+ method: GET
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}
+
+ - name: log_target
+ type: console
+ with:
+ message: |
+ TARGET JOB: {{ inputs.job_id }}
+ Source indices: {{ steps.get_target_job.output.jobs[0].datafeed_config.indices }}
+
+ Searching for jobs with overlapping source indices...
+
+ - name: find_related_jobs
+ type: foreach
+ foreach: "${{ steps.get_target_job.output.jobs[0].datafeed_config.indices }}"
+ steps:
+ - name: search_matching_datafeeds
+ type: elasticsearch.search
+ with:
+ index: .ml-config
+ size: 200
+ _source: ["job_id", "datafeed_id", "indices"]
+ query:
+ bool:
+ filter:
+ - term:
+ config_type: datafeed
+ - term:
+ indices: "{{ foreach.item }}"
+ must_not:
+ - term:
+ job_id: "{{ inputs.job_id }}"
+
+ - name: log_matches
+ type: console
+ with:
+ message: |
+ [{{ foreach.index | plus: 1 }}/{{ foreach.total }}] Index: {{ foreach.item }}
+ Matching jobs ({{ steps.search_matching_datafeeds.output.hits.total.value }}):
+ {{ steps.search_matching_datafeeds.output.hits.hits | json:2 }}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_estimate_memory_requirement.yaml b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_estimate_memory_requirement.yaml
new file mode 100644
index 0000000..5230bfb
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_estimate_memory_requirement.yaml
@@ -0,0 +1,61 @@
+name: ad_estimate_memory_requirement
+description: >
+ Compute a principled model_memory_limit estimate by automatically sampling cardinality from source data and calling
+ the Estimate Model Memory API. Dramatically better than guessing or peak_model_bytes * 1.3 because it uses the exact
+ same estimation algorithm Elasticsearch uses internally. Steps: (1) retrieve job+datafeed config, (2) identify fields
+ requiring cardinality estimates, (3) compute overall_cardinality via cardinality aggregations, (4) compute
+ max_bucket_cardinality via date_histogram + cardinality + max_bucket pipeline, (5) call Estimate Model Memory API, (6)
+ compare with current state, (7) produce recommendation.
+enabled: true
+tags: ["anomaly-detection", "troubleshooting", "memory"]
+
+inputs:
+ - name: job_id
+ type: string
+ description: The anomaly detection job ID
+
+triggers:
+ - type: manual
+
+steps:
+ # Step 1: Retrieve job configuration
+ - name: get_job_config
+ type: elasticsearch.request
+ with:
+ method: GET
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}
+
+ - name: get_job_stats
+ type: elasticsearch.request
+ with:
+ method: GET
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}/_stats
+
+ # Step 2: Log what we found
+ # The agent (or manual reviewer) inspects the config to identify:
+ # - overall_fields: all partition/by/over field names from detectors
+ # - pure_influencers: influencer fields NOT used in any detector split
+ # - datafeed source indices and query
+ - name: log_config
+ type: console
+ with:
+ message: |
+ Job config retrieved for {{ inputs.job_id }}.
+ Analysis config: {{ steps.get_job_config.output | json:2 }}
+ Current stats: {{ steps.get_job_stats.output | json:2 }}
+
+ MANUAL STEP REQUIRED:
+ 1. From analysis_config.detectors, extract all unique by_field_name,
+ over_field_name, partition_field_name values → these are "overall_fields"
+ 2. From analysis_config.influencers, find fields NOT in any detector →
+ these are "pure_influencers"
+ 3. From datafeed_config, get indices[] and query{}
+ 4. Run cardinality aggregations on those indices for each field
+ 5. Run date_histogram(bucket_span) + cardinality sub-agg + max_bucket
+ pipeline for each pure influencer
+ 6. Call POST _ml/anomaly_detectors/_estimate_model_memory with the
+ analysis_config and computed cardinalities
+ 7. Compare the estimate with current model_memory_limit and model_bytes
+
+ See tools/workflow/ad_estimate_memory_requirement.json for the full
+ step-by-step algorithm.
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_get_calendar_events.yaml b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_get_calendar_events.yaml
new file mode 100644
index 0000000..81496d5
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_get_calendar_events.yaml
@@ -0,0 +1,28 @@
+name: ad_get_calendar_events
+description: >
+ Get scheduled events from calendars (maintenance windows, holidays). These suppress anomaly detection during known
+ downtime periods.
+enabled: true
+tags: ["anomaly-detection", "config"]
+
+inputs:
+ - name: calendar_id
+ type: string
+ description: The calendar ID
+
+triggers:
+ - type: manual
+
+steps:
+ - name: get_events
+ type: elasticsearch.request
+ with:
+ method: GET
+ path: /_ml/calendars/{{ inputs.calendar_id }}/events
+
+ - name: result
+ type: console
+ with:
+ message: |
+ Calendar events for {{ inputs.calendar_id }}:
+ {{ steps.get_events.output | json:2 }}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_get_job_datafeed_config.yaml b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_get_job_datafeed_config.yaml
new file mode 100644
index 0000000..fdb1ddc
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_get_job_datafeed_config.yaml
@@ -0,0 +1,36 @@
+name: ad_get_job_datafeed_config
+description: >
+ Fetch complete job and datafeed configuration in one call: detectors, by/over/partition fields, bucket_span,
+ frequency, query_delay, delayed_data_check_config, source indices, and datafeed query. Essential for troubleshooting
+ and for ad_estimate_memory_requirement.
+enabled: true
+tags: ["anomaly-detection", "config"]
+
+inputs:
+ - name: job_id
+ type: string
+ description: The anomaly detection job ID
+
+triggers:
+ - type: manual
+
+steps:
+ - name: get_job
+ type: elasticsearch.request
+ with:
+ method: GET
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}
+
+ - name: get_job_stats
+ type: elasticsearch.request
+ with:
+ method: GET
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}/_stats
+
+ - name: summary
+ type: console
+ with:
+ message: |
+ Job: {{ inputs.job_id }}
+ Config: {{ steps.get_job.output | json:2 }}
+ Stats: {{ steps.get_job_stats.output | json:2 }}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_get_log_categories.yaml b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_get_log_categories.yaml
new file mode 100644
index 0000000..4bbac2b
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_get_log_categories.yaml
@@ -0,0 +1,32 @@
+name: ad_get_log_categories
+description: >
+ Retrieve ML log category details — terms, regex pattern, and example messages — for a specific category from a log
+ categorization job. Use when investigating anomalies where by_field_name == "mlcategory" to understand what type of
+ log message the category represents before comparing samples across time windows.
+enabled: true
+tags: ["anomaly-detection", "log-categorization"]
+
+inputs:
+ - name: job_id
+ type: string
+ description: The categorization job ID
+ - name: category_id
+ type: string
+ description: "The category ID from the anomaly record's by_field_value"
+
+triggers:
+ - type: manual
+
+steps:
+ - name: get_categories
+ type: elasticsearch.request
+ with:
+ method: GET
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}/results/categories/{{ inputs.category_id }}
+
+ - name: result
+ type: console
+ with:
+ message: |
+ Log category {{ inputs.category_id }} for job {{ inputs.job_id }}:
+ {{ steps.get_categories.output | json:2 }}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_manage_datafeed.yaml b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_manage_datafeed.yaml
new file mode 100644
index 0000000..8c79c63
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_manage_datafeed.yaml
@@ -0,0 +1,31 @@
+name: ad_manage_datafeed
+description: >
+ Start or stop a datafeed via POST _ml/datafeeds/{id}/{_start|_stop}. Used in remediation sequences (for example stop
+ before updating query_delay, then restart). For payload preview use ad_preview_datafeed_with_latency (GET
+ _ml/datafeeds/{id}/_preview); preview is not supported here because it requires GET, not POST.
+enabled: true
+tags: ["anomaly-detection", "remediation"]
+
+inputs:
+ - name: datafeed_id
+ type: string
+ description: "The datafeed ID (typically 'datafeed-{job_id}')"
+ - name: action
+ type: string
+ description: "Action to perform: _start or _stop only"
+
+triggers:
+ - type: manual
+
+steps:
+ - name: manage_datafeed
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/datafeeds/{{ inputs.datafeed_id }}/{{ inputs.action }}
+
+ - name: result
+ type: console
+ with:
+ message: |
+ Datafeed {{ inputs.datafeed_id }} action {{ inputs.action }}: {{ steps.manage_datafeed.output | json:2 }}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_open_job.yaml b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_open_job.yaml
new file mode 100644
index 0000000..9e9182a
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_open_job.yaml
@@ -0,0 +1,27 @@
+name: ad_open_job
+description: >
+ Open an anomaly detection job so it can receive data and run analysis (POST _ml/anomaly_detectors/{job_id}/_open).
+enabled: true
+tags: ["anomaly-detection", "management"]
+
+inputs:
+ - name: job_id
+ type: string
+ description: The anomaly detection job ID
+
+triggers:
+ - type: manual
+
+steps:
+ - name: open_job
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}/_open
+
+ - name: result
+ type: console
+ with:
+ message: |
+ Opened job {{ inputs.job_id }}:
+ {{ steps.open_job.output | json:2 }}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_preview_datafeed_with_latency.yaml b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_preview_datafeed_with_latency.yaml
new file mode 100644
index 0000000..151ac43
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_preview_datafeed_with_latency.yaml
@@ -0,0 +1,28 @@
+name: ad_preview_datafeed_with_latency
+description: >
+ Preview a datafeed's source payload and measure effective latency before tuning query_delay. Shows what data the
+ datafeed would see at query time, helping identify fields available for latency measurement.
+enabled: true
+tags: ["anomaly-detection", "troubleshooting"]
+
+inputs:
+ - name: datafeed_id
+ type: string
+ description: The datafeed ID to preview
+
+triggers:
+ - type: manual
+
+steps:
+ - name: preview_datafeed
+ type: elasticsearch.request
+ with:
+ method: GET
+ path: /_ml/datafeeds/{{ inputs.datafeed_id }}/_preview
+
+ - name: result
+ type: console
+ with:
+ message: |
+ Datafeed preview for {{ inputs.datafeed_id }}:
+ {{ steps.preview_datafeed.output | json:2 }}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_revert_model_snapshot.yaml b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_revert_model_snapshot.yaml
new file mode 100644
index 0000000..e17ab55
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_revert_model_snapshot.yaml
@@ -0,0 +1,55 @@
+name: ad_revert_model_snapshot
+description: >
+ Revert a job's model to a previous snapshot to 'unlearn' bad data. Stops the datafeed, closes the job, reverts to the
+ snapshot, reopens, and restarts the datafeed from the snapshot timestamp.
+enabled: true
+tags: ["anomaly-detection", "remediation"]
+
+inputs:
+ - name: job_id
+ type: string
+ description: The anomaly detection job ID
+ - name: snapshot_id
+ type: string
+ description: The snapshot ID to revert to
+
+triggers:
+ - type: manual
+
+steps:
+ - name: stop_datafeed
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/datafeeds/datafeed-{{ inputs.job_id }}/_stop
+
+ - name: close_job
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}/_close
+
+ - name: revert_snapshot
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}/model_snapshots/{{ inputs.snapshot_id }}/_revert
+
+ - name: open_job
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}/_open
+
+ - name: start_datafeed
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/datafeeds/datafeed-{{ inputs.job_id }}/_start
+
+ - name: result
+ type: console
+ with:
+ message: |
+ Reverted {{ inputs.job_id }} to snapshot {{ inputs.snapshot_id }}.
+ Job reopened and datafeed restarted from snapshot timestamp.
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_search_log_category_examples.yaml b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_search_log_category_examples.yaml
new file mode 100644
index 0000000..e3f7b96
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_search_log_category_examples.yaml
@@ -0,0 +1,54 @@
+name: ad_search_log_category_examples
+description: >
+ Search a source log index for messages matching specific ML category terms within a time window. Returns concrete log
+ examples belonging to a category during a specific period. Call twice — once for the anomaly window, once for a
+ baseline period (e.g. 24h prior) — to compare log content and identify what changed in the variable parts (IPs,
+ hostnames, error codes) that may reveal the root cause.
+enabled: true
+tags: ["anomaly-detection", "log-categorization", "evidence"]
+
+inputs:
+ - name: source_index
+ type: string
+ description: "The source log index (from ad_get_job_datafeed_config)"
+ default: "it_ops_logs"
+ - name: search_terms
+ type: string
+ description: "Category terms copied from ad_get_log_categories output"
+ - name: start_time
+ type: string
+ description: "Start of the time window (ISO 8601)"
+ - name: end_time
+ type: string
+ description: "End of the time window (ISO 8601)"
+
+triggers:
+ - type: manual
+
+steps:
+ - name: search_logs
+ type: elasticsearch.search
+ with:
+ index: "{{ inputs.source_index }}"
+ size: 20
+ sort: "@timestamp:desc"
+ query:
+ bool:
+ must:
+ - match:
+ message:
+ query: "{{ inputs.search_terms }}"
+ minimum_should_match: "70%"
+ filter:
+ - range:
+ "@timestamp":
+ gte: "{{ inputs.start_time }}"
+ lte: "{{ inputs.end_time }}"
+
+ - name: result
+ type: console
+ with:
+ message: |
+ Log examples matching category terms in {{ inputs.source_index }}
+ Time window: {{ inputs.start_time }} to {{ inputs.end_time }}
+ Results: {{ steps.search_logs.output.hits.hits | json:2 }}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_ts_ccs_diagnostics.yaml b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_ts_ccs_diagnostics.yaml
new file mode 100644
index 0000000..ed4b397
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_ts_ccs_diagnostics.yaml
@@ -0,0 +1,29 @@
+name: ad_ts_ccs_diagnostics
+description: >
+ Diagnose cross-cluster search (CCS) issues for datafeeds that query remote clusters. Checks remote cluster
+ connectivity, latency, and error rates. Helps identify per-cluster skew contributing to missing data.
+enabled: true
+tags: ["anomaly-detection", "troubleshooting", "ccs"]
+
+triggers:
+ - type: manual
+
+steps:
+ - name: check_remote_clusters
+ type: elasticsearch.request
+ with:
+ method: GET
+ path: /_remote/info
+
+ - name: check_cluster_health
+ type: elasticsearch.request
+ with:
+ method: GET
+ path: /_cluster/health
+
+ - name: result
+ type: console
+ with:
+ message: |
+ Remote clusters: {{ steps.check_remote_clusters.output | json:2 }}
+ Cluster health: {{ steps.check_cluster_health.output | json:2 }}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_update_datafeed_query_delay.yaml b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_update_datafeed_query_delay.yaml
new file mode 100644
index 0000000..fe0e3e8
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_update_datafeed_query_delay.yaml
@@ -0,0 +1,45 @@
+name: ad_update_datafeed_query_delay
+description: >
+ Update the query_delay setting on a datafeed. The datafeed must be stopped first. Larger query_delay captures more
+ late-arriving data but delays anomaly alerts. Recommended: set to P95 ingest latency + buffer.
+enabled: true
+tags: ["anomaly-detection", "remediation"]
+
+inputs:
+ - name: datafeed_id
+ type: string
+ description: The datafeed ID
+ - name: new_query_delay
+ type: string
+ description: "New query_delay value (e.g., '3m', '120s', '5m')"
+
+triggers:
+ - type: manual
+
+steps:
+ - name: stop_datafeed
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/datafeeds/{{ inputs.datafeed_id }}/_stop
+
+ - name: update_query_delay
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/datafeeds/{{ inputs.datafeed_id }}/_update
+ body:
+ query_delay: "{{ inputs.new_query_delay }}"
+
+ - name: start_datafeed
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/datafeeds/{{ inputs.datafeed_id }}/_start
+
+ - name: result
+ type: console
+ with:
+ message: |
+ Updated query_delay on {{ inputs.datafeed_id }} to {{ inputs.new_query_delay }}.
+ Datafeed restarted.
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_update_delayed_data_check_config.yaml b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_update_delayed_data_check_config.yaml
new file mode 100644
index 0000000..78e475b
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_update_delayed_data_check_config.yaml
@@ -0,0 +1,47 @@
+name: ad_update_delayed_data_check_config
+description: >
+ Update the delayed_data_check_config on a job to control how aggressively delayed data is detected. POST nests under
+ analysis_config.delayed_data_check_config. A data.parseJson step builds JSON so enabled is a boolean (not a quoted
+ string) and check_window is omitted when the input is empty or whitespace-only after trim.
+enabled: true
+tags: ["anomaly-detection", "remediation"]
+
+inputs:
+ - name: job_id
+ type: string
+ description: The anomaly detection job ID
+ - name: enabled
+ type: boolean
+ description: Whether to enable delayed data checks
+ - name: check_window
+ type: string
+ description: "Time window to check for delayed data (e.g., '2h'). Leave empty, omit, or whitespace-only to exclude check_window from the update body."
+
+triggers:
+ - type: manual
+
+steps:
+ - name: compose_payload
+ type: data.parseJson
+ source: |
+ {%- assign cw = inputs.check_window | default: "" | strip -%}
+ {%- if cw != "" -%}
+ {"analysis_config":{"delayed_data_check_config":{"enabled":{{ inputs.enabled }},"check_window":{{ cw | json }}}}
+ {%- else -%}
+ {"analysis_config":{"delayed_data_check_config":{"enabled":{{ inputs.enabled }}}}
+ {%- endif -%}
+ with: {}
+
+ - name: update_delayed_data
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}/_update
+ body: "${{ steps.compose_payload.output }}"
+
+ - name: result
+ type: console
+ with:
+ message: |
+ Updated delayed_data_check_config on {{ inputs.job_id }}:
+ {{ steps.update_delayed_data.output | json:2 }}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_update_model_memory_limit.yaml b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_update_model_memory_limit.yaml
new file mode 100644
index 0000000..0198d95
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_update_model_memory_limit.yaml
@@ -0,0 +1,60 @@
+name: ad_update_model_memory_limit
+description: >
+ Remediation workflow to change analysis_limits.model_memory_limit on an anomaly detector (for example after
+ ad_estimate_memory_requirement). Stops the datafeed, closes the job, POSTs /_ml/anomaly_detectors/{job_id}/_update with
+ the new limit, opens the job, and starts the datafeed again. Expects the default datafeed id datafeed-{job_id}. You
+ cannot decrease model_memory_limit below current model_bytes — clone the job to shrink.
+enabled: true
+tags: ["anomaly-detection", "remediation"]
+
+inputs:
+ - name: job_id
+ type: string
+ description: The anomaly detection job ID
+ - name: model_memory_limit
+ type: string
+ description: "New model_memory_limit value (e.g., '512mb', '1gb')"
+
+triggers:
+ - type: manual
+
+steps:
+ - name: stop_datafeed
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/datafeeds/datafeed-{{ inputs.job_id }}/_stop
+
+ - name: close_job
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}/_close
+
+ - name: update_model_memory_limit
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}/_update
+ body:
+ analysis_limits:
+ model_memory_limit: "{{ inputs.model_memory_limit }}"
+
+ - name: open_job
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}/_open
+
+ - name: start_datafeed
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/datafeeds/datafeed-{{ inputs.job_id }}/_start
+
+ - name: result
+ type: console
+ with:
+ message: |
+ Updated model_memory_limit on {{ inputs.job_id }} to {{ inputs.model_memory_limit }}.
+ Job reopened and datafeed restarted.
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_validate_job_spec.yaml b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_validate_job_spec.yaml
new file mode 100644
index 0000000..ad36027
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_validate_job_spec.yaml
@@ -0,0 +1,31 @@
+name: ad_validate_job_spec
+description: >
+ Validate an anomaly detection job configuration before creation. POSTs to the ML validate endpoint with the same JSON
+ document shape as PUT job creation. Requires manage_ml (cluster).
+enabled: true
+tags: ["anomaly-detection", "management"]
+
+inputs:
+ - name: job_body
+ type: string
+ description: >
+ Full job configuration as JSON text (same shape as job creation). Parsed with json_parse so the POST body is a
+ structured object for _validate, not a JSON string value.
+
+triggers:
+ - type: manual
+
+steps:
+ - name: validate_job
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_ml/anomaly_detectors/_validate
+ body: "${{ inputs.job_body | json_parse }}"
+
+ - name: result
+ type: console
+ with:
+ message: |
+ Validation result:
+ {{ steps.validate_job.output | json:2 }}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_validate_ml_tool_permissions.yaml b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_validate_ml_tool_permissions.yaml
new file mode 100644
index 0000000..a31104b
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_validate_ml_tool_permissions.yaml
@@ -0,0 +1,41 @@
+name: ad_validate_ml_tool_permissions
+description: >
+ Preflight check for core ML result and config indices via _has_privileges. Verifies read + view_index_metadata on
+ .ml-anomalies-*, .ml-config, .ml-annotations-*, and .ml-notifications-*. Does not check privileges on job-specific
+ source data indices — validate those separately before ad_rca_source_evidence, datafeed preview, or ad_wf_ts_field_cardinality.
+enabled: true
+tags: ["anomaly-detection", "diagnostics"]
+
+triggers:
+ - type: manual
+
+steps:
+ - name: check_security
+ type: elasticsearch.request
+ with:
+ method: GET
+ path: /_security/_authenticate
+
+ - name: check_ml_privileges
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_security/user/_has_privileges
+ body:
+ index:
+ - names: [".ml-anomalies-*"]
+ privileges: ["read", "view_index_metadata"]
+ - names: [".ml-config"]
+ privileges: ["read", "view_index_metadata"]
+ - names: [".ml-annotations-*"]
+ privileges: ["read", "view_index_metadata"]
+ - names: [".ml-notifications-*"]
+ privileges: ["read", "view_index_metadata"]
+
+ - name: result
+ type: console
+ with:
+ message: |
+ User: {{ steps.check_security.output.username }}
+ Roles: {{ steps.check_security.output.roles | json }}
+ ML index privileges: {{ steps.check_ml_privileges.output | json:2 }}
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_wf_troubleshoot_anomaly_score.yaml b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_wf_troubleshoot_anomaly_score.yaml
new file mode 100644
index 0000000..5915174
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_wf_troubleshoot_anomaly_score.yaml
@@ -0,0 +1,128 @@
+name: ad_wf_troubleshoot_anomaly_score
+description: >
+ Stored workflow for troubleshooting unexpectedly high or low anomaly scores. Implements a branching decision tree: (0)
+ gate checks (sufficient data, memory status, delayed data, UI aggregation), (1) UI display vs real score
+ (renormalization), (2) job configuration analysis (bucket_span, detector function, partition/influencer
+ fields, custom rules),
+ (3) model learning and data characteristics (insufficient history, high variance
+ penalty, model adaptation),
+ (4) score factor education (anomaly_score_explanation breakdown). Trigger: 'Why is my score low?', 'Expected anomaly
+ not detected', 'Score too high/low'.
+enabled: true
+tags: ["anomaly-detection", "troubleshooting", "scores"]
+
+inputs:
+ - name: job_id
+ type: string
+ description: The anomaly detection job ID
+ - name: record_timestamp
+ type: string
+ description: "Optional: ISO 8601 timestamp of the specific anomaly record to investigate"
+
+triggers:
+ - type: manual
+
+steps:
+ # Gate check 0a: Has the job processed enough data?
+ - name: get_job_stats
+ type: elasticsearch.request
+ with:
+ method: GET
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}/_stats
+
+ - name: gate_check
+ type: console
+ with:
+ message: |
+ GATE CHECKS for {{ inputs.job_id }}:
+ Job stats: {{ steps.get_job_stats.output | json:2 }}
+
+ CHECK 1 - Sufficient data:
+ Model needs >= 3 weeks for weekly seasonality, >= 2 full cycles.
+ Check data_counts.processed_record_count and earliest_record_timestamp.
+
+ CHECK 2 - Memory status:
+ If memory_status is soft_limit or hard_limit, the model is degraded.
+ Fix memory first before investigating scores.
+
+ CHECK 3 - Delayed data:
+ If the job has delayed data warnings, bucket scores may be based on
+ incomplete data. Check .ml-annotations-* for delayed data annotations.
+
+ # Step 1: Compare record_score vs initial_record_score
+ - name: get_anomaly_records
+ type: elasticsearch.search
+ with:
+ index: .ml-anomalies-*
+ size: 10
+ sort: "record_score:desc"
+ query:
+ bool:
+ filter:
+ - term:
+ job_id: "{{ inputs.job_id }}"
+ - term:
+ result_type: record
+
+ - name: score_comparison
+ type: console
+ with:
+ message: |
+ SCORE ANALYSIS for {{ inputs.job_id }}:
+ Top records: {{ steps.get_anomaly_records.output.hits.hits | json:2 }}
+
+ Compare initial_record_score vs record_score:
+ - If initial >> current: renormalization lowered the score after more
+ extreme anomalies appeared later. This is EXPECTED behavior.
+ - initial_record_score is the score at detection time.
+ - record_score is the current (renormalized) score.
+
+ # Step 2: Get job configuration for analysis
+ - name: get_job_config
+ type: elasticsearch.request
+ with:
+ method: GET
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}
+
+ - name: config_analysis
+ type: console
+ with:
+ message: |
+ JOB CONFIGURATION ANALYSIS:
+ Config: {{ steps.get_job_config.output | json:2 }}
+
+ Check these factors:
+ 1. bucket_span: Too large → dilutes anomalies. Too small → noisy.
+ 2. Detector function: mean vs high_mean vs low_mean affects directionality.
+ 3. Partition fields: High cardinality partitions split the model thin.
+ 4. custom_rules: May be suppressing valid anomalies.
+ 5. use_null: If false (default), missing entities produce no anomalies.
+
+ # Step 3-4: Score factor explanation
+ - name: score_education
+ type: console
+ with:
+ message: |
+ ANOMALY SCORE FACTORS (from anomaly_score_explanation):
+
+ 1. anomaly_length: How many consecutive buckets are anomalous.
+ Longer sequences → higher scores.
+
+ 2. single_bucket_impact: How extreme this single bucket is.
+ Driven by probability (lower p → higher impact).
+
+ 3. multi_bucket_impact: Positive (0-5) when anomaly spans multiple
+ buckets. Values >= 3 suggest genuine behavioral shift.
+
+ 4. anomaly_characteristics_impact: Nature of the anomaly (mean shift
+ vs variance change).
+
+ 5. high_variance_penalty: REDUCES score when the model's confidence
+ bounds are wide. Common early in model training or with noisy data.
+ Wide bounds → model is uncertain → anomalies appear less surprising.
+
+ 6. incomplete_bucket_penalty: REDUCES score when the bucket doesn't
+ have the expected amount of data (e.g., due to delayed data).
+
+ To see these factors, query the specific record from .ml-anomalies-*
+ and inspect the anomaly_score_explanation field.
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_wf_troubleshoot_memory_limit.yaml b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_wf_troubleshoot_memory_limit.yaml
new file mode 100644
index 0000000..f2bfa58
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_wf_troubleshoot_memory_limit.yaml
@@ -0,0 +1,165 @@
+name: ad_wf_troubleshoot_memory_limit
+description: >
+ Stored workflow for troubleshooting model_memory_limit issues (hard_limit/soft_limit). Implements a 7-step branching
+ decision tree: (1) identify memory status, (2) check downstream false alarms (hard_limit causing missing-doc
+ warnings), (3) analyze growth trend, (4) inspect model_size_stats thresholds, (5) compute principled estimate via
+ Estimate Model Memory API, (6) CCS-specific checks, (7) recommend action (increase limit / reduce data / restructure).
+ Trigger: 'My job hit memory limit', 'hard_limit', 'soft_limit'.
+enabled: true
+tags: ["anomaly-detection", "troubleshooting", "memory"]
+
+inputs:
+ - name: job_id
+ type: string
+ description: The anomaly detection job ID to troubleshoot
+
+triggers:
+ - type: manual
+
+steps:
+ # Step 1: Get current memory state
+ - name: get_stats
+ type: elasticsearch.request
+ with:
+ method: GET
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}/_stats
+
+ - name: get_config
+ type: elasticsearch.request
+ with:
+ method: GET
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}
+
+ - name: memory_status
+ type: console
+ with:
+ message: |
+ MEMORY STATUS for {{ inputs.job_id }}:
+ Stats: {{ steps.get_stats.output | json:2 }}
+
+ # Step 2: Check for downstream false alarms
+ - name: check_false_alarms
+ type: elasticsearch.search
+ with:
+ index: .ml-notifications-*
+ size: 10
+ sort: "timestamp:desc"
+ query:
+ bool:
+ filter:
+ - term:
+ job_id: "{{ inputs.job_id }}"
+ - terms:
+ level: ["warning", "error"]
+
+ - name: false_alarm_analysis
+ type: console
+ with:
+ message: |
+ DOWNSTREAM FALSE ALARM CHECK:
+ Recent warnings/errors: {{ steps.check_false_alarms.output.hits.hits | json:2 }}
+
+ If hard_limit AND you see "missing documents" warnings:
+ → These are SYMPTOMS of memory exhaustion, NOT ingest lag.
+ The model cannot track new entities, so events for unknown entities
+ are skipped, appearing as "missing documents".
+ Fix: Increase model_memory_limit first.
+
+ # Step 3: Analyze memory growth trend
+ - name: memory_trend
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /.ml-anomalies-*/_search
+ body:
+ size: 0
+ query:
+ bool:
+ filter:
+ - term:
+ job_id: "{{ inputs.job_id }}"
+ - term:
+ result_type: model_size_stats
+ aggs:
+ memory_over_time:
+ date_histogram:
+ field: timestamp
+ fixed_interval: 1d
+ aggs:
+ model_bytes:
+ max:
+ field: model_bytes
+ total_entities:
+ max:
+ field: total_by_field_count
+
+ - name: trend_analysis
+ type: console
+ with:
+ message: |
+ MEMORY GROWTH TREND:
+ {{ steps.memory_trend.output.aggregations | json:2 }}
+
+ Classify the trend:
+ - Stable plateau: Memory stabilized. Current limit may be appropriate
+ if status is ok, or barely insufficient if soft_limit.
+ - Linear growth: Entity cardinality is growing steadily (new hosts,
+ users, services). Will eventually hit limit.
+ - Exponential growth: Rapid cardinality explosion. Likely data or
+ config issue (unbounded by_field, logging explosion).
+
+ # Step 4-5: Inspect thresholds and estimate
+ - name: thresholds_and_estimate
+ type: console
+ with:
+ message: |
+ THRESHOLD INSPECTION:
+ From model_size_stats, check:
+ - total_by_field_count > 100K? → by_field cardinality too high
+ - total_partition_field_count > 10K? → partition explosion
+ - total_category_count > 10K? → categorization unbounded
+ Identify which field drives memory consumption.
+
+ PRINCIPLED ESTIMATION:
+ Run ad_estimate_memory_requirement workflow for this job.
+ It will:
+ 1. Sample cardinality from source data
+ 2. Call POST _ml/anomaly_detectors/_estimate_model_memory
+ 3. Compare estimate vs current limit vs actual usage
+
+ # Step 6-7: CCS checks and recommendations
+ - name: recommendations
+ type: console
+ with:
+ message: |
+ CCS CHECK:
+ If datafeed uses cross-cluster indices (remote_cluster:index pattern):
+ - Check remote cluster connectivity: GET _remote/info
+ - Note cardinality aggregation covers all clusters
+ - Re-run estimation if remote cluster recently added
+
+ RECOMMENDATIONS:
+
+ Branch A — Increase limit:
+ 1. Use the estimate from ad_estimate_memory_requirement, rounded up
+ 2. Fallback: max(estimate, peak_model_bytes * 1.3)
+ 3. Remediation sequence:
+ a) Stop datafeed: POST _ml/datafeeds/datafeed-{{ inputs.job_id }}/_stop
+ b) Close job: POST _ml/anomaly_detectors/{{ inputs.job_id }}/_close
+ c) Update: POST _ml/anomaly_detectors/{{ inputs.job_id }}/_update
+ { "analysis_limits": { "model_memory_limit": "NEW_VALUE" } }
+ d) Open job: POST _ml/anomaly_detectors/{{ inputs.job_id }}/_open
+ e) Start datafeed: POST _ml/datafeeds/datafeed-{{ inputs.job_id }}/_start
+ Use workflow: ad_update_model_memory_limit
+
+ Branch B — Reduce data:
+ - Filter datafeed query to exclude noisy/irrelevant data
+ - Reduce influencer count (each adds memory overhead)
+ - Partition into multiple focused jobs instead of one big job
+ - Exclude high-cardinality fields from analysis
+
+ Branch C — Extreme cases (multi-GB jobs):
+ - Architectural restructuring required
+ - Consider splitting by partition_field into separate jobs
+ - Use population analysis (over_field) instead of per-entity (by_field)
+ when possible — population uses less memory per entity
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_wf_troubleshoot_query_delay.yaml b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_wf_troubleshoot_query_delay.yaml
new file mode 100644
index 0000000..9ea1eda
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_wf_troubleshoot_query_delay.yaml
@@ -0,0 +1,131 @@
+name: ad_wf_troubleshoot_query_delay
+description: >
+ Stored workflow for troubleshooting missing documents and query_delay warnings. Implements a branching decision tree:
+ (1) checks for hard_limit categorization false alarms, (2) retrieves delayed data annotations, (3) checks bucket event
+ gaps, (4) measures ingest latency (two methods depending on available fields), (5) recommends query_delay value, (6)
+ suggests additional remediation. Trigger: 'My job reports missing documents', 'query_delay'.
+enabled: true
+tags: ["anomaly-detection", "troubleshooting", "query-delay"]
+
+inputs:
+ - name: job_id
+ type: string
+ description: The anomaly detection job ID to troubleshoot
+
+triggers:
+ - type: manual
+
+steps:
+ # Step 0: Get job/datafeed configuration
+ - name: get_config
+ type: elasticsearch.request
+ with:
+ method: GET
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}
+
+ - name: get_stats
+ type: elasticsearch.request
+ with:
+ method: GET
+ path: /_ml/anomaly_detectors/{{ inputs.job_id }}/_stats
+
+ # Step 1: Check for hard_limit categorization false alarm
+ # If job has categorization AND memory_status == hard_limit, the missing
+ # doc warning is a SYMPTOM of hard_limit, not ingest lag.
+ - name: log_hard_limit_check
+ type: console
+ with:
+ message: |
+ GATE CHECK: hard_limit categorization false alarm
+ Memory status: {{ steps.get_stats.output }}
+ If categorization_field_name is set AND memory_status == "hard_limit":
+ → STOP: Missing doc warning is caused by hard_limit, not ingest lag.
+ The per-partition categorizer cannot create new categories; events
+ are skipped, causing event_count mismatch.
+ Fix: Increase model_memory_limit first (use ad_update_model_memory_limit).
+
+ # Step 2: Check delayed data annotations
+ - name: check_delayed_annotations
+ type: elasticsearch.search
+ with:
+ index: .ml-annotations-*
+ size: 20
+ sort: "timestamp:desc"
+ query:
+ bool:
+ filter:
+ - term:
+ job_id: "{{ inputs.job_id }}"
+ - term:
+ type: annotation
+ - match:
+ annotation: "delayed data"
+
+ - name: log_delayed_annotations
+ type: console
+ with:
+ message: |
+ Delayed data annotations for {{ inputs.job_id }}:
+ Found {{ steps.check_delayed_annotations.output.hits.total.value }} delayed data annotations.
+ Recent annotations: {{ steps.check_delayed_annotations.output.hits.hits | json:2 }}
+
+ # Step 3: Check bucket event gaps
+ - name: check_bucket_gaps
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /.ml-anomalies-*/_search
+ body:
+ size: 0
+ query:
+ bool:
+ filter:
+ - term:
+ job_id: "{{ inputs.job_id }}"
+ - term:
+ result_type: bucket
+ aggs:
+ zero_count_buckets:
+ filter:
+ term:
+ event_count: 0
+ low_count_buckets:
+ filter:
+ range:
+ event_count:
+ gt: 0
+ lte: 5
+
+ - name: log_bucket_gaps
+ type: console
+ with:
+ message: |
+ Bucket event gap analysis for {{ inputs.job_id }}:
+ {{ steps.check_bucket_gaps.output.aggregations | json:2 }}
+
+ # Step 4-6: Recommendations
+ - name: recommendations
+ type: console
+ with:
+ message: |
+ RECOMMENDATIONS for {{ inputs.job_id }}:
+
+ 1. MEASURE INGEST LATENCY:
+ - If event.ingested field exists: compare event.ingested vs @timestamp
+ to get exact latency distribution (P50, P95, P99)
+ - If not: use date_histogram comparison (recent bucket doc count vs
+ same bucket queried later) to estimate stabilization window
+
+ 2. SET QUERY_DELAY:
+ - Recommended: P95 ingest latency + 30s buffer
+ - Trade-off: larger query_delay = slower anomaly alerts
+ - To update: stop datafeed → update query_delay → restart datafeed
+ - Use workflow: ad_update_datafeed_query_delay
+
+ 3. ADDITIONAL REMEDIATION:
+ a) Add ingest timestamp via ingest pipeline for future diagnosis:
+ PUT _ingest/pipeline/add-ingest-ts
+ with a "set" processor that copies the _ingest.timestamp into "event.ingested"
+ b) Consider using ingest timestamp as the job's time_field
+ c) If catastrophically late data: revert model snapshot + backfill
+ d) Note: missed documents warning may persist 24h after fix
diff --git a/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_wf_ts_field_cardinality.yaml b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_wf_ts_field_cardinality.yaml
new file mode 100644
index 0000000..5f93f12
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/kibana/workflows/ad_wf_ts_field_cardinality.yaml
@@ -0,0 +1,50 @@
+name: ad_wf_ts_field_cardinality
+description: >
+ Estimate cardinality of a split field (by_field, over_field, partition_field) in SOURCE data via ES|QL
+ POST /_query. The column for COUNT_DISTINCT is interpolated into the query text (not a ? parameter - those bind only
+ literals). Pass split_field_esql as a valid ES|QL column reference (for example service.keyword or `host.name.keyword`)
+ from ad_get_job_datafeed_config. Compare distinct_count with total_*_count from ad_ts_model_memory_health; if source
+ cardinality is much larger, entities may be dropped. For CCS, run per cluster and sum. Prefer ad_estimate_memory_requirement
+ for full sizing; this workflow answers how many distinct values the field has in the window.
+enabled: true
+tags: ["anomaly-detection", "diagnostics"]
+
+inputs:
+ - name: source_index
+ type: string
+ description: >
+ Source index name or LIKE pattern (* wildcard). From the job datafeed config (same as ES|QL FROM * METADATA _index filter).
+ - name: split_field_esql
+ type: string
+ description: >
+ Exact ES|QL column expression for COUNT_DISTINCT (not quoted as a string). Examples: service.keyword, `host.name.keyword`.
+ Must match a field on matched documents; use only values taken from job analysis config to avoid query injection.
+ - name: start_time
+ type: string
+ description: Start of time range in ISO 8601 format
+ - name: end_time
+ type: string
+ description: End of time range in ISO 8601 format
+
+triggers:
+ - type: manual
+
+steps:
+ - name: cardinality_esql
+ type: elasticsearch.request
+ with:
+ method: POST
+ path: /_query
+ body:
+ query: "FROM * METADATA _index | WHERE _index LIKE ?source_index AND @timestamp >= ?start_time AND @timestamp <= ?end_time | STATS distinct_count = COUNT_DISTINCT({{ inputs.split_field_esql }}) | LIMIT 1"
+ params:
+ source_index: "{{ inputs.source_index }}"
+ start_time: "{{ inputs.start_time }}"
+ end_time: "{{ inputs.end_time }}"
+
+ - name: result
+ type: console
+ with:
+ message: |
+ Split-field cardinality ({{ inputs.split_field_esql }}) on indices matching {{ inputs.source_index }} ({{ inputs.start_time }}-{{ inputs.end_time }}):
+ {{ steps.cardinality_esql.output | json:2 }}
diff --git a/skills/kibana/kibana-anomaly-detection/references/observability-anomaly-expert.md b/skills/kibana/kibana-anomaly-detection/references/observability-anomaly-expert.md
new file mode 100644
index 0000000..bb69981
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/observability-anomaly-expert.md
@@ -0,0 +1,104 @@
+# Observability / SRE framing — Elastic ML anomaly detection
+
+**Role:** Treat Elasticsearch ML anomaly detection as a reliability signal for SRE and platform work: degradation,
+incident scope, and capacity decisions. Combine the **Investigate**, **Explain**, and **Troubleshoot** modes of the
+parent skill, biased toward reliability interpretation.
+
+---
+
+## Reliability-first interpretation
+
+Interpret anomalies through three reliability lenses:
+
+1. **Incident detection** — is this active degradation? What is the scope?
+2. **Change attribution** — tie signal to deployments, config changes, dependencies when possible.
+3. **Capacity signals** — separate acute incidents from resource-exhaustion trajectories.
+
+## Signal → reliability mapping
+
+| Anomaly pattern | Reliability interpretation | Action |
+| -------------------------------------------------------------- | -------------------------------------------------- | ------------------------------------ |
+| Latency spike + error rate spike (same service, same time) | Service degradation in progress | Incident response |
+| Throughput drop (`actual << typical` with `count`/`low_count`) | Service unavailable or upstream dependency failure | Check dependencies, circuit breakers |
+| Cross-service entity anomalies with temporal chain | Cascading failure / blast propagation | Identify blast radius, isolate |
+| Memory/CPU creep (`multi_bucket_impact ≥ 3`) | Resource exhaustion trajectory | Capacity intervention before OOM |
+| Anomaly onset matches deployment timestamp | Deployment regression | Rollback candidate |
+| Single service anomaly, no related job co-firing | Isolated issue, contained | Service-level investigation |
+| Anomaly during known maintenance window | Expected — suppress via calendar event | `ad_create_calendar_event` |
+
+## SRE investigation protocol
+
+### Phase 1 — Incident scoping (Investigate mode)
+
+1. `ad_get_available_metadata` — identify observability jobs (latency, error rate, throughput, saturation, request
+ count).
+2. `ad_query_anomaly_timeline` (`job_id_pattern='*'`) — establish incident start time and breadth.
+3. `ad_rca_multi_job_entities` (`min_job_count=2`) — co-firing metrics on the same entity = the degraded service.
+4. `ad_rca_blast_radius` — scope: which downstream services are affected.
+
+### Phase 2 — Root cause attribution (Investigate mode)
+
+1. `ad_discover_jobs_by_datafeed_index` — find all jobs monitoring the same infrastructure layer.
+2. `ad_rca_cross_job_entity_match` — confirm which services are actively co-firing.
+3. `ad_rca_correlation` sorted by timestamp — leading metric (first anomaly) = root cause; lagging = symptoms.
+4. `ad_rca_detector_fingerprint` — characterize failure type (latency? saturation? error rate? throughput drop?).
+
+### Phase 3 — Evidence and context (Investigate mode)
+
+1. `ad_get_job_datafeed_config` → source index → `ad_rca_source_evidence` — actual metric values and dimensions.
+2. `ad_query_influencers` — which specific service instances, pods, or hosts are contributing.
+3. `ad_rca_entity_profile` — full behavioral history for the suspect service/host.
+
+### Phase 4 — Deployment regression check (Explain mode)
+
+When incident onset aligns with a recent deployment:
+
+1. `ad_rca_score_reassessment` — confirm whether a score drop reflects renormalization instead of real recovery.
+2. `ad_get_model_plot` — confirm the anomaly sits outside expected bounds instead of being a model artifact.
+3. `ad_rca_source_evidence` — compare metric values before and after the deployment timestamp.
+
+### Phase 5 — Capacity planning (Explain + Troubleshoot modes)
+
+For sustained `multi_bucket_impact ≥ 3` anomalies that look like trajectories instead of spikes:
+
+1. `ad_ts_model_memory_health` — confirm ML memory pressure is not degrading detections.
+2. `ad_query_anomaly_records` filtered to `multi_bucket_impact ≥ 3` — extract resource saturation trends.
+3. `ad_estimate_memory_requirement` — size memory for expanded infrastructure.
+
+### Phase 6 — Maintenance suppression (Troubleshoot mode)
+
+For planned deployments or maintenance windows:
+
+1. `ad_create_calendar_event` — suppress false positives, reduce alert fatigue, protect model health.
+
+## Reliability-specific rules
+
+- **`multi_bucket_impact ≥ 3`** is the primary capacity signal: sustained shifts indicate trajectory. These need
+ capacity planning, not just incident response.
+- **`actual << typical` with throughput detectors** = service unavailability. Treat as SEV-1 until proven otherwise.
+- **Temporal ordering matters**: in cascading failures, the first anomaly timestamp points to root cause, not the
+ highest score.
+- **`initial_record_score >> record_score`**: renormalization — the score dropped because a worse event occurred later.
+ Do not interpret as "resolved." Use Explain mode to communicate this to stakeholders.
+- **All jobs firing simultaneously**: shared infrastructure layer (database, message bus, shared network path).
+ Investigate shared dependencies first.
+- **`ad_validate_ml_tool_permissions`**: run as a preflight when tool calls fail unexpectedly.
+
+## Job health before trusting signals
+
+- `ad_ts_model_memory_health` — a job at `hard_limit` stops learning new entities (new pods/services), risking missed
+ anomalies for those entities.
+- `ad_ts_delayed_data_annotations` — delayed data delays alerts. Raise `query_delay` toward P95 ingest latency + buffer
+ when the pipeline is slow; otherwise expect missed real-time detection.
+- `ad_create_calendar_event` — add maintenance windows to suppress false positives during planned deployments.
+
+## Escalation decision framework
+
+| Signal | SRE action |
+| ------------------------------------------------------- | ------------------------------------------- |
+| Multi-job co-fire + blast radius > 1 service | Declare incident, page on-call |
+| Leading metric identified + deployment timestamp match | Rollback candidate — page owning team |
+| Sustained `multi_bucket_impact ≥ 3` + resource detector | Capacity review, no immediate incident |
+| Single-job anomaly + no downstream impact | Service-level investigation, no incident |
+| Anomaly during known maintenance | Add calendar event, dismiss |
+| Score drop only (renormalization) | Use Explain mode to communicate — no action |
diff --git a/skills/kibana/kibana-anomaly-detection/references/permissions-matrix.md b/skills/kibana/kibana-anomaly-detection/references/permissions-matrix.md
new file mode 100644
index 0000000..4a46869
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/permissions-matrix.md
@@ -0,0 +1,144 @@
+# Permissions Matrix
+
+Maps each tool to the Elasticsearch and Kibana privileges it requires.
+
+Run `ad_validate_ml_tool_permissions` as a preflight check on core `.ml-*` indices (see workflow description). Confirm
+**source data index** privileges separately before previews, `ad_wf_ts_field_cardinality`, or `ad_rca_source_evidence`.
+
+---
+
+## Required Index Privileges
+
+| Index | Privilege | Purpose |
+| --------------------- | --------------------- | ----------------------------------------------------------------------------- |
+| `.ml-anomalies-*` | `read` | Query anomaly records, buckets, influencers, category definitions |
+| `.ml-anomalies-*` | `view_index_metadata` | ESQL `FROM` and metadata access on anomaly indices |
+| `.ml-config` | `read` | Query job and datafeed configurations |
+| `.ml-config` | `view_index_metadata` | ESQL `FROM` against the config index |
+| `.ml-annotations-*` | `read` | Query delayed data annotations |
+| `.ml-annotations-*` | `view_index_metadata` | ESQL `FROM` and metadata access on annotation indices |
+| `.ml-notifications-*` | `read` | Query job messages and notifications |
+| `.ml-notifications-*` | `view_index_metadata` | ESQL `FROM` and metadata access on notification indices |
+| Source data indices | `read` | `ad_rca_source_evidence`, `ad_search_log_category_examples`, cardinality aggs |
+| Source data indices | `view_index_metadata` | ESQL `FROM` with `METADATA _index` on source data |
+
+---
+
+## Tool → Permission Matrix
+
+### ES|QL Tools (Read-only)
+
+| Tool | `.ml-anomalies-*` | `.ml-config` | `.ml-annotations-*` | Source indices |
+| --------------------------------- | ----------------- | --------------- | ------------------- | --------------- |
+| `ad_get_available_metadata` | — | read + metadata | — | — |
+| `ad_get_jobs` | — | read + metadata | — | — |
+| `ad_discover_related_jobs` | — | read + metadata | — | — |
+| `ad_query_anomaly_records` | read + metadata | — | — | — |
+| `ad_query_anomaly_timeline` | read + metadata | — | — | — |
+| `ad_query_influencers` | read + metadata | — | — | — |
+| `ad_rca_multi_job_entities` | read + metadata | — | — | — |
+| `ad_rca_cross_job_entity_match` | read + metadata | — | — | — |
+| `ad_rca_detector_fingerprint` | read + metadata | — | — | — |
+| `ad_rca_correlation` | read + metadata | — | — | — |
+| `ad_rca_blast_radius` | read + metadata | — | — | — |
+| `ad_rca_entity_profile` | read + metadata | — | — | — |
+| `ad_rca_source_evidence` | — | — | — | read + metadata |
+| `ad_rca_score_reassessment` | read + metadata | — | — | — |
+| `ad_get_categories` | read + metadata | — | — | — |
+| `ad_search_log_category_examples` | — | — | — | read + metadata |
+| `ad_get_job_messages` | — | — | — | — |
+| `ad_get_model_snapshots` | read + metadata | — | — | — |
+| `ad_get_model_plot` | read + metadata | — | — | — |
+| `ad_get_forecast_results` | read + metadata | — | — | — |
+| `ad_ts_delayed_data_annotations` | — | — | read + metadata | — |
+| `ad_ts_bucket_event_gaps` | read + metadata | — | — | — |
+| `ad_ts_ingest_latency_estimate` | — | — | — | read |
+| `ad_ts_model_memory_health` | read + metadata | — | — | — |
+
+### Workflow Tools
+
+| Tool | ML API privilege | Source indices | Notes |
+| ------------------------------------- | -------------------------- | -------------- | --------------------------------------------------------------- |
+| `ad_get_job_datafeed_config` | `monitor_ml` | — | Reads job config and stats |
+| `ad_discover_jobs_by_datafeed_index` | `monitor_ml` | — | Reads job config + .ml-config search |
+| `ad_manage_datafeed` | `manage_ml` | — | Start/stop requires write privilege |
+| `ad_preview_datafeed_with_latency` | `monitor_ml` | read | Preview requires source index read |
+| `ad_update_datafeed_query_delay` | `manage_ml` | — | Write operation |
+| `ad_update_delayed_data_check_config` | `manage_ml` | — | Write operation |
+| `ad_estimate_memory_requirement` | `monitor_ml` | read | Cardinality aggs on source indices |
+| `ad_wf_ts_field_cardinality` | `monitor_ml` | read | POST /\_query COUNT_DISTINCT on source split field |
+| `ad_update_model_memory_limit` | `manage_ml` | — | Write operation; job must be closed |
+| `ad_revert_model_snapshot` | `manage_ml` | — | Write operation; job must be closed |
+| `ad_get_calendar_events` | `monitor_ml` | — | — |
+| `ad_create_calendar_event` | `manage_ml` | — | Write operation |
+| `ad_create_job` | `manage_ml` | — | Write operation |
+| `ad_validate_ml_tool_permissions` | `monitor` (cluster) | — | Uses `_security` API |
+| `ad_ts_ccs_diagnostics` | `monitor` (cluster) | — | Uses `_remote/info`, `_cluster/health` |
+| `ad_wf_troubleshoot_anomaly_score` | `monitor_ml` | — | Read-only workflow |
+| `ad_wf_troubleshoot_memory_limit` | `monitor_ml` + `manage_ml` | read | Includes estimate (read) and optional update (write) |
+| `ad_wf_troubleshoot_query_delay` | `monitor_ml` + `manage_ml` | read | Includes latency measurement (read) and optional update (write) |
+
+---
+
+## Privilege Definitions
+
+| Privilege | Scope | What it allows |
+| ----------------------------- | ------- | -------------------------------------------------------------------------------------- | ---------------------- |
+| `read` (index) | Index | Search, GET, ES | QL FROM |
+| `view_index_metadata` (index) | Index | Required for ES | QL FROM and field caps |
+| `monitor_ml` (cluster) | Cluster | GET job configs, stats, snapshots, calendar events |
+| `manage_ml` (cluster) | Cluster | Create/update/delete jobs, datafeeds, calendars; start/stop datafeed; revert snapshots |
+| `monitor` (cluster) | Cluster | Cluster health, remote info, security authenticate |
+
+---
+
+## Minimum Role for Read-only Investigation
+
+```json
+{
+ "cluster": ["monitor_ml"],
+ "indices": [
+ {
+ "names": [".ml-anomalies-*", ".ml-config", ".ml-annotations-*", ".ml-notifications-*"],
+ "privileges": ["read", "view_index_metadata"]
+ },
+ {
+ "names": [""],
+ "privileges": ["read", "view_index_metadata"]
+ }
+ ]
+}
+```
+
+## Minimum Role for Full Remediation
+
+```json
+{
+ "cluster": ["monitor_ml", "manage_ml", "monitor"],
+ "indices": [
+ {
+ "names": [".ml-anomalies-*", ".ml-config", ".ml-annotations-*", ".ml-notifications-*"],
+ "privileges": ["read", "view_index_metadata"]
+ },
+ {
+ "names": [""],
+ "privileges": ["read", "view_index_metadata"]
+ }
+ ]
+}
+```
+
+---
+
+## Troubleshooting Permission Errors
+
+| Symptom | Likely missing privilege |
+| ---------------------------------------------- | ---------------------------------------------- | ------------------------------------------ |
+| ES | QL FROM `.ml-anomalies-*` returns no results | `view_index_metadata` on `.ml-anomalies-*` |
+| `ad_rca_source_evidence` returns empty | `read` on source data indices |
+| Workflow tools return 403 | `monitor_ml` or `manage_ml` cluster privilege |
+| `ad_validate_ml_tool_permissions` fails | `monitor` cluster privilege |
+| `ad_ts_delayed_data_annotations` returns empty | `read` on `.ml-annotations-*` |
+| Job config missing from `ad_get_jobs` | `read` + `view_index_metadata` on `.ml-config` |
+
+Run `ad_validate_ml_tool_permissions` to get a definitive list of which specific privileges are missing.
diff --git a/skills/kibana/kibana-anomaly-detection/references/protocols/investigation.md b/skills/kibana/kibana-anomaly-detection/references/protocols/investigation.md
new file mode 100644
index 0000000..0a29b5d
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/protocols/investigation.md
@@ -0,0 +1,152 @@
+# Investigation Protocol (14 Steps)
+
+Canonical workflow for root cause analysis of Elastic ML anomaly detection events.
+
+> For a worked example, see [../worked-example.md](../worked-example.md).
+
+---
+
+## When to Use This Protocol
+
+- Starting from a single alert and need to determine the root cause
+- Multiple jobs are co-firing and you need to find the common denominator
+- Asked "what broke?", "which entity caused this?", or "why is service X slow?"
+
+For **score explanation questions** (why is my score low/high?), see [../score-reference.md](../score-reference.md)
+instead.
+
+---
+
+## Three-Layer Job Discovery
+
+Before beginning analysis, identify all related jobs using these signals in priority order:
+
+1. **Shared datafeed index patterns** (strongest) — Jobs reading from the same source indices monitor the same
+ underlying system.
+2. **Shared entity field names** (config signal) — Jobs that split by the same field names analyze the same entity
+ dimensions.
+3. **Shared entity values in results** (active incident) — During an active incident, find all jobs where a specific
+ entity is currently co-firing.
+
+---
+
+## The 14 Steps
+
+### Phase 1: Discovery
+
+**Step 1 — Discover** Call `ad_get_available_metadata` to learn available jobs, fields, and functions. Always start here
+when jobs are unknown.
+
+**Step 2 — Find related jobs** Use `ad_discover_jobs_by_datafeed_index` with the job of interest — it retrieves that
+job's `datafeed_config.indices`, then finds all other jobs sharing the same source index. Also use
+`ad_discover_related_jobs` to find jobs sharing entity field names (partition/by/over). Fallback: compare
+`datafeed_config.indices` manually via `ad_get_jobs`.
+
+**Step 3 — Scope** Use `ad_query_anomaly_timeline` with `job_id_pattern` set to the related job group (e.g.,
+`rcaeval-*`) or `*` for all jobs. Identify the incident time window and count of affected jobs. Cross-job composite
+scores reveal coordinated events.
+
+---
+
+### Phase 2: Entity Attribution
+
+**Step 4 — Expand from alert** Extract entity values from the alert (`partition_field_value`, `by_field_value`,
+`over_field_value`). Use `ad_rca_cross_job_entity_match` to find all related jobs with anomalies for that entity. Note
+`first_anomaly` per job for chronology reconstruction.
+
+**Step 5 — Multi-job entities** Use `ad_rca_multi_job_entities` with `min_job_count=2`. Entities anomalous in 2+ jobs
+simultaneously are the strongest root cause signal — they are prime suspects. Single-job entities are likely downstream
+victims.
+
+> Resource faults (CPU, memory, disk) affect multiple metrics → multi-job. Network faults (packet loss) affect latency
+> but not resource metrics → single-job.
+
+**Step 6 — Fingerprint** Use `ad_rca_detector_fingerprint` with the related job group as `job_id_pattern`. Understand
+which system aspects are anomalous: CPU? Latency? Error rate? Memory? The combination of anomalous detectors
+characterizes the fault type.
+
+---
+
+### Phase 3: Deep Analysis
+
+**Step 7 — Drill down per job** Use `ad_query_anomaly_records` with an exact `job_id_pattern` to examine a specific
+job's anomalies in detail, without cross-job noise.
+
+**Step 8 — Attribute** Use `ad_query_influencers` with the related job group as `job_id_pattern` and a low `min_score`
+(25) for shared influencer discovery. Filter for `job_count > 1` to surface entities that are influencers in multiple
+co-firing jobs — the common denominator.
+
+**Step 9 — Profile** Use `ad_rca_entity_profile` to build a complete dossier on the suspect entity: all anomalies across
+all jobs and field types, sorted by timestamp.
+
+**Step 10 — Characterize** Examine `multi_bucket_impact` in results:
+
+- `≥ 3` → sustained behavioral shift (system change), not a transient spike
+- `0–2` → isolated event (one-off anomaly)
+
+---
+
+### Phase 4: Root Cause Confirmation
+
+**Step 11 — Cascade** Use `ad_rca_correlation` sorted by timestamp. The job with the **earliest anomaly** for the
+suspect entity points toward the root cause. Reconstruct chronology: which metric became anomalous first?
+
+**Step 12 — Evidence** Get the source index from `ad_get_job_datafeed_config`, then call `ad_rca_source_evidence` to
+retrieve raw source documents. This shows the actual values that triggered the anomaly at the point of ingestion.
+
+**Step 13 — Log categories** _(only when `by_field_name == "mlcategory"`)_ For log categorization jobs:
+
+1. `ad_get_categories` → find the category matching the anomaly's `by_field_value` (category ID). Examine its terms,
+ regex, and examples.
+2. `ad_search_log_category_examples` twice — once for a **baseline window** (24h before anomaly), once for the **anomaly
+ window**.
+3. Compare: look for changed field values in the variable parts of the log structure (IPs, hostnames, error codes,
+ paths, credentials).
+4. Cross-reference changed entities with influencers from other related jobs to confirm root cause.
+
+---
+
+### Phase 5: Synthesis
+
+**Step 14 — Synthesize** Present findings as a structured RCA report:
+
+| Section | Content |
+| ------------------------ | ------------------------------------------------------------- |
+| **Root cause entity** | The entity (host, service, user) responsible |
+| **Affected systems** | Which jobs/metrics were impacted |
+| **Temporal progression** | Which metric became anomalous first (from Step 11) |
+| **Fault type** | Resource (CPU/memory/disk) / Network / Application / Pipeline |
+| **Severity** | `record_score` range, `multi_bucket_impact`, duration |
+| **Recommended actions** | Remediation steps |
+
+---
+
+## Quick Reference: Tool → Step Mapping
+
+| Tool | Step |
+| ------------------------------------ | ---- |
+| `ad_get_available_metadata` | 1 |
+| `ad_discover_jobs_by_datafeed_index` | 2 |
+| `ad_discover_related_jobs` | 2 |
+| `ad_query_anomaly_timeline` | 3 |
+| `ad_rca_cross_job_entity_match` | 4 |
+| `ad_rca_multi_job_entities` | 5 |
+| `ad_rca_detector_fingerprint` | 6 |
+| `ad_query_anomaly_records` | 7 |
+| `ad_query_influencers` | 8 |
+| `ad_rca_entity_profile` | 9 |
+| `ad_rca_correlation` | 11 |
+| `ad_get_job_datafeed_config` | 12 |
+| `ad_rca_source_evidence` | 12 |
+| `ad_get_categories` | 13 |
+| `ad_search_log_category_examples` | 13 |
+
+---
+
+## Key Decision Rules
+
+- **Low scores across many jobs** > one high score — composite cross-job signal often indicates systemic root cause.
+- **`actual << typical` with count/low_count** → absence/outage, not just a numerically low value.
+- **Entities in 2+ jobs** → prime suspects (resource fault or systemic failure).
+- **Entities in only 1 job** → likely downstream victims or surface-level effects.
+- **`first_anomaly` chronology** → the earliest metric to become anomalous is closest to the root cause.
diff --git a/skills/kibana/kibana-anomaly-detection/references/score-reference.md b/skills/kibana/kibana-anomaly-detection/references/score-reference.md
new file mode 100644
index 0000000..c409573
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/score-reference.md
@@ -0,0 +1,120 @@
+# Anomaly Score Reference
+
+Canonical definitions for all score and impact fields in Elastic ML anomaly detection.
+
+---
+
+## Score Types
+
+| Field | Scope | Range | Description |
+| ----------------------- | --------------------- | ----- | -------------------------------------------------------------------------------------- |
+| `record_score` | Single anomaly record | 0–100 | Current normalized severity. May change over time as the model sees more extreme data. |
+| `initial_record_score` | Single anomaly record | 0–100 | Score at detection time — never changes. Use for alerting on fresh anomalies. |
+| `anomaly_score` | Bucket (time window) | 0–100 | Aggregate severity across all detectors in a bucket. |
+| `initial_anomaly_score` | Bucket | 0–100 | Bucket score at detection time — never changes. |
+| `influencer_score` | Entity × bucket | 0–100 | How anomalous a specific entity (host, user, service) is within that bucket. |
+
+---
+
+## `record_score` Severity Bands
+
+| Band | Range | Interpretation |
+| ------------- | ----- | ---------------------------------------------------------- |
+| Critical | > 75 | High-confidence anomaly; warrants immediate investigation |
+| Warning | 50–75 | Notable deviation; triage and correlate with other signals |
+| Minor | 25–50 | Potentially interesting; aggregate with cross-job signals |
+| Informational | < 25 | Weak signal; useful for context, not standalone action |
+
+> **Cross-job composite signal**: Low scores (25–50) across many jobs simultaneously are often more significant than a
+> single high score. Five jobs each scoring 30 = composite signal 150, pointing to a systemic root cause.
+
+---
+
+## `multi_bucket_impact`
+
+Scale from -5 to +5 indicating whether the anomaly spans multiple consecutive time buckets.
+
+| Value | Meaning |
+| -------- | ---------------------------------------------------- |
+| 0 | One-off event, no sustained pattern |
+| 1–2 | Mild persistence across a few buckets |
+| ≥ 3 | Genuine behavioral shift — not a transient spike |
+| Negative | Anomaly is suppressed by surrounding normal behavior |
+
+Values ≥ 3 strongly suggest a real system change (e.g., a resource exhaustion event that persists) rather than a
+momentary blip.
+
+---
+
+## `initial_record_score` vs `record_score`
+
+Elasticsearch continuously renormalizes scores relative to the most extreme anomaly ever seen by the job. A score of 90
+today may become 60 if a more extreme event appears later — by design, so the "worst ever" event always scores near 100.
+
+**When to use each:**
+
+| Use case | Field |
+| ------------------------------------------------------------ | ------------------------------------------------------------------------------------ |
+| Alerting on newly detected anomalies | `initial_record_score` — captures severity at detection time |
+| Ranking historical anomalies by current importance | `record_score` — reflects how bad this was relative to all history |
+| Detecting renormalization (model calibrated away an anomaly) | Compare: if `initial_record_score >> record_score`, the model saw worse events later |
+
+**Quantify drift:** `score_drift = initial_record_score - record_score`
+
+- Large positive drift = renormalized away (model calibrated)
+- Small drift = score is stable and genuine
+
+---
+
+## `anomaly_score_explanation` Components
+
+When available, this field explains the factors that contributed to the final score.
+
+| Component | Effect on score | What it means |
+| -------------------------------- | --------------- | ------------------------------------------------------------------- |
+| `anomaly_length` | ↑ increases | More consecutive anomalous buckets — sustained deviation |
+| `single_bucket_impact` | ↑ increases | Lower statistical probability → more surprising → higher impact |
+| `multi_bucket_impact` | ↑ increases | Contribution from sustained pattern across multiple buckets |
+| `anomaly_characteristics_impact` | ↑ increases | Mean shift (value moved) vs. variance change (volatility increased) |
+| `high_variance_penalty` | ↓ decreases | Historically noisy data; wide confidence bounds absorb the spike |
+| `incomplete_bucket_penalty` | ↓ decreases | Bucket has less data than expected (ingest lag, sparse events) |
+
+---
+
+## Absence Anomalies
+
+When `actual << typical` with `count`, `low_count`, `low_mean`, or `low_sum` functions, a low or zero value indicates a
+real-world absence — not just a numerically low observation:
+
+- Zero `count` when traffic is normally constant → pipeline stopped, service unavailable
+- `low_mean(response_time)` → requests completing too fast (cache hit storm, bypassed processing)
+- Very low `sum(bytes_sent)` → network partition or data source failure
+
+**Key insight:** A `record_score` of 80 with `actual = 0` and `typical = 5000` is an outage signal, not just a low
+number.
+
+---
+
+## Why a Score Is Unexpectedly Low
+
+1. **`high_variance_penalty`** — Metric is historically noisy; wide model bounds absorb the spike.
+2. **Renormalization** — A more extreme anomaly appeared later, pushing this score down.
+3. **Insufficient training** — Model needs ≥ 3 weeks for weekly seasonality, ≥ 2 full cycles for any period.
+4. **`bucket_span` too large** — Long span smooths short-duration spikes; use smaller span for high-frequency detection.
+5. **Detector function mismatch** — `mean` vs `high_mean`, `count` vs `high_count`. Wrong function = missed direction.
+6. **`incomplete_bucket_penalty`** — Ingest latency or sparse events reduced bucket data volume.
+7. **`custom_rules`** — A detector filter may be suppressing or conditioning the anomaly.
+
+## Why a Score Is Unexpectedly High
+
+1. **Insufficient training history** — Early training: moderate deviations flag as extreme.
+2. **High-cardinality split** — Too few data points per entity per bucket → unreliable probabilities.
+3. **`use_null: true`** — Missing entities produce "null" anomalies that may not be operationally meaningful.
+
+---
+
+## See Also
+
+- [anomaly-detection-functions.md](anomaly-detection-functions.md) — Function selection guide
+- [protocols/investigation.md](protocols/investigation.md) — 14-step investigation workflow
+- [worked-example.md](worked-example.md) — End-to-end investigation walkthrough
diff --git a/skills/kibana/kibana-anomaly-detection/references/security-anomaly-expert.md b/skills/kibana/kibana-anomaly-detection/references/security-anomaly-expert.md
new file mode 100644
index 0000000..4e0f75f
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/security-anomaly-expert.md
@@ -0,0 +1,103 @@
+# Security framing — Elastic ML anomaly detection
+
+**Role:** Treat Elasticsearch ML anomaly detection as a behavioral threat-detection surface. Assume unusual behavior is
+suspicious until benign intent is proven. Combine the **Investigate**, **Explain**, and **Troubleshoot** modes of the
+parent skill, biased toward attack-first interpretation.
+
+---
+
+## Threat-first interpretation
+
+Treat operational monitoring as benign-first; treat security anomalies as attack-first. Then:
+
+1. Map behavioral deviations to known attack patterns.
+2. Reconstruct attacker chains from cross-job signals.
+3. Separate attacker behavior from benign operational noise.
+4. Classify threats with MITRE ATT&CK context.
+
+## Signal mapping
+
+| Anomaly pattern | Threat hypothesis | MITRE tactic |
+| -------------------------------------------------------------- | ----------------------------------------------- | ------------------------------- |
+| Unusual auth failures for a user/host | Brute force, credential stuffing | Credential Access (TA0006) |
+| `actual << typical` with `low_count` on auth/process | Service stop, log clearing, defense evasion | Defense Evasion (TA0005) |
+| New/rare entity (first-seen IP, user, process) | Initial access, new implant, new C2 | Initial Access (TA0001) |
+| Entity anomalous in multiple jobs simultaneously | Active compromise, lateral movement in progress | Lateral Movement (TA0008) |
+| Unusual data volume (bytes_out spike) | Data exfiltration | Exfiltration (TA0010) |
+| Rare process execution (high influencer_score on process name) | Malware execution, living-off-the-land | Execution (TA0002) |
+| Auth success following prior auth failures | Successful credential compromise | Credential Access → Persistence |
+| Privilege escalation patterns (sudo, admin role changes) | Admin abuse, shadow IT, misconfiguration | Privilege Escalation (TA0004) |
+| Regular low-volume network spikes (beaconing) | C2 communication | Command & Control (TA0011) |
+
+## Investigation questions
+
+For each anomalous entity, determine:
+
+1. **Known vs first-seen entity** — treat first-seen entities as higher risk.
+2. **Blast radius** — count how many jobs or systems co-fire.
+3. **Temporal chain** — treat auth failure → auth success → lateral movement as a compromise chain hypothesis.
+4. **Source evidence** — treat raw logs as the ground truth.
+5. **MITRE mapping** — map the pattern to the closest tactic and technique.
+
+## Investigation protocol
+
+### Phase 1 — Triage (Investigate mode)
+
+1. `ad_get_available_metadata` — identify security-relevant jobs (auth, network, process, DNS, endpoint).
+2. `ad_query_anomaly_timeline` — establish incident time window.
+3. `ad_rca_multi_job_entities` (`min_job_count=2`) — multi-job entities in security = active threat actors.
+
+### Phase 2 — Entity attribution (Investigate mode)
+
+1. `ad_rca_cross_job_entity_match` — expand from single alert to full entity activity chain.
+2. `ad_query_influencers` (low `min_score`, broad `job_id_pattern`) — surface all associated entities.
+3. `ad_rca_entity_profile` — complete behavioral dossier on suspect user/host/IP.
+
+### Phase 3 — Attack chain reconstruction (Investigate mode)
+
+1. `ad_rca_correlation` sorted by timestamp — reconstruct chronological order. First anomaly = entry point hypothesis.
+2. `ad_rca_blast_radius` — determine lateral spread: how many systems/accounts affected.
+3. `ad_rca_detector_fingerprint` — which behavioral dimensions are anomalous (auth? process? network? data volume?).
+
+### Phase 4 — Evidence collection (Investigate mode)
+
+1. `ad_get_job_datafeed_config` → source index → `ad_rca_source_evidence` — raw forensic ground truth.
+2. For log categorization jobs: `ad_get_categories` + `ad_search_log_category_examples` — compare baseline vs. incident
+ window for changed IPs, credentials, command-line arguments, file paths.
+
+### Phase 5 — Score validation (Explain mode)
+
+1. If score seems low for a suspicious pattern: `ad_rca_score_reassessment` — check renormalization drift.
+ `initial_record_score` may reveal a threat that was renormalized away.
+2. `ad_get_model_plot` — confirm actual exceeds model bounds.
+
+### Phase 6 — Threat report
+
+- **Threat classification**: attack type + MITRE ATT&CK tactic/technique
+- **Confidence**: High/Medium/Low with reasoning
+- **Affected entities**: users, hosts, IPs, processes
+- **Attack timeline**: reconstructed from `first_anomaly` per job
+- **Evidence summary**: key anomalous values from source documents
+- **Recommended response**: containment, investigation, tuning actions
+
+## Security-specific rules
+
+- **Absence anomalies are high priority**: `actual << typical` on auth or process jobs = log clearing or service killing
+ = defense evasion.
+- **`initial_record_score >> record_score`**: do not dismiss. Score was renormalized after a more extreme event — the
+ original anomaly is still a valid threat indicator.
+- **Low scores across many jobs > one high score**: sophisticated attackers stay below single-job thresholds. Composite
+ cross-job signals are the primary detection mechanism.
+- **New entities**: first-seen host or user is higher priority than a known entity with a moderate score.
+- **`multi_bucket_impact ≥ 3`**: sustained shift = persistent access, beaconing, or ongoing exfiltration.
+- Run `ad_validate_ml_tool_permissions` if tools fail — permission errors are common in multi-tenant security
+ environments.
+
+## Escalation vs. tuning
+
+| Signal | Action |
+| ------------------------------------------------ | ------------------------------------------------- |
+| Multi-job entity + source evidence + MITRE match | Escalate as confirmed threat |
+| Multi-job entity + no source evidence | Escalate for manual log review |
+| Single-job, explainable by operational event | Document and tune (calendar event or custom rule) |
+| Renormalization-only score drop | Explain to stakeholder — use Explain mode |
diff --git a/skills/kibana/kibana-anomaly-detection/references/tools.md b/skills/kibana/kibana-anomaly-detection/references/tools.md
new file mode 100644
index 0000000..95c7773
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/tools.md
@@ -0,0 +1,145 @@
+# Tool Reference — Elastic Anomaly Detection Agent Builder
+
+## ES|QL Tools (24) — read-only, query `.ml-anomalies-*` and `.ml-config`
+
+### Discovery & Metadata
+
+| Tool | Description |
+| --------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| `ad_get_available_metadata` | Discover all jobs, fields, functions, entity fields, bucket_spans. **Call first** when jobs are unknown. Returns a single summary row from `.ml-config`. |
+| `ad_get_jobs` | List all jobs with full config: bucket_span, detectors, field names, memory limit, groups, description. |
+| `ad_discover_related_jobs` | Find groups of jobs sharing `partition_field_name`, `by_field_name`, or `over_field_name`. |
+
+### Anomaly Records & Influencers
+
+| Tool | Params | Description |
+| --------------------------- | ----------------------------------------------- | ------------------------------------------------------------------------------------------ |
+| `ad_query_anomaly_records` | job_id_pattern, min_score, start_time, end_time | Search anomaly records cross-job or scoped. Use `*` for overview, exact ID for drill-down. |
+| `ad_query_anomaly_timeline` | job_id_pattern, start_time, end_time | Bucket timeline / composite signal across jobs. |
+| `ad_query_influencers` | job_id_pattern, min_score, start_time, end_time | Find most anomalous entities. Filter `job_count > 1` for cross-job shared influencers. |
+
+### RCA Tools
+
+| Tool | Description |
+| ------------------------------- | --------------------------------------------------------------------------------------------------------------------------- |
+| `ad_rca_multi_job_entities` | Entities anomalous in multiple jobs (min_job_count=2). Prime root-cause signal. |
+| `ad_rca_cross_job_entity_match` | All jobs where a specific entity value appears anomalous right now. Returns per-job first_anomaly timestamp for chronology. |
+| `ad_rca_detector_fingerprint` | Detector-level incident fingerprint — what aspects of the system are anomalous (CPU? latency?). |
+| `ad_rca_correlation` | Temporal correlation / cascade. Sort by timestamp — earliest anomaly for the entity hints at root cause. |
+| `ad_rca_blast_radius` | Blast radius across jobs for an entity/time window. |
+| `ad_rca_entity_profile` | Complete dossier on a suspect entity. |
+| `ad_rca_source_evidence` | Raw source documents from the original data index. Get index from `ad_get_job_datafeed_config`. |
+| `ad_rca_score_reassessment` | Score drift (`score_drift = initial_record_score - record_score`). Quantify renormalization. |
+
+### Log Categorization
+
+| Tool | Description |
+| --------------------------------- | ------------------------------------------------------------------------------------------------------- |
+| `ad_get_categories` | Category definitions (terms, regex, examples) for categorization jobs. |
+| `ad_search_log_category_examples` | Log samples for a category in a time window. Run twice (baseline + anomaly) and compare variable parts. |
+
+### Model Insight
+
+| Tool | Description |
+| ------------------------- | --------------------------------------------------------------------------------------------------------------- |
+| `ad_get_model_plot` | Model upper/lower/median bounds over time. Most visual explanation for non-technical users. |
+| `ad_get_forecast_results` | Forecast predictions with upper/lower bounds for capacity planning. |
+| `ad_get_model_snapshots` | Available model snapshots for a job. Used before `ad_revert_model_snapshot`. |
+| `ad_get_job_messages` | All notifications from `.ml-notifications-*`: datafeed warnings, delayed data, memory limits, lifecycle events. |
+
+### Time-Series Diagnostics
+
+| Tool | Params | Description |
+| -------------------------------- | ------------------------------------- | --------------------------------------------------------------------------------------------------------- |
+| `ad_ts_model_memory_health` | job_id, limit (1=snapshot, 500=trend) | Memory status time series: model_bytes, limit, memory_status, entity counts. |
+| `ad_ts_ingest_latency_estimate` | source_index, start_time, end_time | P50/P95/P99 ingest latency. Requires `event.ingested` field. If P95 > query_delay → data will be lost. |
+| `ad_ts_bucket_event_gaps` | job_id, start_time, end_time | Buckets with zero or low event counts. Correlate with delayed data annotations. |
+| `ad_ts_delayed_data_annotations` | job_id | All delayed data annotations from `.ml-annotations-*`. Starting point for missing-document investigation. |
+
+---
+
+## Workflow Tools (23 YAML files) — REST API + YAML, management and remediation
+
+### Discovery & Config
+
+| Tool | Description |
+| ------------------------------------ | ------------------------------------------------------------------------------------------------------------------- |
+| `ad_discover_jobs_by_datafeed_index` | Jobs sharing datafeed source indices — strongest related-job signal. Iterates job's indices, queries `.ml-config`. |
+| `ad_get_job_datafeed_config` | Full job + datafeed config: detectors, fields, bucket_span, query_delay, delayed_data_check_config, source indices. |
+
+### Datafeed Operations
+
+| Tool | Description |
+| ---------------------------------- | ------------------------------------------------------------------------------------------- |
+| `ad_manage_datafeed` | Start or stop a datafeed (`_start` / `_stop`). Preview: `ad_preview_datafeed_with_latency`. |
+| `ad_preview_datafeed_with_latency` | Preview datafeed source payload and measure effective latency before tuning query_delay. |
+
+### Config Updates
+
+| Tool | Description |
+| ------------------------------------- | ----------------------------------------------------------------------------------------------------------------------- |
+| `ad_update_datafeed_query_delay` | Update `query_delay`. Stop datafeed first. Set to P95 ingest latency + buffer. |
+| `ad_update_delayed_data_check_config` | Enable/disable delayed data checks or adjust `check_window`. |
+| `ad_update_model_memory_limit` | Update `model_memory_limit`. Requires stop/close/update/open/start sequence. Cannot decrease below current model_bytes. |
+
+### Sizing & Estimation
+
+| Tool | Description |
+| -------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| `ad_wf_ts_field_cardinality` | ES | QL `POST /_query`: COUNT_DISTINCT on a source split field; column name is spliced into the query (`?` params are literals only). Use values from job analysis config only. |
+| `ad_estimate_memory_requirement` | Principled memory sizing: auto-samples cardinality from source data → calls Estimate Model Memory API. Better than `peak_model_bytes * 1.3` (ignores pure influencer memory). |
+
+### Permissions & CCS
+
+| Tool | Description |
+| --------------------------------- | --------------------------------------------------------------------------------------------------------------------- |
+| `ad_validate_ml_tool_permissions` | Preflight: read + view_index_metadata on `.ml-anomalies-*`, `.ml-config`, `.ml-annotations-*`, `.ml-notifications-*`. |
+| `ad_ts_ccs_diagnostics` | CCS diagnostics: remote cluster connectivity, latency, error rates for cross-cluster datafeeds. |
+
+### Lifecycle & Recovery
+
+| Tool | Description |
+| -------------------------- | -------------------------------------------------------------- |
+| `ad_revert_model_snapshot` | Revert to a previous model snapshot. Job must be closed first. |
+| `ad_validate_job_spec` | POST `/_ml/anomaly_detectors/_validate` — validate job JSON |
+| `ad_create_job` | PUT `/_ml/anomaly_detectors/{job_id}` — create job |
+| `ad_create_datafeed` | PUT `/_ml/datafeeds/{datafeed_id}` — create datafeed |
+| `ad_open_job` | POST `/_ml/anomaly_detectors/{job_id}/_open` |
+
+### Calendars
+
+| Tool | Description |
+| -------------------------- | ------------------------------------------------------------------------ |
+| `ad_get_calendar_events` | Get scheduled events from calendars (maintenance windows, holidays). |
+| `ad_create_calendar_event` | Add a scheduled event to suppress false positives during known downtime. |
+
+### Stored Troubleshooting Workflows (decision trees from support runbooks)
+
+| Tool | Trigger |
+| ---------------------------------- | ----------------------------------------------------------------------------- |
+| `ad_wf_troubleshoot_anomaly_score` | "Why is my score low?", "Expected anomaly not detected", "Score too high/low" |
+| `ad_wf_troubleshoot_query_delay` | "My job reports missing documents", "Datafeed has missed X documents" |
+| `ad_wf_troubleshoot_memory_limit` | "My job hit memory limit", "hard_limit", "soft_limit" |
+
+---
+
+## Key System Indices
+
+| Index | `result_type` values |
+| --------------------- | ----------------------------------------------------------------------------------------------------------------------------- |
+| `.ml-anomalies-*` | `bucket`, `record`, `influencer`, `model_size_stats`, `model_plot`, `model_forecast`, `model_snapshot`, `category_definition` |
+| `.ml-annotations-*` | delayed data (`event == "delayed_data"`) |
+| `.ml-notifications-*` | job messages (datafeed warnings, memory limits, lifecycle) |
+| `.ml-config` | job/datafeed documents (used for discovery — all jobs visible, even never-run) |
+
+## Registration
+
+Requires Node.js 18+. Defaults to `elastic`/`changeme` when no credentials supplied.
+
+```bash
+cd skills/kibana/kibana-anomaly-detection
+node scripts/kibana-agent-builder.mjs all register --kibana-url http://localhost:5601
+```
+
+Workflow tools are skipped automatically until Elastic Workflows (preview) is enabled. Configure exclusions in
+`scripts/agent_builder_constants.json`.
diff --git a/skills/kibana/kibana-anomaly-detection/references/troubleshoot-anomaly-tool-reference.md b/skills/kibana/kibana-anomaly-detection/references/troubleshoot-anomaly-tool-reference.md
new file mode 100644
index 0000000..5787d8d
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/troubleshoot-anomaly-tool-reference.md
@@ -0,0 +1,431 @@
+# Troubleshoot mode — tool reference
+
+ES|QL and workflow tool parameters, REST fallbacks, and decision trees for the **Troubleshoot** mode of the parent
+[SKILL.md](../SKILL.md).
+
+## ES|QL Tools (8)
+
+### `ad_get_available_metadata`
+
+Discover all jobs and metadata. **Call first.**
+
+```esql
+FROM .ml-config
+| WHERE job_type == "anomaly_detector"
+| STATS job_count = COUNT(*), job_ids = VALUES(job_id),
+ functions = VALUES(`analysis_config.detectors.function`),
+ bucket_spans = VALUES(`analysis_config.bucket_span`)
+```
+
+_No parameters._
+
+---
+
+### `ad_get_jobs`
+
+List all jobs with config, memory limit, and state context.
+
+```esql
+FROM .ml-config
+| WHERE job_type == "anomaly_detector"
+| KEEP job_id, `analysis_config.bucket_span`, `analysis_config.detectors.function`,
+ `analysis_config.detectors.partition_field_name`,
+ `analysis_config.detectors.by_field_name`,
+ `analysis_limits.model_memory_limit`, groups, description
+| SORT job_id ASC | LIMIT 100
+```
+
+_No parameters._
+
+---
+
+### `ad_get_job_messages`
+
+All notifications from `.ml-notifications-*`: datafeed warnings, delayed data, memory limits, lifecycle events, errors.
+
+| Parameter | Type | Description |
+| --------- | ---- | ----------- |
+| `job_id` | text | Job ID |
+
+```esql
+FROM .ml-notifications-*
+| WHERE job_id == ?job_id
+| SORT timestamp DESC
+| KEEP timestamp, level, message, node_name, job_id
+| LIMIT 50
+```
+
+---
+
+### `ad_get_model_snapshots`
+
+Available model snapshots. Review before using `ad_revert_model_snapshot`.
+
+| Parameter | Type | Description |
+| --------- | ---- | ----------- |
+| `job_id` | text | Job ID |
+
+```esql
+FROM .ml-anomalies-*
+| WHERE result_type == "model_snapshot" AND job_id == ?job_id
+| SORT timestamp DESC
+| KEEP job_id, timestamp, description, snapshot_doc_count
+| LIMIT 20
+```
+
+---
+
+### `ad_ts_model_memory_health`
+
+Memory status time series. `limit=1` for current snapshot; `limit=500` for trend and trajectory.
+
+**Interpretation:** `hard_limit` = CRITICAL (job blind to new entities). `soft_limit` = WARNING (pruning).
+`model_bytes / model_bytes_memory_limit > 0.8` = APPROACHING LIMIT.
+
+| Parameter | Type | Description |
+| --------- | ------- | ----------------------------- |
+| `job_id` | text | Job ID |
+| `limit` | integer | 1 = current, 500 = full trend |
+
+```esql
+FROM .ml-anomalies-*
+| WHERE result_type == "model_size_stats" AND job_id == ?job_id
+| SORT timestamp DESC
+| KEEP job_id, timestamp, model_bytes, peak_model_bytes, model_bytes_memory_limit,
+ model_bytes_exceeded, memory_status,
+ total_by_field_count, total_over_field_count, total_partition_field_count,
+ bucket_allocation_failures_count
+| LIMIT ?limit
+```
+
+---
+
+### `ad_ts_ingest_latency_estimate`
+
+Measure actual ingest latency via `event.ingested`. If P95 > `query_delay` → data is being lost.
+
+| Parameter | Type | Description |
+| -------------- | ---- | --------------------------------- |
+| `source_index` | text | LIKE pattern from datafeed config |
+| `start_time` | text | ISO 8601 |
+| `end_time` | text | ISO 8601 |
+
+```esql
+FROM * METADATA _index
+| WHERE _index LIKE ?source_index
+ AND @timestamp >= ?start_time AND @timestamp <= ?end_time
+ AND event.ingested IS NOT NULL
+| EVAL latency_seconds = DATE_DIFF("second", @timestamp, event.ingested)
+| STATS p50_latency = PERCENTILE(latency_seconds, 50),
+ p95_latency = PERCENTILE(latency_seconds, 95),
+ p99_latency = PERCENTILE(latency_seconds, 99),
+ max_latency = MAX(latency_seconds), doc_count = COUNT(*)
+| LIMIT 1
+```
+
+---
+
+### `ad_ts_bucket_event_gaps`
+
+Buckets with zero or low event counts. Correlate with delayed data annotations to confirm false positives from missing
+data.
+
+| Parameter | Type | Description |
+| ------------ | ---- | ----------- |
+| `job_id` | text | Job ID |
+| `start_time` | text | ISO 8601 |
+| `end_time` | text | ISO 8601 |
+
+```esql
+FROM .ml-anomalies-*
+| WHERE result_type == "bucket"
+ AND job_id == ?job_id
+ AND timestamp >= ?start_time AND timestamp <= ?end_time
+| SORT timestamp ASC
+| KEEP timestamp, event_count, anomaly_score, bucket_span, is_interim
+| LIMIT 500
+```
+
+---
+
+### `ad_ts_delayed_data_annotations`
+
+Delayed data annotations — starting point for any missing documents investigation.
+
+| Parameter | Type | Description |
+| --------- | ---- | ----------- |
+| `job_id` | text | Job ID |
+
+```esql
+FROM .ml-annotations-*
+| WHERE job_id == ?job_id AND event == "delayed_data"
+| SORT timestamp DESC
+| KEEP job_id, timestamp, end_timestamp, annotation
+| LIMIT 100
+```
+
+---
+
+## Workflow Tools (15)
+
+### `ad_get_job_datafeed_config`
+
+Full job + datafeed config: `bucket_span`, `query_delay`, `delayed_data_check_config`, source indices. Essential for all
+troubleshooting and memory estimation.
+
+| Parameter | Type | Required |
+| --------- | ------ | -------- |
+| `job_id` | string | yes |
+
+```text
+Step 1: GET _ml/anomaly_detectors/{job_id}
+Step 2: GET _ml/anomaly_detectors/{job_id}/_stats
+```
+
+---
+
+### `ad_manage_datafeed`
+
+Start or stop a datafeed via `POST _ml/datafeeds/{datafeed_id}/{_start|_stop}`. For preview, use
+`ad_preview_datafeed_with_latency` (`GET _ml/datafeeds/{datafeed_id}/_preview`).
+
+| Parameter | Type | Required | Description |
+| ------------- | ------ | -------- | --------------------------- |
+| `datafeed_id` | string | yes | Usually `datafeed-{job_id}` |
+| `action` | string | yes | `_start` or `_stop` only |
+
+```text
+POST _ml/datafeeds/{datafeed_id}/{action}
+```
+
+---
+
+### `ad_preview_datafeed_with_latency`
+
+Preview datafeed payload and measure effective latency before tuning query_delay.
+
+| Parameter | Type | Required |
+| ------------- | ------ | -------- |
+| `datafeed_id` | string | yes |
+
+```text
+GET _ml/datafeeds/{datafeed_id}/_preview
+```
+
+---
+
+### `ad_update_datafeed_query_delay`
+
+Update `query_delay`. **Stop datafeed first.** Set to P95 ingest latency + buffer.
+
+| Parameter | Type | Required | Description |
+| ----------------- | ------ | -------- | ----------------- |
+| `datafeed_id` | string | yes | |
+| `new_query_delay` | string | yes | e.g. `3m`, `120s` |
+
+```text
+POST _ml/datafeeds/{datafeed_id}/_update
+Body: { "query_delay": "{new_query_delay}" }
+```
+
+---
+
+### `ad_update_delayed_data_check_config`
+
+Enable/disable delayed data checks or adjust `check_window`.
+
+| Parameter | Type | Required | Description |
+| -------------- | ------- | -------- | --------------------------------------------------------------------- |
+| `job_id` | string | yes | |
+| `enabled` | boolean | yes | |
+| `check_window` | string | no | e.g. `2h`; omitted from the update body when empty or whitespace-only |
+
+```text
+Step 1: data.parseJson — build update body (enabled as JSON boolean; check_window only if non-blank after trim)
+Step 2: POST _ml/anomaly_detectors/{job_id}/_update with that body
+```
+
+---
+
+### `ad_update_model_memory_limit`
+
+Update `model_memory_limit`. **Cannot decrease below current `model_bytes`** — clone job to shrink. Requires
+stop/close/update/open/start sequence.
+
+| Parameter | Type | Required | Description |
+| ----------- | ------ | -------- | ------------------- |
+| `job_id` | string | yes | |
+| `new_limit` | string | yes | e.g. `256mb`, `1gb` |
+
+```text
+POST _ml/anomaly_detectors/{job_id}/_update
+Body: { "analysis_limits": { "model_memory_limit": "{new_limit}" } }
+```
+
+---
+
+### `ad_estimate_memory_requirement`
+
+**Best practice for memory sizing.** Auto-samples cardinality from source → calls
+`POST _ml/anomaly_detectors/_estimate_model_memory`. More accurate than `peak_model_bytes * 1.3`.
+
+| Parameter | Type | Required | Description |
+| -------------- | ------ | -------- | --------------------------------- |
+| `job_id` | string | yes | |
+| `sample_start` | string | no | ISO 8601, defaults to 30 days ago |
+| `sample_end` | string | no | ISO 8601, defaults to now |
+
+```text
+Step 1: GET _ml/anomaly_detectors/{job_id}
+Step 2: Extract split fields and pure influencers
+Step 3: POST {datafeed_indices}/_search?size=0 → overall_cardinality per field
+Step 4: POST {datafeed_indices}/_search?size=0 → max_bucket_cardinality
+Step 5: POST _ml/anomaly_detectors/_estimate_model_memory
+Step 6: Compare estimate vs model_memory_limit vs model_bytes
+Step 7: Recommend (increase / reduce data / restructure)
+```
+
+---
+
+### `ad_wf_ts_field_cardinality`
+
+Cardinality of a split field in **source** data. If source cardinality >> the model's `total_*_count` from
+`ad_ts_model_memory_health`, entities may be dropped.
+
+This is a **workflow** (not an ES|QL Agent Builder tool): it runs `POST /_query` and **splices** `split_field_esql` into
+the query text as the `COUNT_DISTINCT` column. ES|QL `?` parameters bind **literals only**, so field names cannot be
+passed as `?` parameters.
+
+| Parameter | Type | Required | Description |
+| ------------------ | ------ | -------- | ------------------------------------------------------------------------------------------------ |
+| `source_index` | string | yes | LIKE pattern from datafeed config |
+| `split_field_esql` | string | yes | Column expression for `COUNT_DISTINCT` (for example `service.keyword`); from job analysis config |
+| `start_time` | string | yes | ISO 8601 |
+| `end_time` | string | yes | ISO 8601 |
+
+```text
+POST /_query
+Body includes "query" with COUNT_DISTINCT() and "params" for ?source_index, ?start_time, ?end_time
+```
+
+---
+
+### `ad_validate_ml_tool_permissions`
+
+Preflight check on core `.ml-*` indices (`read` + `view_index_metadata` on `.ml-anomalies-*`, `.ml-config`,
+`.ml-annotations-*`, `.ml-notifications-*`). Does **not** assert privileges on job source data indices — check those
+before previews or `ad_rca_source_evidence`.
+
+_No parameters._
+
+```text
+Step 1: GET _security/_authenticate
+Step 2: POST _security/user/_has_privileges
+ (.ml-anomalies-*, .ml-config, .ml-annotations-*, .ml-notifications-*)
+```
+
+---
+
+### `ad_ts_ccs_diagnostics`
+
+Cross-cluster search diagnostics: remote cluster connectivity, latency, error rates.
+
+_No parameters._
+
+```text
+Step 1: GET _remote/info
+Step 2: GET _cluster/health
+```
+
+---
+
+### `ad_revert_model_snapshot`
+
+Revert to a previous model snapshot to "unlearn" bad data. Job must be **closed** first. After reverting, reopen and
+restart datafeed from the snapshot timestamp.
+
+| Parameter | Type | Required |
+| ------------- | ------ | -------- |
+| `job_id` | string | yes |
+| `snapshot_id` | string | yes |
+
+```text
+POST _ml/anomaly_detectors/{job_id}/model_snapshots/{snapshot_id}/_revert
+```
+
+---
+
+### `ad_get_calendar_events`
+
+Get scheduled events from a calendar (maintenance windows, holidays).
+
+| Parameter | Type | Required |
+| ------------- | ------ | -------- |
+| `calendar_id` | string | yes |
+
+```text
+GET _ml/calendars/{calendar_id}/events
+```
+
+---
+
+### `ad_create_calendar_event`
+
+Add a scheduled event to suppress false positives during known downtime.
+
+| Parameter | Type | Required | Description |
+| ------------- | ------ | -------- | --------------------------------- |
+| `calendar_id` | string | yes | |
+| `description` | string | yes | e.g. `Planned maintenance window` |
+| `start_time` | string | yes | ISO 8601 or epoch_millis |
+| `end_time` | string | yes | ISO 8601 or epoch_millis |
+
+```text
+POST _ml/calendars/{calendar_id}/events
+Body: { "events": [{ "description": ..., "start_time": ..., "end_time": ... }] }
+```
+
+---
+
+### `ad_wf_troubleshoot_query_delay`
+
+Full automated decision tree for missing documents diagnosis.
+
+| Parameter | Type | Required |
+| --------- | ------ | -------- |
+| `job_id` | string | yes |
+
+Decision tree:
+
+1. Retrieve config + `memory_status`.
+2. **Gate:** Categorization job + `memory_status == hard_limit` → EXIT EARLY. Missing-doc warning is a false alarm from
+ memory exhaustion. Fix memory first.
+3. `ad_ts_delayed_data_annotations` — frequency, severity, affected ranges.
+4. `ad_ts_bucket_event_gaps` — zero/low event count buckets.
+5. `ad_ts_ingest_latency_estimate` — P50/P95/P99. If P95 > `query_delay` → data lost.
+6. Recommend `query_delay` = P95 + buffer.
+7. Additional remediation: add ingest pipeline, use ingest timestamp as `time_field`, revert snapshot + backfill.
+
+---
+
+### `ad_wf_troubleshoot_memory_limit`
+
+Full automated decision tree for memory limit diagnosis.
+
+| Parameter | Type | Required |
+| --------- | ------ | -------- |
+| `job_id` | string | yes |
+
+Decision tree:
+
+1. `ad_ts_model_memory_health` (limit=1): classify as `ok` / `soft_limit` / `hard_limit`.
+2. If `hard_limit`: check notifications for "missed documents" warnings — these are SYMPTOMS of hard_limit, not ingest
+ lag.
+3. `ad_ts_model_memory_health` (limit=500): stable / linear / exponential growth. Predict time-to-limit.
+4. Inspect `model_size_stats`: `total_by_field_count > 100K`, `total_partition_field_count > 10K`,
+ `total_category_count > 10K` → identify dominant driver.
+5. `ad_estimate_memory_requirement`: principled sizing.
+6. Recommend: **A** — Increase limit; **B** — Reduce data (filter datafeed, reduce influencers); **C** — Multi-GB
+ architectural restructuring.
+
+---
diff --git a/skills/kibana/kibana-anomaly-detection/references/worked-example.md b/skills/kibana/kibana-anomaly-detection/references/worked-example.md
new file mode 100644
index 0000000..3219870
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/worked-example.md
@@ -0,0 +1,263 @@
+# Worked Investigation Example
+
+End-to-end walkthrough of a multi-job anomaly investigation using the [14-step protocol](protocols/investigation.md).
+
+---
+
+## Scenario
+
+**Alert received:** `rcaeval-ob-cpu` — `partition_field_value: frontend` — `record_score: 72` — `2024-03-15T14:30:00Z`
+
+The on-call engineer receives this alert and needs to determine: Is `frontend` the root cause, or a victim of something
+upstream?
+
+---
+
+## Phase 1: Discovery
+
+### Step 1 — Discover available jobs
+
+Call `ad_get_available_metadata` (no parameters).
+
+**Result summary:**
+
+```text
+job_count: 6
+job_ids: [rcaeval-ob-cpu, rcaeval-ob-latency, rcaeval-ob-memory,
+ rcaeval-ob-errors, rcaeval-nw-throughput, rcaeval-logs-app]
+functions: [mean, high_mean, count, rare]
+partition_fields: [service]
+bucket_spans: [5m]
+```
+
+**Interpretation:** 6 jobs all partitioned by `service`. All likely monitor the same system from different angles.
+
+---
+
+### Step 2 — Find related jobs
+
+Call `ad_discover_jobs_by_datafeed_index` with `job_id: rcaeval-ob-cpu`.
+
+**Result:**
+
+```text
+Jobs sharing index rcaeval-re1-ob:
+ rcaeval-ob-cpu, rcaeval-ob-latency, rcaeval-ob-memory, rcaeval-ob-errors
+ match_count: 4
+
+Jobs sharing index rcaeval-re1-nw:
+ rcaeval-nw-throughput
+ match_count: 1
+```
+
+Then call `ad_discover_related_jobs` with `job_id: rcaeval-ob-cpu`.
+
+**Result:**
+
+```text
+entity_field: service
+ jobs: [rcaeval-ob-cpu, rcaeval-ob-latency, rcaeval-ob-memory,
+ rcaeval-ob-errors, rcaeval-nw-throughput]
+ job_count: 5
+```
+
+**Interpretation:** 5 jobs all split by `service`. The `rcaeval-logs-app` job uses `mlcategory` (log categorization),
+not `service`. The 5 observability jobs are our related group.
+
+---
+
+### Step 3 — Scope the incident
+
+Call `ad_query_anomaly_timeline` with:
+
+- `job_id_pattern: rcaeval-*`
+- `min_score: 25`
+- `start_time: 2024-03-15T13:00:00Z`
+- `end_time: 2024-03-15T16:00:00Z`
+
+**Result:**
+
+```text
+timestamp max_score job_count composite_score jobs
+2024-03-15T14:00:00Z 48 2 76 [rcaeval-ob-cpu, rcaeval-ob-latency]
+2024-03-15T14:15:00Z 61 3 138 [rcaeval-ob-cpu, rcaeval-ob-latency, rcaeval-ob-memory]
+2024-03-15T14:30:00Z 78 4 198 [rcaeval-ob-cpu, rcaeval-ob-latency, rcaeval-ob-memory, rcaeval-ob-errors]
+2024-03-15T14:45:00Z 72 4 201 [rcaeval-ob-cpu, rcaeval-ob-latency, rcaeval-ob-memory, rcaeval-ob-errors]
+2024-03-15T15:00:00Z 45 3 112 [rcaeval-ob-latency, rcaeval-ob-memory, rcaeval-ob-errors]
+```
+
+**Interpretation:** Peak at 14:30 with 4 jobs co-firing (composite score 198). Incident started ~14:00, peak 14:30,
+declining by 15:00. This is a 1-hour incident, not a transient spike. CPU was first to fire (14:00).
+
+---
+
+## Phase 2: Entity Attribution
+
+### Step 4 — Expand from alert
+
+Call `ad_rca_cross_job_entity_match` with:
+
+- `entity_value: frontend`
+- `min_score: 10`
+- `start_time: 2024-03-15T13:30:00Z`
+- `end_time: 2024-03-15T15:30:00Z`
+
+**Result:**
+
+```text
+job_id max_score anomaly_count first_anomaly functions
+rcaeval-ob-cpu 78 8 2024-03-15T14:00:00Z [high_mean]
+rcaeval-ob-latency 74 7 2024-03-15T14:05:00Z [high_mean]
+rcaeval-ob-memory 61 5 2024-03-15T14:15:00Z [high_mean]
+rcaeval-ob-errors 52 4 2024-03-15T14:20:00Z [high_count]
+```
+
+**Interpretation:** `frontend` is anomalous in 4 jobs. CPU was first (14:00), then latency (14:05), then memory (14:15),
+then errors (14:20). This cascading pattern suggests CPU is upstream — high CPU caused latency, which caused memory
+pressure, which caused errors.
+
+---
+
+### Step 5 — Multi-job entities
+
+Call `ad_rca_multi_job_entities` with:
+
+- `min_score: 25`
+- `min_job_count: 2`
+- `start_time: 2024-03-15T14:00:00Z`
+- `end_time: 2024-03-15T15:00:00Z`
+
+**Result:**
+
+```text
+partition_field_value job_count max_score functions
+frontend 4 78 [high_mean, high_count]
+backend 1 31 [high_mean]
+```
+
+**Interpretation:** `frontend` is the only entity anomalous in 4+ jobs — confirmed prime suspect. `backend` appears in
+only 1 job at score 31 — likely incidental.
+
+---
+
+### Step 6 — Fingerprint
+
+Call `ad_rca_detector_fingerprint` with:
+
+- `job_id_pattern: rcaeval-ob-*`
+- `min_score: 25`
+- `start_time: 2024-03-15T14:00:00Z`
+- `end_time: 2024-03-15T15:00:00Z`
+
+**Result:**
+
+```text
+job_id function field_name max_score count
+rcaeval-ob-cpu high_mean system.cpu.percent 78 8
+rcaeval-ob-latency high_mean http.response_time 74 7
+rcaeval-ob-memory high_mean system.memory.used 61 5
+rcaeval-ob-errors high_count http.error_count 52 4
+```
+
+**Interpretation:** Classic resource exhaustion pattern: CPU spike → latency increase → memory pressure → errors. All
+metrics elevated on `frontend` simultaneously.
+
+---
+
+## Phase 3: Deep Analysis
+
+### Steps 7–10 — Drill down, attribute, profile, characterize
+
+`ad_query_anomaly_records` for `rcaeval-ob-cpu` with `min_score: 50`:
+
+```text
+record_score: 78
+actual: 94.7 (% CPU)
+typical: 31.2
+multi_bucket_impact: 4
+initial_record_score: 79
+```
+
+`multi_bucket_impact: 4` → sustained shift across 4+ buckets, not a transient spike. `initial ≈ current` → no
+renormalization; score is stable and genuine.
+
+---
+
+## Phase 4: Root Cause Confirmation
+
+### Step 11 — Cascade
+
+Call `ad_rca_correlation` with:
+
+- `job_id_pattern: rcaeval-ob-*`
+- `min_score: 25`
+- Window: 13:30–15:30
+
+**Result (sorted by timestamp):**
+
+```text
+timestamp job_id record_score function field_name
+2024-03-15T14:00:00Z rcaeval-ob-cpu 78 high_mean cpu.percent → FIRST
+2024-03-15T14:05:00Z rcaeval-ob-latency 74 high_mean response_time
+2024-03-15T14:15:00Z rcaeval-ob-memory 61 high_mean memory.used
+2024-03-15T14:20:00Z rcaeval-ob-errors 52 high_count error_count
+```
+
+**Interpretation:** CPU anomaly at 14:00 is the earliest — this is the root cause signal.
+
+### Step 12 — Evidence
+
+Get source index from `ad_get_job_datafeed_config` → `rcaeval-re1-ob`.
+
+Call `ad_rca_source_evidence`:
+
+- `source_index: rcaeval-re1-ob`
+- Window: 13:50–14:10
+
+**Sample raw docs (14:00):**
+
+```json
+{"service": "frontend", "system.cpu.percent": 96.2, "event": "metricset", "@timestamp": "2024-03-15T14:00:34Z"}
+{"service": "frontend", "system.cpu.percent": 93.8, "event": "metricset", "@timestamp": "2024-03-15T14:01:05Z"}
+{"service": "frontend", "system.cpu.percent": 94.1, "event": "metricset", "@timestamp": "2024-03-15T14:01:41Z"}
+```
+
+**Interpretation:** CPU is genuinely at ~95% on `frontend` starting at 14:00, confirming the ML anomaly matches reality.
+
+---
+
+## Step 14 — RCA Report
+
+**Root cause:** `frontend` service experienced CPU exhaustion starting at 2024-03-15T14:00Z.
+
+**Evidence:**
+
+- CPU rose from baseline 31% to 94–97% at 14:00 (score: 78, `multi_bucket_impact: 4` — sustained shift)
+- CPU anomaly preceded all other anomalies by 5–20 minutes
+- `frontend` was the only entity anomalous in 4 jobs simultaneously — no other service implicated
+
+**Cascading impact:**
+
+1. **14:00** — CPU saturates (94%+)
+2. **14:05** — HTTP latency climbs (CPU-bound request processing)
+3. **14:15** — Memory rises (queued/retried requests consuming heap)
+4. **14:20** — Error rate spikes (timeouts from downstream callers)
+
+**Fault type:** Resource exhaustion — CPU-bound processing on `frontend` pod(s)
+
+**Recommended actions:**
+
+1. Check `frontend` deployment for runaway process or CPU-hungry code path (profiling)
+2. Review recent deploys to `frontend` in the 30 min before 14:00
+3. Horizontal scale or vertical CPU limit increase as immediate mitigation
+4. Add CPU throttling alert at 80% to catch before next saturation
+
+**Severity:** Score 78, `multi_bucket_impact: 4`, duration ~1 hour — **high severity**
+
+---
+
+## See Also
+
+- [protocols/investigation.md](protocols/investigation.md) — Full 14-step protocol
+- [score-reference.md](score-reference.md) — Score field definitions and severity bands
+- [anomaly-detection-functions.md](anomaly-detection-functions.md) — Function selection guide
diff --git a/skills/kibana/kibana-anomaly-detection/references/workflow-tools.md b/skills/kibana/kibana-anomaly-detection/references/workflow-tools.md
new file mode 100644
index 0000000..7118945
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/references/workflow-tools.md
@@ -0,0 +1,536 @@
+# Workflow Tools Reference
+
+Full documentation for all workflow tools. Most call Elasticsearch ML HTTP APIs (REST) via the Kibana Agent Builder. A
+few run ES|QL via `POST /_query` when a column identifier must be spliced into the query text (ES|QL `?` parameters bind
+literals only, not field names).
+
+> For ES|QL read tools, see the skill SKILL.md files. For permissions required, see
+> [permissions-matrix.md](permissions-matrix.md).
+
+---
+
+## Availability Note
+
+Workflow tools require **Kibana Elastic Agent Workflows** to be enabled. Not all deployments have this feature. If a
+workflow tool is unavailable, the description includes a manual fallback — prefer ES|QL queries first, then REST API
+calls if ES|QL cannot express the operation.
+
+---
+
+## Job & Datafeed Configuration
+
+### `ad_get_job_datafeed_config`
+
+Fetch complete job and datafeed configuration in one call.
+
+| Parameter | Type | Required | Description |
+| --------- | ------ | -------- | ---------------------------- |
+| `job_id` | string | yes | The anomaly detection job ID |
+
+**API calls:**
+
+```text
+GET _ml/anomaly_detectors/{job_id}
+GET _ml/anomaly_detectors/{job_id}/_stats
+```
+
+**Returns:** `analysis_config` (detectors, by/over/partition fields, bucket_span, frequency), `datafeed_config` (source
+indices, query, query_delay, delayed_data_check_config), and runtime stats (memory_status, model_bytes, data_counts,
+state).
+
+**When to use:** Required before calling `ad_rca_source_evidence` (to get the source index). Also used in
+troubleshooting workflows to inspect query_delay, bucket_span, and custom_rules.
+
+---
+
+### `ad_discover_jobs_by_datafeed_index`
+
+Given a job of interest, find all other jobs whose datafeed reads from the same source indices.
+
+| Parameter | Type | Required | Description |
+| --------- | ------ | -------- | ---------------------------------------- |
+| `job_id` | string | yes | The job ID whose source indices to match |
+
+**API calls:**
+
+```text
+GET _ml/anomaly_detectors/{job_id} ← get target job's datafeed_config.indices
+GET .ml-config/_search ← find all other jobs with overlapping indices
+```
+
+**Returns:** Jobs grouped by shared index pattern, with match count per group.
+
+**When to use:** Step 2 of every investigation. Jobs reading from the same source index monitor the same underlying
+system — strongest config-level relatedness signal.
+
+\*\*Fallback if unavailable: Use ESQL to find get target job's datafeed_config.indices then all other jobs with
+overlapping indices
+
+```esql
+FROM .ml-config
+| WHERE job_type == "anomaly_detector"
+```
+
+---
+
+## Datafeed Lifecycle
+
+### `ad_manage_datafeed`
+
+Start or stop a datafeed (`POST _ml/datafeeds/{datafeed_id}/_start` or `/_stop`). Preview uses a separate GET endpoint;
+use `ad_preview_datafeed_with_latency` instead of `_preview` on this workflow.
+
+| Parameter | Type | Required | Description |
+| ------------- | ------ | -------- | ----------------------------- |
+| `datafeed_id` | string | yes | Typically `datafeed-{job_id}` |
+| `action` | string | yes | `_start` or `_stop` only |
+
+**API call:**
+
+```text
+POST _ml/datafeeds/{datafeed_id}/{action}
+```
+
+**When to use:** Required as part of remediation sequences:
+
+- Stop datafeed before updating `query_delay` or `model_memory_limit`
+- Restart datafeed after updating job config
+- To preview extracted rows before starting, call `ad_preview_datafeed_with_latency`
+ (`GET _ml/datafeeds/{datafeed_id}/_preview`)
+
+---
+
+### `ad_preview_datafeed_with_latency`
+
+Preview a datafeed's output to inspect data quality and measure effective latency.
+
+| Parameter | Type | Required | Description |
+| ------------- | ------ | -------- | -------------------------- |
+| `datafeed_id` | string | yes | The datafeed ID to preview |
+
+**API call:**
+
+```text
+GET _ml/datafeeds/{datafeed_id}/_preview
+```
+
+**Returns:** Sample documents that the datafeed would extract, including field values and timestamps. Use to verify: Is
+`event.ingested` available? Are all expected fields present? What does the data look like at query time?
+
+**When to use:** Before tuning `query_delay` — understand what data the datafeed sees and which timestamp fields are
+available for latency measurement.
+
+---
+
+### `ad_update_datafeed_query_delay`
+
+Update the `query_delay` on a datafeed to capture more late-arriving data.
+
+| Parameter | Type | Required | Description |
+| ----------------- | ------ | -------- | ----------------------------------------- |
+| `datafeed_id` | string | yes | The datafeed ID |
+| `new_query_delay` | string | yes | New delay value, e.g., `3m`, `120s`, `5m` |
+
+**API call:**
+
+```text
+POST _ml/datafeeds/{datafeed_id}/_update
+Body: {"query_delay": "{new_query_delay}"}
+```
+
+**Prerequisites:** Datafeed must be stopped first (`ad_manage_datafeed` with `_stop`).
+
+**Trade-off:** Larger `query_delay` = more late-arriving data captured = slower anomaly alerts. Set to P95 ingest
+latency + a buffer (e.g., if P95 latency is 90s, set to `2m`).
+
+---
+
+### `ad_update_delayed_data_check_config`
+
+Control how aggressively delayed data is detected and annotated.
+
+| Parameter | Type | Required | Description |
+| -------------- | ------- | -------- | -------------------------------------------------------------------------------------------------------- |
+| `job_id` | string | yes | The anomaly detection job ID |
+| `enabled` | boolean | yes | Enable or disable delayed data checks |
+| `check_window` | string | no | Time window to scan for delayed data, e.g., `2h`; omitted from the request when empty or whitespace-only |
+
+**API call:**
+
+```text
+Step 1: data.parseJson — build JSON (enabled as boolean; check_window only when non-blank after trim)
+Step 2: POST _ml/anomaly_detectors/{job_id}/_update
+Body (example with window): {"analysis_config": {"delayed_data_check_config": {"enabled": true, "check_window": "2h"}}}
+Body (example without window): {"analysis_config": {"delayed_data_check_config": {"enabled": true}}}
+```
+
+**When to use:** Disable if delayed data checks are generating false positives. Increase `check_window` if late-arriving
+data is coming in very late (>1h after event time).
+
+---
+
+## Memory Management
+
+### `ad_estimate_memory_requirement`
+
+Compute a principled `model_memory_limit` estimate using the same algorithm Elasticsearch uses internally.
+
+| Parameter | Type | Required | Description |
+| -------------- | ------ | -------- | ------------------------------------------------------------------------- |
+| `job_id` | string | yes | The anomaly detection job ID |
+| `sample_start` | string | no | Start of cardinality sampling period (ISO 8601). Defaults to 30 days ago. |
+| `sample_end` | string | no | End of cardinality sampling period (ISO 8601). Defaults to now. |
+
+**API calls (7 steps):**
+
+```text
+1. GET _ml/anomaly_detectors/{job_id} ← get config
+2. [identify cardinality fields]
+3. POST {indices}/_search?size=0 ← overall_cardinality (aggs)
+4. POST {indices}/_search?size=0 ← max_bucket_cardinality (date_histogram)
+5. POST _ml/anomaly_detectors/_estimate_model_memory ← official estimate
+6. [compare estimate vs current limit vs actual usage]
+7. [produce recommendation]
+```
+
+**Returns:** Estimated memory requirement, comparison against current `model_memory_limit` and `peak_model_bytes`, and a
+recommendation (increase / current is appropriate / reduce data).
+
+**When to use:** Before increasing `model_memory_limit`. Much more accurate than `peak_model_bytes * 1.3` because it
+uses the actual cardinality of your data.
+
+---
+
+### `ad_wf_ts_field_cardinality`
+
+Approximate **distinct value count** for a split field (`partition_field`, `by_field`, or `over_field`) in **source**
+data over a time window via ES|QL `POST /_query`. `source_index`, `start_time`, and `end_time` are sent as **named
+literal** `params` for `?` placeholders; `split_field_esql` is **interpolated** into the query as the `COUNT_DISTINCT`
+column (for example `service.keyword` or `` `host.name.keyword` `` from `ad_get_job_datafeed_config`). If source
+cardinality is much larger than `total_*_count` from `ad_ts_model_memory_health`, entities may be dropped. For CCS, run
+per cluster and sum. Prefer `ad_estimate_memory_requirement` for full sizing.
+
+#### Parameters
+
+- `source_index` (string, required): index name or LIKE pattern from the datafeed config (same filter as
+ `FROM * METADATA _index`).
+- `split_field_esql` (string, required): valid ES|QL column expression for `COUNT_DISTINCT` — not a string literal; use
+ values from job analysis config only.
+- `start_time` (string, required): ISO 8601 start of window.
+- `end_time` (string, required): ISO 8601 end of window.
+
+**API call:**
+
+```text
+POST /_query
+Body: { "query": "... COUNT_DISTINCT() ...", "params": { "source_index": "...", "start_time": "...", "end_time": "..." } }
+```
+
+**When to use:** After `ad_ts_model_memory_health` shows memory pressure or `total_*_count` lower than expected — check
+whether the source still has more distinct split values than the model retains.
+
+---
+
+### `ad_update_model_memory_limit`
+
+Update the `model_memory_limit` on a job.
+
+| Parameter | Type | Required | Description |
+| ----------- | ------ | -------- | ------------------------------------- |
+| `job_id` | string | yes | The anomaly detection job ID |
+| `new_limit` | string | yes | New limit value, e.g., `256mb`, `1gb` |
+
+**API call:**
+
+```text
+POST _ml/anomaly_detectors/{job_id}/_update
+Body: {"analysis_limits": {"model_memory_limit": "{new_limit}"}}
+```
+
+**Prerequisites:** Job must be closed first. Full remediation sequence:
+
+1. `ad_manage_datafeed` with `_stop`
+2. `POST _ml/anomaly_detectors/{job_id}/_close`
+3. `ad_update_model_memory_limit`
+4. `POST _ml/anomaly_detectors/{job_id}/_open`
+5. `ad_manage_datafeed` with `_start`
+
+**Constraint:** Cannot decrease below current `model_bytes`. To shrink, clone the job with a lower limit.
+
+---
+
+## Model Snapshots
+
+### `ad_revert_model_snapshot`
+
+Revert a job's model to a previous snapshot to "unlearn" bad data.
+
+| Parameter | Type | Required | Description |
+| ------------- | ------ | -------- | ------------------------------------------------------------ |
+| `job_id` | string | yes | The anomaly detection job ID |
+| `snapshot_id` | string | yes | The snapshot ID to revert to (from `ad_get_model_snapshots`) |
+
+**API call:**
+
+```text
+POST _ml/anomaly_detectors/{job_id}/model_snapshots/{snapshot_id}/_revert
+```
+
+**Prerequisites:** Job must be closed before reverting.
+
+**Post-revert steps:**
+
+1. Reopen the job: `POST _ml/anomaly_detectors/{job_id}/_open`
+2. Restart datafeed from the snapshot timestamp to reprocess data (prevents gap in analysis)
+
+**When to use:** When bad/anomalous training data has corrupted the model — e.g., a 2-week outage that the model
+"learned" as normal. Revert to a snapshot from before the bad data, then reprocess.
+
+---
+
+## Calendar Management
+
+### `ad_get_calendar_events`
+
+Retrieve scheduled events from a calendar.
+
+| Parameter | Type | Required | Description |
+| ------------- | ------ | -------- | --------------- |
+| `calendar_id` | string | yes | The calendar ID |
+
+**API call:**
+
+```text
+GET _ml/calendars/{calendar_id}/events
+```
+
+**Returns:** List of events with `description`, `start_time`, `end_time`.
+
+**When to use:** Before adding a new event, verify what's already scheduled. Also use to audit whether a score anomaly
+was suppressed by an existing calendar event.
+
+---
+
+### `ad_create_calendar_event`
+
+Add a scheduled event to suppress false positives during known downtime.
+
+| Parameter | Type | Required | Description |
+| ------------- | ------ | -------- | -------------------------------------------------------------- |
+| `calendar_id` | string | yes | The calendar ID (must already exist) |
+| `description` | string | yes | Human-readable description, e.g., `Planned maintenance window` |
+| `start_time` | string | yes | ISO 8601 or epoch_millis |
+| `end_time` | string | yes | ISO 8601 or epoch_millis |
+
+**API call:**
+
+```text
+POST _ml/calendars/{calendar_id}/events
+Body: {"events": [{"description": "...", "start_time": "...", "end_time": "..."}]}
+```
+
+**Effect:** During the event window, the ML model does not produce anomaly results — it continues learning but
+suppresses output. Results resume normally after the window ends.
+
+**When to use:** Planned maintenance, deployments, known data pipeline downtime, holidays with predictable traffic
+changes.
+
+---
+
+## Job Creation
+
+Workflows `ad_validate_job_spec`, `ad_create_job`, and `ad_create_datafeed` accept JSON **text** inputs (`job_body`,
+`datafeed_body`). Each step uses Liquid `json_parse` with typed `${{ }}` interpolation so Elasticsearch receives a
+structured JSON object in the HTTP body, not a JSON value that is itself a quoted string.
+
+### `ad_validate_job_spec`
+
+Validate a job configuration before create.
+
+| Parameter | Type | Required | Description |
+| ---------- | ------ | -------- | --------------------------------------------------------------------------- |
+| `job_body` | string | yes | Full job JSON text (PUT create shape); parsed to an object in the workflow. |
+
+**API call:**
+
+```text
+POST _ml/anomaly_detectors/_validate
+Body: {full job configuration JSON document}
+```
+
+### `ad_create_job`
+
+Create a new anomaly detection job from a configuration.
+
+| Parameter | Type | Required | Description |
+| ---------- | ------ | -------- | -------------------------------------------------------- |
+| `job_id` | string | yes | The new job ID to create |
+| `job_body` | string | yes | Full job JSON text; parsed to an object in the workflow. |
+
+**API call:**
+
+```text
+PUT _ml/anomaly_detectors/{job_id}
+Body: {full job configuration JSON document}
+```
+
+**When to use:** Advanced scenario — when the agent has explored the data and designed a job configuration. Requires a
+complete `analysis_config` including detectors and `data_description`; create the datafeed with `ad_create_datafeed`
+before open/start.
+
+### `ad_create_datafeed`
+
+Create or replace a datafeed.
+
+| Parameter | Type | Required | Description |
+| --------------- | ------ | -------- | ------------------------------------------------------------- |
+| `datafeed_id` | string | yes | Typically `datafeed-{job_id}` |
+| `datafeed_body` | string | yes | Full datafeed JSON text; parsed to an object in the workflow. |
+
+**API call:**
+
+```text
+PUT _ml/datafeeds/{datafeed_id}
+Body: {full datafeed configuration JSON document}
+```
+
+### `ad_open_job`
+
+Open a job after configuration exists.
+
+| Parameter | Type | Required | Description |
+| --------- | ------ | -------- | -------------- |
+| `job_id` | string | yes | Job ID to open |
+
+**API call:**
+
+```text
+POST _ml/anomaly_detectors/{job_id}/_open
+```
+
+---
+
+## Permissions & Diagnostics
+
+### `ad_validate_ml_tool_permissions`
+
+Preflight check for core `.ml-*` indices used by packaged tools (see workflow YAML for exact `_has_privileges` payload).
+Does **not** cover job-specific source indices — validate those separately.
+
+| Parameter | Type | Required | Description |
+| --------- | ---- | -------- | ----------- |
+| _(none)_ | — | — | — |
+
+**API calls:**
+
+```text
+GET _security/_authenticate ← identify current user/API key
+POST _security/user/_has_privileges ← check index permissions
+Body: {
+ "index": [
+ {"names": [".ml-anomalies-*"], "privileges": ["read", "view_index_metadata"]},
+ {"names": [".ml-config"], "privileges": ["read", "view_index_metadata"]},
+ {"names": [".ml-annotations-*"], "privileges": ["read", "view_index_metadata"]},
+ {"names": [".ml-notifications-*"], "privileges": ["read", "view_index_metadata"]}
+ ]
+}
+```
+
+**Returns:** Current identity and a boolean per index/privilege combination.
+
+**When to use:** Early sanity check for `.ml-*` access; still verify source index privileges before datafeed preview or
+evidence queries.
+
+---
+
+### `ad_ts_ccs_diagnostics`
+
+Diagnose cross-cluster search (CCS) issues for datafeeds querying remote clusters.
+
+| Parameter | Type | Required | Description |
+| --------- | ---- | -------- | ----------- |
+| _(none)_ | — | — | — |
+
+**API calls:**
+
+```text
+GET _remote/info ← list configured remote clusters and connection status
+GET _cluster/health ← check local cluster health
+```
+
+**Returns:** Remote cluster connectivity status (connected/disconnected), latency, and error rates.
+
+**When to use:** When datafeed source indices use CCS patterns (e.g., `remote1:logs-*`) and the job reports missing
+documents or delayed data that can't be explained by local ingest latency.
+
+---
+
+## Troubleshooting Workflows (Decision Trees)
+
+### `ad_wf_troubleshoot_anomaly_score`
+
+Branching decision tree for unexpectedly high or low anomaly scores.
+
+| Parameter | Type | Required | Description |
+| ------------------ | ------ | -------- | --------------------------------------------------------- |
+| `job_id` | string | yes | The anomaly detection job ID |
+| `record_timestamp` | string | no | ISO 8601 timestamp of the specific anomaly to investigate |
+| `entity_value` | string | no | Entity value to focus investigation on |
+
+**Decision tree steps:**
+
+1. **Gate checks**: sufficient training data? memory status ok? delayed data present?
+2. **Score comparison**: compare `record_score` vs `initial_record_score` — large gap = renormalization
+3. **Job config analysis**: `bucket_span`, detector function, `custom_rules`, `use_null`
+4. **Model learning**: model plot (wide bounds = high variance), score reassessment for drift
+5. **Score factor education**: explain each `anomaly_score_explanation` component
+
+**Trigger phrases:** "Why is my score low?", "Expected anomaly not detected", "Score too high"
+
+---
+
+### `ad_wf_troubleshoot_memory_limit`
+
+Branching decision tree for `soft_limit` / `hard_limit` memory issues.
+
+| Parameter | Type | Required | Description |
+| --------- | ------ | -------- | -------------------------------------------- |
+| `job_id` | string | yes | The anomaly detection job ID to troubleshoot |
+
+**Decision tree steps:**
+
+1. **Memory status**: `model_bytes`, limit, `memory_status`, entity counts, allocation failures
+2. **False alarm check**: if `hard_limit`, check for missing-doc warnings that are symptoms of memory (not ingest lag)
+3. **Growth trend**: classify as stable plateau / linear growth / exponential growth
+4. **Threshold inspection**: `total_by_field_count > 100K`? `total_partition_field_count > 10K`? Identify which field
+ drives memory.
+5. **Principled estimate**: `ad_estimate_memory_requirement` — compare vs current limit vs actual usage
+6. **CCS check**: if datafeed uses CCS, account for cross-cluster cardinality
+7. **Recommendation**: increase limit / reduce data (filter datafeed, exclude datasets, reduce influencers) /
+ restructure job
+
+**Trigger phrases:** "My job hit memory limit", "hard_limit", "soft_limit"
+
+---
+
+### `ad_wf_troubleshoot_query_delay`
+
+Branching decision tree for missing documents and query_delay warnings.
+
+| Parameter | Type | Required | Description |
+| --------- | ------ | -------- | -------------------------------------------- |
+| `job_id` | string | yes | The anomaly detection job ID to troubleshoot |
+
+**Decision tree steps:**
+
+1. **Hard_limit false alarm**: if job has `categorization_field_name` AND `memory_status == hard_limit` → fix memory
+ first (missing docs = symptom, not cause)
+2. **Delayed data annotations**: frequency, severity, affected time ranges
+3. **Bucket event gaps**: buckets with zero or abnormally low event counts
+4. **Ingest latency measurement**: if `event.ingested` available, use it; otherwise use date_histogram comparison
+ fallback
+5. **Recommend `query_delay`**: P95 latency + buffer; trade-off: larger delay = slower alerts
+6. **Additional remediation**: add ingest pipeline timestamp, consider `event.ingested` as `time_field`, revert model
+ snapshot if catastrophically late data
+
+**Trigger phrases:** "Missing documents", "Datafeed has missed X documents", "query_delay"
diff --git a/skills/kibana/kibana-anomaly-detection/scripts/agent_builder_constants.json b/skills/kibana/kibana-anomaly-detection/scripts/agent_builder_constants.json
new file mode 100644
index 0000000..bf490f5
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/scripts/agent_builder_constants.json
@@ -0,0 +1,33 @@
+{
+ "default_agent_id": "elastic-ai-agent",
+ "default_agent_enable_elastic_capabilities": true,
+ "workflow_tool_exclusions": [
+ "ad_get_job_datafeed_config",
+ "ad_ts_ccs_diagnostics",
+ "ad_get_calendar_events",
+ "ad_create_calendar_event",
+ "ad_discover_jobs_by_datafeed_index"
+ ],
+ "fallback_tools": {
+ "ad_get_index_mappings": "platform.core.get_index_mapping"
+ },
+ "workflow_prefixes": [
+ "ad_wf_",
+ "ad_create_",
+ "ad_manage_",
+ "ad_open_",
+ "ad_update_",
+ "ad_revert_",
+ "ad_preview_",
+ "ad_validate_",
+ "ad_estimate_"
+ ],
+ "builtin_tools": [
+ "platform.core.search",
+ "platform.core.list_indices",
+ "platform.core.get_index_mapping",
+ "platform.core.execute_esql",
+ "platform.core.generate_esql",
+ "platform.core.product_documentation"
+ ]
+}
diff --git a/skills/kibana/kibana-anomaly-detection/scripts/kibana-agent-builder.mjs b/skills/kibana/kibana-anomaly-detection/scripts/kibana-agent-builder.mjs
new file mode 100755
index 0000000..8170482
--- /dev/null
+++ b/skills/kibana/kibana-anomaly-detection/scripts/kibana-agent-builder.mjs
@@ -0,0 +1,1481 @@
+#!/usr/bin/env node
+/**
+ * Kibana Agent Builder helpers for anomaly-detection — connection pattern aligned with
+ * elastic/agent-skills `kibana-dashboards.js`.
+ *
+ * Requires Node.js 18+ (global fetch). Optional `node:undici` / `undici` for TLS bypass without
+ * mutating process-wide NODE_TLS_REJECT_UNAUTHORIZED when available.
+ *
+ * Usage (run from repo root; script lives under skills/kibana/kibana-anomaly-detection/scripts/):
+ * node skills/kibana/kibana-anomaly-detection/scripts/kibana-agent-builder.mjs test
+ * node skills/kibana/kibana-anomaly-detection/scripts/kibana-agent-builder.mjs tools register [--dry-run]
+ * node skills/kibana/kibana-anomaly-detection/scripts/kibana-agent-builder.mjs skills register [--dry-run]
+ * node skills/kibana/kibana-anomaly-detection/scripts/kibana-agent-builder.mjs all register [--dry-run] # tools → workflows → skills → merge skill_ids on elastic-ai-agent
+ *
+ * Kibana compatibility: targets **9.4+** Agent Builder / Workflows. Tool and skill updates use PUT when the API
+ * supports it (9.5+); if PUT is missing or returns 404/405, registration falls back to DELETE + POST so idempotent
+ * runs still succeed on older minors.
+ *
+ * Environment variables (same as kibana-dashboards.js):
+ * KIBANA_URL, KIBANA_CLOUD_ID / ELASTICSEARCH_CLOUD_ID,
+ * KIBANA_USERNAME / ELASTICSEARCH_USERNAME,
+ * KIBANA_PASSWORD / ELASTICSEARCH_PASSWORD,
+ * KIBANA_API_KEY / ELASTICSEARCH_API_KEY,
+ * KIBANA_SPACE_ID, KIBANA_INSECURE
+ */
+
+import { readFileSync, readdirSync, existsSync } from "fs";
+import { join, dirname, basename } from "path";
+import { fileURLToPath } from "url";
+import { createRequire } from "module";
+
+const require = createRequire(import.meta.url);
+
+const __dirname = dirname(fileURLToPath(import.meta.url));
+const KIBANA_REF_DIR = join(__dirname, "..", "references", "kibana");
+const AGENT_DIR = join(KIBANA_REF_DIR, "agent");
+const TOOLS_DIR = join(KIBANA_REF_DIR, "tools");
+const WORKFLOWS_DIR = join(KIBANA_REF_DIR, "workflows");
+const PLUGIN_ROOT = join(__dirname, "..");
+const SKILLS_DIR = join(PLUGIN_ROOT, "skills");
+const CONSTANTS_PATH = join(__dirname, "agent_builder_constants.json");
+
+const CONSTANTS = JSON.parse(readFileSync(CONSTANTS_PATH, "utf8"));
+const DEFAULT_AGENT_ID = CONSTANTS.default_agent_id ?? "elastic-ai-agent";
+/** When true (default), PUT sets configuration.enable_elastic_capabilities so bundle skills are active in the Elastic AI Agent. Set false in agent_builder_constants.json to skip. */
+const DEFAULT_AGENT_ENABLE_ELASTIC_CAPABILITIES = CONSTANTS.default_agent_enable_elastic_capabilities !== false;
+const WORKFLOW_TOOL_EXCLUSIONS = new Set(CONSTANTS.workflow_tool_exclusions);
+const WORKFLOW_PREFIXES = CONSTANTS.workflow_prefixes;
+const BUILTIN_TOOLS = CONSTANTS.builtin_tools;
+const FALLBACK_TOOLS = CONSTANTS.fallback_tools || {};
+const BUILTIN_TOOLS_SET = new Set(BUILTIN_TOOLS);
+
+/** Kibana Agent Builder rejects skills with more than this many tool_ids. */
+const MAX_SKILL_TOOL_IDS = 5;
+
+let kibanaFetchImpl = globalThis.fetch.bind(globalThis);
+let insecureDispatcher = null;
+try {
+ const u = require("node:undici");
+ kibanaFetchImpl = u.fetch.bind(u);
+ insecureDispatcher = new u.Agent({ connect: { rejectUnauthorized: false } });
+} catch {
+ try {
+ const u = require("undici");
+ kibanaFetchImpl = u.fetch.bind(u);
+ insecureDispatcher = new u.Agent({ connect: { rejectUnauthorized: false } });
+ } catch {
+ insecureDispatcher = null;
+ }
+}
+
+/**
+ * HTTP fetch for Kibana: uses undici Agent for insecure TLS when possible (no global env toggle).
+ */
+async function kibanaHttpFetch(url, config, init = {}) {
+ const merged = {
+ ...init,
+ headers: { ...init.headers },
+ };
+
+ if (config.insecure && insecureDispatcher) {
+ merged.dispatcher = insecureDispatcher;
+ return kibanaFetchImpl(url, merged);
+ }
+
+ if (config.insecure && !insecureDispatcher) {
+ const prev = process.env.NODE_TLS_REJECT_UNAUTHORIZED;
+ try {
+ process.env.NODE_TLS_REJECT_UNAUTHORIZED = "0";
+ return await globalThis.fetch(url, merged);
+ } finally {
+ if (prev === undefined) {
+ delete process.env.NODE_TLS_REJECT_UNAUTHORIZED;
+ } else {
+ process.env.NODE_TLS_REJECT_UNAUTHORIZED = prev;
+ }
+ }
+ }
+
+ return globalThis.fetch(url, merged);
+}
+
+/**
+ * Read JSON from a fetch Response body; keeps raw text if JSON.parse fails.
+ */
+async function readJsonBody(res) {
+ const text = await res.text();
+ if (!text?.trim()) return { text: "", parsed: null };
+ try {
+ return { text, parsed: JSON.parse(text) };
+ } catch {
+ return { text, parsed: null };
+ }
+}
+
+// -----------------------------------------------------------------------------
+// Kibana client (aligned with kibana-dashboards.js)
+// -----------------------------------------------------------------------------
+
+function kibanaUrlFromCloudId(cloudId) {
+ try {
+ const parts = cloudId.split(":");
+ if (parts.length !== 2) return null;
+ const decoded = Buffer.from(parts[1], "base64").toString("utf8");
+ const decodedParts = decoded.split("$");
+ if (decodedParts.length < 3 || !decodedParts[2]) return null;
+ const domain = decodedParts[0];
+ const kibanaUuid = decodedParts[2];
+ let host = domain;
+ let port = "";
+ if (domain.includes(":")) {
+ const splitDomain = domain.split(":");
+ host = splitDomain[0];
+ port = `:${splitDomain[1]}`;
+ } else {
+ port = ":443";
+ }
+ return `https://${kibanaUuid}.${host}${port}`;
+ } catch {
+ return null;
+ }
+}
+
+function resolveApiKey(cli) {
+ if (cli.apiKeyFromCli) {
+ const t = (cli.apiKey ?? "").trim();
+ return { apiKey: t || undefined, apiKeyCliEmpty: !t };
+ }
+ for (const name of ["KIBANA_API_KEY", "ELASTICSEARCH_API_KEY"]) {
+ const raw = process.env[name];
+ if (raw == null) continue;
+ const t = String(raw).trim();
+ if (t) return { apiKey: t, apiKeyCliEmpty: false };
+ }
+ return { apiKey: undefined, apiKeyCliEmpty: false };
+}
+
+const DEFAULT_KIBANA_URL = "http://localhost:5601";
+
+export function getKibanaConfig(cli = {}) {
+ const cloudId = process.env.KIBANA_CLOUD_ID || process.env.ELASTICSEARCH_CLOUD_ID;
+ let url = cli.kibanaUrl || process.env.KIBANA_URL;
+
+ if (!url && cloudId) {
+ url = kibanaUrlFromCloudId(cloudId);
+ }
+
+ const usingDefaults = !url && !cloudId;
+ if (usingDefaults) url = DEFAULT_KIBANA_URL;
+
+ const { apiKey, apiKeyCliEmpty } = resolveApiKey(cli);
+ const username =
+ cli.username ??
+ process.env.KIBANA_USERNAME ??
+ process.env.ELASTICSEARCH_USERNAME ??
+ (apiKey ? undefined : "elastic");
+ const password =
+ cli.password ??
+ process.env.KIBANA_PASSWORD ??
+ process.env.ELASTICSEARCH_PASSWORD ??
+ (apiKey ? undefined : "changeme");
+ let spaceId = cli.spaceId ?? process.env.KIBANA_SPACE_ID;
+ if (spaceId === "default") spaceId = undefined;
+
+ const insecureFlag =
+ cli.insecure === true || ["1", "true", "yes"].includes((process.env.KIBANA_INSECURE || "").toLowerCase());
+
+ return {
+ url,
+ username,
+ password,
+ apiKey,
+ apiKeyCliEmpty,
+ spaceId,
+ insecure: insecureFlag,
+ usingDefaults,
+ };
+}
+
+function warnUsingDefaults() {
+ console.log(`No Kibana URL configured — using default: ${DEFAULT_KIBANA_URL} (elastic/changeme)`);
+ console.log("");
+ console.log("If you want to override Kibana configuration, you can set one of:");
+ console.log(" 1. Elastic Cloud: KIBANA_CLOUD_ID + KIBANA_API_KEY");
+ console.log(" 2. URL + API Key: KIBANA_URL + KIBANA_API_KEY");
+ console.log(" 3. Basic Auth: KIBANA_URL + KIBANA_USERNAME + KIBANA_PASSWORD");
+ console.log(" 4. CLI flags: --kibana-url, --username, --password, --api-key");
+ console.log("");
+}
+
+function getHeaders(config) {
+ const headers = {
+ "Content-Type": "application/json",
+ "kbn-xsrf": "true",
+ "x-elastic-internal-origin": "kibana",
+ "User-Agent": "elastic-agentic-anomaly-detection",
+ };
+
+ if (config.apiKey) {
+ headers.Authorization = `ApiKey ${config.apiKey}`;
+ } else if (config.username && config.password) {
+ const auth = Buffer.from(`${config.username}:${config.password}`).toString("base64");
+ headers.Authorization = `Basic ${auth}`;
+ }
+
+ return headers;
+}
+
+function getBasePath(config) {
+ let basePath = (config.url || "").replace(/\/$/, "");
+ if (config.spaceId && config.spaceId !== "default") {
+ basePath += `/s/${config.spaceId}`;
+ }
+ return basePath;
+}
+
+function validateConfig(config, { dryRun }) {
+ if (config.apiKeyCliEmpty) {
+ console.error("Error: --api-key cannot be empty or whitespace-only.");
+ return false;
+ }
+ if (config.apiKey) return true;
+ if (config.username && config.password) return true;
+ if (dryRun) return true;
+ console.error("Error: Authentication required (API key or username + password).");
+ return false;
+}
+
+async function kibanaFetch(config, path, options = {}) {
+ const basePath = getBasePath(config);
+ const url = `${basePath}${path}`;
+
+ const fetchOptions = {
+ ...options,
+ headers: {
+ ...getHeaders(config),
+ ...options.headers,
+ },
+ };
+
+ const response = await kibanaHttpFetch(url, config, fetchOptions);
+ const contentType = response.headers.get("content-type");
+ let data;
+ if (contentType && contentType.includes("application/json")) {
+ data = await response.json();
+ } else {
+ data = await response.text();
+ }
+
+ return { ok: response.ok, status: response.status, data };
+}
+
+async function kibanaEsRequest(config, method, esPath, body) {
+ const query = new URLSearchParams({
+ path: esPath,
+ method: method.toUpperCase(),
+ }).toString();
+
+ const options = { method: "POST" };
+ if (body !== undefined) {
+ options.body = JSON.stringify(body);
+ }
+
+ return kibanaFetch(config, `/api/console/proxy?${query}`, options);
+}
+
+// -----------------------------------------------------------------------------
+// Tool registration
+// -----------------------------------------------------------------------------
+
+function loadToolDefs() {
+ const tools = [];
+ for (const subdir of ["esql"]) {
+ const dir = join(TOOLS_DIR, subdir);
+ if (!existsSync(dir)) continue;
+ for (const name of readdirSync(dir).sort()) {
+ if (!name.endsWith(".json")) continue;
+ try {
+ const def = JSON.parse(readFileSync(join(dir, name), "utf8"));
+ def._sourceFile = `${subdir}/${name}`;
+ tools.push(def);
+ } catch (e) {
+ console.warn(`Skipping ${subdir}/${name}: ${e.message}`);
+ }
+ }
+ }
+ return tools;
+}
+
+function transformToolDef(def) {
+ const { _sourceFile: _, ...rest } = def;
+ const payload = {
+ id: rest.name || rest.id || "unnamed",
+ type: rest.type || "esql",
+ description: rest.description || "",
+ };
+ if (rest.tags) payload.tags = rest.tags;
+ const config = { ...(rest.configuration || {}) };
+ if (payload.type === "esql") config.params = rest.parameters || {};
+ payload.configuration = config;
+ return payload;
+}
+
+/**
+ * True when PUT failed because the update route/method is unavailable (older Kibana), not request validation.
+ * In that case we fall back to DELETE + POST with the full create payload.
+ */
+function agentBuilderPutUnsupported(status, errorText) {
+ const t = (errorText || "").slice(0, 800);
+ if (status === 404 || status === 405 || status === 501) return true;
+ if (status === 400 && /\b(route not found|method not allowed|no handler|cannot\s+(PUT|patch))\b/i.test(t)) {
+ return true;
+ }
+ return false;
+}
+
+function sleep(ms) {
+ return new Promise((resolve) => setTimeout(resolve, ms));
+}
+
+/**
+ * Retry Kibana Agent Builder writes when SNAPSHOT / testcontainers stacks return transient errors during bulk
+ * registration (common after many consecutive tool POSTs).
+ */
+async function kibanaHttpFetchWithRetry(url, config, init, { attempts = 5 } = {}) {
+ let res;
+ for (let attempt = 0; attempt < attempts; attempt++) {
+ res = await kibanaHttpFetch(url, config, init);
+ if (res.ok) return res;
+ if (![408, 429, 502, 503, 504].includes(res.status)) return res;
+ try {
+ await res.text();
+ } catch {
+ /* ignore body read errors */
+ }
+ const backoff = 150 * 2 ** attempt + Math.floor(Math.random() * 100);
+ await sleep(backoff);
+ }
+ return res;
+}
+
+async function registerTool(config, def, dryRun) {
+ const payload = transformToolDef(def);
+ console.log(`Tool: ${payload.id} (${def._sourceFile})`);
+
+ if (dryRun) {
+ console.log(JSON.stringify(payload, null, 2));
+ return true;
+ }
+
+ const basePath = getBasePath(config);
+ const toolsUrl = `${basePath}/api/agent_builder/tools`;
+ const toolByIdUrl = `${toolsUrl}/${encodeURIComponent(payload.id)}`;
+ const headers = { ...getHeaders(config), "kbn-xsrf": "true" };
+
+ /** PUT body — id comes from path; `type` is immutable (POST-only). */
+ const updateBody = {
+ description: payload.description,
+ configuration: payload.configuration,
+ };
+ if (payload.tags) updateBody.tags = payload.tags;
+
+ let res = await kibanaHttpFetchWithRetry(
+ toolsUrl,
+ config,
+ {
+ method: "POST",
+ headers,
+ body: JSON.stringify(payload),
+ },
+ { attempts: 5 },
+ );
+
+ if (res.ok) {
+ console.log(` Registered: ${payload.id}`);
+ return true;
+ }
+
+ const text = await res.text();
+ const alreadyExists = res.status === 409 || (res.status === 400 && /already exists|duplicate/i.test(text));
+
+ if (alreadyExists) {
+ console.log(` Already exists — updating (PUT): ${payload.id}`);
+ const putRes = await kibanaHttpFetchWithRetry(
+ toolByIdUrl,
+ config,
+ {
+ method: "PUT",
+ headers,
+ body: JSON.stringify(updateBody),
+ },
+ { attempts: 5 },
+ );
+ if (putRes.ok) {
+ console.log(` Updated: ${payload.id}`);
+ return true;
+ }
+ const putText = await putRes.text();
+ if (agentBuilderPutUnsupported(putRes.status, putText)) {
+ console.log(` PUT unsupported — deleting and re-creating: ${payload.id}`);
+ await kibanaHttpFetchWithRetry(toolByIdUrl, config, { method: "DELETE", headers }, { attempts: 3 });
+ const recRes = await kibanaHttpFetchWithRetry(
+ toolsUrl,
+ config,
+ {
+ method: "POST",
+ headers,
+ body: JSON.stringify(payload),
+ },
+ { attempts: 5 },
+ );
+ if (recRes.ok) {
+ console.log(` Re-created: ${payload.id}`);
+ return true;
+ }
+ const recText = await recRes.text();
+ console.error(` Failed: ${recRes.status} ${recText.slice(0, 200)}`);
+ return false;
+ }
+ console.error(` Failed to update tool: ${putRes.status} ${putText.slice(0, 200)}`);
+ return false;
+ }
+
+ console.error(` Failed: ${res.status} ${text.slice(0, 200)}`);
+ return false;
+}
+
+async function cmdToolsRegister(argv) {
+ let dryRun = false;
+ const cli = {};
+ for (let i = 0; i < argv.length; i++) {
+ const a = argv[i];
+ if (a === "--dry-run") dryRun = true;
+ else if (a === "--kibana-url") cli.kibanaUrl = argv[++i];
+ else if (a === "--username") cli.username = argv[++i];
+ else if (a === "--password") cli.password = argv[++i];
+ else if (a === "--api-key") {
+ cli.apiKeyFromCli = true;
+ cli.apiKey = argv[++i];
+ } else if (a === "--space-id") cli.spaceId = argv[++i];
+ else if (a === "--insecure") cli.insecure = true;
+ }
+
+ const config = getKibanaConfig(cli);
+ if (!validateConfig(config, { dryRun })) process.exit(1);
+ if (config.usingDefaults && !dryRun) warnUsingDefaults();
+
+ const defs = loadToolDefs();
+ console.log(`Loaded ${defs.length} tool definitions`);
+
+ if (!dryRun) {
+ const st = await kibanaFetch(config, "/api/status");
+ if (!st.ok) {
+ console.error("Cannot reach Kibana:", st.status, st.data);
+ process.exit(1);
+ }
+ console.log("Connected to Kibana", st.data?.version?.number || "?");
+ }
+
+ let succeeded = 0,
+ failed = 0,
+ skipped = 0;
+ for (const def of defs) {
+ const primaryBuiltin = FALLBACK_TOOLS[def.name];
+ if (primaryBuiltin && BUILTIN_TOOLS_SET.has(primaryBuiltin)) {
+ console.log(` Skipping fallback tool: ${def.name} (${primaryBuiltin} is available as a builtin)`);
+ skipped++;
+ continue;
+ }
+ const ok = await registerTool(config, def, dryRun);
+ ok ? succeeded++ : failed++;
+ if (!dryRun) await sleep(120);
+ }
+
+ if (!dryRun) {
+ console.log(`\nRegistration complete: ${succeeded} succeeded, ${failed} failed, ${skipped} skipped`);
+ }
+}
+
+// -----------------------------------------------------------------------------
+// ML job scaffolding
+// -----------------------------------------------------------------------------
+
+function parseJobsCreateServiceHealthArgs(argv) {
+ const cli = {};
+ const opts = {
+ prefix: "svc",
+ metricsIndex: "metrics-*",
+ logsIndex: "logs-*",
+ apmIndex: "apm-*",
+ bucketSpan: "15m",
+ queryDelay: "120s",
+ memoryLimit: "256mb",
+ dryRun: false,
+ };
+
+ for (let i = 0; i < argv.length; i++) {
+ const a = argv[i];
+ if (a === "--dry-run") opts.dryRun = true;
+ else if (a === "--prefix") opts.prefix = argv[++i];
+ else if (a === "--metrics-index") opts.metricsIndex = argv[++i];
+ else if (a === "--logs-index") opts.logsIndex = argv[++i];
+ else if (a === "--apm-index") opts.apmIndex = argv[++i];
+ else if (a === "--bucket-span") opts.bucketSpan = argv[++i];
+ else if (a === "--query-delay") opts.queryDelay = argv[++i];
+ else if (a === "--memory-limit") opts.memoryLimit = argv[++i];
+ else if (a === "--kibana-url") cli.kibanaUrl = argv[++i];
+ else if (a === "--username") cli.username = argv[++i];
+ else if (a === "--password") cli.password = argv[++i];
+ else if (a === "--api-key") {
+ cli.apiKeyFromCli = true;
+ cli.apiKey = argv[++i];
+ } else if (a === "--space-id") cli.spaceId = argv[++i];
+ else if (a === "--insecure") cli.insecure = true;
+ }
+
+ return { cli, opts };
+}
+
+function serviceHealthPlans(opts) {
+ const makePlan = (suffix, description, indices, datafeedQuery, detectors, influencers) => {
+ const jobId = `${opts.prefix}-${suffix}`;
+ const datafeedId = `datafeed-${jobId}`;
+ return {
+ jobId,
+ datafeedId,
+ jobBody: {
+ description,
+ analysis_config: {
+ bucket_span: opts.bucketSpan,
+ detectors,
+ influencers,
+ },
+ analysis_limits: {
+ model_memory_limit: opts.memoryLimit,
+ },
+ data_description: {
+ time_field: "@timestamp",
+ },
+ },
+ datafeedBody: {
+ datafeed_id: datafeedId,
+ job_id: jobId,
+ indices,
+ query_delay: opts.queryDelay,
+ scroll_size: 1000,
+ query: datafeedQuery,
+ },
+ };
+ };
+
+ const serviceInfluencers = ["service.name", "host.name", "kubernetes.pod.name"];
+
+ return [
+ makePlan(
+ "cpu-high-mean",
+ "Detect sustained CPU pressure per service from metrics data.",
+ [opts.metricsIndex],
+ {
+ bool: {
+ filter: [
+ { exists: { field: "@timestamp" } },
+ { exists: { field: "service.name" } },
+ { exists: { field: "system.cpu.total.norm.pct" } },
+ ],
+ },
+ },
+ [
+ {
+ function: "high_mean",
+ field_name: "system.cpu.total.norm.pct",
+ partition_field_name: "service.name",
+ },
+ ],
+ serviceInfluencers,
+ ),
+ makePlan(
+ "memory-high-mean",
+ "Detect sustained memory pressure per service from metrics data.",
+ [opts.metricsIndex],
+ {
+ bool: {
+ filter: [
+ { exists: { field: "@timestamp" } },
+ { exists: { field: "service.name" } },
+ { exists: { field: "system.memory.actual.used.pct" } },
+ ],
+ },
+ },
+ [
+ {
+ function: "high_mean",
+ field_name: "system.memory.actual.used.pct",
+ partition_field_name: "service.name",
+ },
+ ],
+ serviceInfluencers,
+ ),
+ makePlan(
+ "latency-high-mean",
+ "Detect service latency spikes from APM transactions.",
+ [opts.apmIndex],
+ {
+ bool: {
+ filter: [
+ { exists: { field: "@timestamp" } },
+ { exists: { field: "service.name" } },
+ { term: { "processor.event": "transaction" } },
+ { exists: { field: "transaction.duration.us" } },
+ ],
+ },
+ },
+ [
+ {
+ function: "high_mean",
+ field_name: "transaction.duration.us",
+ partition_field_name: "service.name",
+ },
+ ],
+ ["service.name", "transaction.type", "host.name"],
+ ),
+ makePlan(
+ "error-rate-high-count",
+ "Detect service-level error surges from logs and failed transactions.",
+ [opts.logsIndex, opts.apmIndex],
+ {
+ bool: {
+ filter: [{ exists: { field: "@timestamp" } }, { exists: { field: "service.name" } }],
+ should: [{ term: { "log.level": "error" } }, { term: { "event.outcome": "failure" } }],
+ minimum_should_match: 1,
+ },
+ },
+ [
+ {
+ function: "high_count",
+ partition_field_name: "service.name",
+ },
+ ],
+ ["service.name", "event.dataset", "host.name", "kubernetes.pod.name"],
+ ),
+ ];
+}
+
+async function upsertServiceHealthPlan(config, plan, dryRun) {
+ const requests = [
+ ["PUT", `/_ml/anomaly_detectors/${encodeURIComponent(plan.jobId)}`, plan.jobBody],
+ ["PUT", `/_ml/datafeeds/${encodeURIComponent(plan.datafeedId)}`, plan.datafeedBody],
+ ["POST", `/_ml/anomaly_detectors/${encodeURIComponent(plan.jobId)}/_open`],
+ ["POST", `/_ml/datafeeds/${encodeURIComponent(plan.datafeedId)}/_start`],
+ ];
+
+ console.log(`\nJob: ${plan.jobId}`);
+ if (dryRun) {
+ for (const [method, path, body] of requests) {
+ console.log(`${method} ${path}`);
+ if (body !== undefined) console.log(JSON.stringify(body, null, 2));
+ }
+ return true;
+ }
+
+ for (const [method, path, body] of requests) {
+ const res = await kibanaEsRequest(config, method, path, body);
+ if (res.ok) {
+ console.log(` ${method} ${path} -> ${res.status}`);
+ continue;
+ }
+ const text = typeof res.data === "string" ? res.data : JSON.stringify(res.data);
+ const alreadyExists =
+ res.status === 409 || (res.status === 400 && /already exists|resource_already_exists_exception/i.test(text));
+ const alreadyStarted =
+ res.status === 409 || (res.status === 400 && /already open|already started|is started/i.test(text));
+ if (alreadyExists || alreadyStarted) {
+ console.log(` ${method} ${path} -> exists/already started; continuing`);
+ continue;
+ }
+ console.error(` ${method} ${path} -> failed (${res.status}): ${text.slice(0, 350)}`);
+ return false;
+ }
+
+ return true;
+}
+
+async function cmdJobsCreateServiceHealth(argv) {
+ const { cli, opts } = parseJobsCreateServiceHealthArgs(argv);
+ const config = getKibanaConfig(cli);
+ if (!validateConfig(config, { dryRun: opts.dryRun })) process.exit(1);
+ if (config.usingDefaults && !opts.dryRun) warnUsingDefaults();
+
+ if (!opts.dryRun) {
+ const st = await kibanaFetch(config, "/api/status");
+ if (!st.ok) {
+ console.error("Cannot reach Kibana:", st.status, st.data);
+ process.exit(1);
+ }
+ console.log("Connected to Kibana", st.data?.version?.number || "?");
+ }
+
+ const plans = serviceHealthPlans(opts);
+ console.log(
+ `Creating ${plans.length} service-health jobs (prefix='${opts.prefix}', bucket_span='${opts.bucketSpan}', query_delay='${opts.queryDelay}')`,
+ );
+
+ let okCount = 0;
+ let failCount = 0;
+ for (const plan of plans) {
+ const ok = await upsertServiceHealthPlan(config, plan, opts.dryRun);
+ if (ok) okCount++;
+ else failCount++;
+ }
+
+ console.log(`\nDone: ${okCount} succeeded, ${failCount} failed`);
+ if (!opts.dryRun) {
+ console.log("Check jobs in Kibana: Stack Management > Machine Learning > Anomaly Detection Jobs");
+ }
+}
+
+// -----------------------------------------------------------------------------
+// Tool classification (skill tool_ids + agent JSON manifests)
+// -----------------------------------------------------------------------------
+
+function isWorkflowTool(toolId) {
+ if (WORKFLOW_TOOL_EXCLUSIONS.has(toolId)) return true;
+ if (toolId in FALLBACK_TOOLS) return true;
+ return WORKFLOW_PREFIXES.some((p) => toolId.startsWith(p));
+}
+
+/** ES|QL tool IDs that `tools register` would POST (same skip rules as cmdToolsRegister). */
+function getRegisterableEsqlToolIds() {
+ const ids = new Set();
+ for (const def of loadToolDefs()) {
+ const primaryBuiltin = FALLBACK_TOOLS[def.name];
+ if (primaryBuiltin && BUILTIN_TOOLS_SET.has(primaryBuiltin)) continue;
+ ids.add(transformToolDef(def).id);
+ }
+ return ids;
+}
+
+function loadAgentToolNames(jsonBasename) {
+ const path = join(AGENT_DIR, jsonBasename);
+ if (!existsSync(path)) return [];
+ try {
+ const data = JSON.parse(readFileSync(path, "utf8"));
+ return Array.isArray(data.tools) ? data.tools : [];
+ } catch {
+ return [];
+ }
+}
+
+function mergedCoreAgentToolNames() {
+ const a = loadAgentToolNames("anomaly_detective.json");
+ const b = loadAgentToolNames("anomaly_explainer.json");
+ const c = loadAgentToolNames("anomaly_maintainer.json");
+ return [...new Set([...a, ...b, ...c])];
+}
+
+/** Maps skill folder id → tool lists from references/kibana/agent/*.json (used only to derive skill tool_ids). */
+const SKILL_ID_AGENT_JSON = {
+ "investigate-anomaly": "anomaly_detective.json",
+ "explain-anomaly-results": "anomaly_explainer.json",
+ "troubleshoot-anomaly-detection-jobs": "anomaly_maintainer.json",
+ "manage-anomaly-detection-job": "anomaly_maintainer.json",
+};
+
+function toolNamesDeclaredForSkill(skillId) {
+ if (
+ skillId === "kibana-anomaly-detection" ||
+ skillId === "observability-anomaly-expert" ||
+ skillId === "security-anomaly-expert"
+ ) {
+ return mergedCoreAgentToolNames();
+ }
+ const agentJson = SKILL_ID_AGENT_JSON[skillId];
+ if (agentJson) return loadAgentToolNames(agentJson);
+ return [];
+}
+
+/**
+ * Full preference list: builtins (in order) then custom ES|QL tools from agent manifests that `tools register`
+ * created (workflow-backed tools omitted).
+ */
+function buildSkillToolIdsFallback(skillId, registeredEsqlIds, missingAccumulator) {
+ const declared = toolNamesDeclaredForSkill(skillId);
+ const nonWorkflow = declared.filter((t) => !isWorkflowTool(t));
+ const nonBuiltin = nonWorkflow.filter((t) => !BUILTIN_TOOLS_SET.has(t));
+ const attached = [];
+ for (const t of nonBuiltin) {
+ if (registeredEsqlIds.has(t)) attached.push(t);
+ else missingAccumulator.add(t);
+ }
+ return [...BUILTIN_TOOLS, ...attached];
+}
+
+const TOOL_ID_SCAN_RE = /\b(ad_[a-z][a-z0-9_]*)\b|\b(platform\.core\.[a-z0-9_]+)\b|\b(observability\.[a-z0-9_]+)\b/g;
+
+/**
+ * Tool ids appearing in *skillMarkdown* (order = first mention wins). Only includes ids Kibana can attach:
+ * builtins from this bundle, or registerable ES|QL tools (non-workflow).
+ */
+function extractMentionedToolIdsInOrder(skillMarkdown, registeredEsqlIds) {
+ const ordered = [];
+ const seen = new Set();
+ if (!skillMarkdown) return ordered;
+ let m;
+ TOOL_ID_SCAN_RE.lastIndex = 0;
+ while ((m = TOOL_ID_SCAN_RE.exec(skillMarkdown)) !== null) {
+ const id = m[1] || m[2] || m[3];
+ if (seen.has(id)) continue;
+ if (isWorkflowTool(id)) continue;
+ if (BUILTIN_TOOLS_SET.has(id)) {
+ seen.add(id);
+ ordered.push(id);
+ continue;
+ }
+ if (registeredEsqlIds.has(id)) {
+ seen.add(id);
+ ordered.push(id);
+ }
+ }
+ return ordered;
+}
+
+/**
+ * At most {@link MAX_SKILL_TOOL_IDS} tools: prefer those mentioned in the skill markdown (full SKILL.md text),
+ * then fill from the manifest-derived fallback list.
+ */
+function buildSkillToolIdsCapped(skillId, skillMarkdownScan, registeredEsqlIds, missingAccumulator) {
+ const mentioned = extractMentionedToolIdsInOrder(skillMarkdownScan, registeredEsqlIds);
+ const fallback = buildSkillToolIdsFallback(skillId, registeredEsqlIds, missingAccumulator);
+ const out = [];
+ const seen = new Set();
+ for (const id of mentioned) {
+ if (out.length >= MAX_SKILL_TOOL_IDS) break;
+ if (seen.has(id)) continue;
+ seen.add(id);
+ out.push(id);
+ }
+ for (const id of fallback) {
+ if (out.length >= MAX_SKILL_TOOL_IDS) break;
+ if (seen.has(id)) continue;
+ seen.add(id);
+ out.push(id);
+ }
+ return out;
+}
+
+function enrichSkillDefsWithToolIds(defs) {
+ const registered = getRegisterableEsqlToolIds();
+ const missing = new Set();
+ for (const def of defs) {
+ const scan = def._toolScanMarkdown ?? "";
+ delete def._toolScanMarkdown;
+ def.tool_ids = buildSkillToolIdsCapped(def.id, scan, registered, missing);
+ }
+ if (missing.size > 0) {
+ const sample = [...missing].sort().slice(0, 16).join(", ");
+ console.warn(
+ ` Note: ${missing.size} tool id(s) declared in agent JSON but not in ES|QL bundle (workflows / skipped fallbacks) — omitted from skill tool_ids: ${sample}${missing.size > 16 ? " …" : ""}`,
+ );
+ }
+ return defs;
+}
+
+async function cmdWorkflowsRegister(argv) {
+ let dryRun = false;
+ const cli = {};
+ for (let i = 0; i < argv.length; i++) {
+ const a = argv[i];
+ if (a === "--dry-run") dryRun = true;
+ else if (a === "--kibana-url") cli.kibanaUrl = argv[++i];
+ else if (a === "--username") cli.username = argv[++i];
+ else if (a === "--password") cli.password = argv[++i];
+ else if (a === "--api-key") {
+ cli.apiKeyFromCli = true;
+ cli.apiKey = argv[++i];
+ } else if (a === "--space-id") cli.spaceId = argv[++i];
+ else if (a === "--insecure") cli.insecure = true;
+ }
+
+ const config = getKibanaConfig(cli);
+ if (!validateConfig(config, { dryRun })) process.exit(1);
+ if (config.usingDefaults && !dryRun) warnUsingDefaults();
+
+ const files = readdirSync(WORKFLOWS_DIR)
+ .filter((f) => f.endsWith(".yaml") || f.endsWith(".yml"))
+ .sort();
+
+ if (files.length === 0) {
+ console.error("No YAML files found in", WORKFLOWS_DIR);
+ process.exit(1);
+ }
+
+ const workflows = files.map((f) => ({
+ _file: f,
+ yaml: readFileSync(join(WORKFLOWS_DIR, f), "utf8"),
+ }));
+
+ console.log(`Loaded ${workflows.length} workflow YAML files`);
+
+ if (dryRun) {
+ for (const { _file, yaml } of workflows) {
+ console.log(`\n--- ${_file} ---`);
+ console.log(yaml.slice(0, 200) + (yaml.length > 200 ? "\n ..." : ""));
+ }
+ return;
+ }
+
+ const st = await kibanaFetch(config, "/api/status");
+ if (!st.ok) {
+ console.error("Cannot reach Kibana:", st.status, st.data);
+ process.exit(1);
+ }
+ console.log("Connected to Kibana", st.data?.version?.number || "?");
+
+ const basePath = getBasePath(config);
+ const url = `${basePath}/api/workflows?overwrite=true`;
+ const payload = { workflows: workflows.map(({ yaml }) => ({ yaml })) };
+
+ const res = await kibanaHttpFetch(url, config, {
+ method: "POST",
+ headers: { ...getHeaders(config), "kbn-xsrf": "true" },
+ body: JSON.stringify(payload),
+ });
+
+ let body;
+ try {
+ body = await res.json();
+ } catch {
+ const text = await res.text().catch(() => "");
+ console.error(`Unexpected response ${res.status}: ${text.slice(0, 300)}`);
+ process.exit(1);
+ }
+
+ if (!res.ok) {
+ console.error(`Failed: ${res.status}`, JSON.stringify(body).slice(0, 400));
+ process.exit(1);
+ }
+
+ for (const { id, name } of body.created ?? []) {
+ console.log(` Registered: ${name} (${id})`);
+ }
+ for (const failure of body.failures ?? []) {
+ console.error(` Failed: ${JSON.stringify(failure)}`);
+ }
+
+ console.log(`\nDone: ${body.created?.length ?? 0} registered, ${body.failures?.length ?? 0} failed`);
+}
+
+async function cmdTest(config) {
+ if (!validateConfig(config, { dryRun: false })) process.exit(1);
+ if (config.usingDefaults) warnUsingDefaults();
+
+ const basePath = getBasePath(config);
+ const res = await kibanaHttpFetch(`${basePath}/api/status`, config, {
+ headers: { ...getHeaders(config), "kbn-xsrf": "true" },
+ });
+ const data = await res.json().catch(() => ({}));
+
+ if (!res.ok) {
+ console.error("Connection failed:", res.status, data);
+ process.exit(1);
+ }
+
+ const version = data.version?.number || "unknown";
+ console.log("Connected to Kibana", version);
+ console.log(" Base URL:", config.url);
+ console.log(" Space:", config.spaceId || "(default)");
+ console.log(" Auth:", config.apiKey ? "ApiKey" : "Basic");
+}
+
+// -----------------------------------------------------------------------------
+// Skill registration
+// -----------------------------------------------------------------------------
+
+function parseSkillFrontmatter(raw) {
+ const fmMatch = raw.match(/^---\n([\s\S]*?)\n---/);
+ if (!fmMatch) return {};
+ const lines = fmMatch[1].split("\n");
+ const result = {};
+ let i = 0;
+ while (i < lines.length) {
+ const keyMatch = lines[i].match(/^(\w+):\s*(.*)/);
+ if (!keyMatch) {
+ i++;
+ continue;
+ }
+ const key = keyMatch[1];
+ const valueStart = keyMatch[2].trim();
+ if (valueStart === ">-" || valueStart === ">") {
+ const parts = [];
+ i++;
+ while (i < lines.length && /^\s/.test(lines[i])) {
+ parts.push(lines[i].trim());
+ i++;
+ }
+ result[key] = parts.join(" ");
+ } else {
+ result[key] = valueStart;
+ i++;
+ }
+ }
+ return result;
+}
+
+function extractSkillContent(raw) {
+ const match = raw.match(/^---\n[\s\S]*?\n---\n([\s\S]*)/);
+ return match ? match[1].trim() : raw.trim();
+}
+
+function readSkillDefFromDir(skillDir, id) {
+ const mdPath = join(skillDir, "SKILL.md");
+ if (!existsSync(mdPath)) return null;
+ let raw;
+ try {
+ raw = readFileSync(mdPath, "utf8");
+ } catch (e) {
+ console.warn(`Skipping ${id}: ${e.message}`);
+ return null;
+ }
+ const fm = parseSkillFrontmatter(raw);
+ if (!fm.name && !fm.description) {
+ console.warn(`Skipping ${id}: no name/description in frontmatter`);
+ return null;
+ }
+ return {
+ id,
+ name: fm.name || id,
+ description: fm.description || "",
+ content: extractSkillContent(raw),
+ /** Full SKILL.md (for tool mention scan); stripped before POST. */
+ _toolScanMarkdown: raw,
+ };
+}
+
+function loadSkillDefs() {
+ const defs = [];
+
+ // Hub skill at plugin root (skills/kibana/kibana-anomaly-detection/SKILL.md)
+ const hubId = basename(PLUGIN_ROOT);
+ const hubDef = readSkillDefFromDir(PLUGIN_ROOT, hubId);
+ if (hubDef) defs.push(hubDef);
+
+ if (!existsSync(SKILLS_DIR)) {
+ console.warn(`Skills directory not found: ${SKILLS_DIR}`);
+ return defs;
+ }
+ for (const entry of readdirSync(SKILLS_DIR).sort()) {
+ const skillPath = join(SKILLS_DIR, entry);
+ const sub = readSkillDefFromDir(skillPath, entry);
+ if (sub) defs.push(sub);
+ }
+ return defs;
+}
+
+async function registerSkill(config, def, dryRun) {
+ const toolIds = def.tool_ids ?? [];
+ const payload = {
+ id: def.id,
+ name: def.name,
+ description: def.description,
+ content: def.content,
+ tool_ids: toolIds,
+ };
+ console.log(`Skill: ${payload.id} (${toolIds.length} tool_ids)`);
+
+ if (dryRun) {
+ const preview = {
+ ...payload,
+ content: payload.content.slice(0, 120) + (payload.content.length > 120 ? "..." : ""),
+ tool_ids: toolIds,
+ };
+ console.log(JSON.stringify(preview, null, 2));
+ return true;
+ }
+
+ const basePath = getBasePath(config);
+ const skillsUrl = `${basePath}/api/agent_builder/skills`;
+ const skillByIdUrl = `${skillsUrl}/${encodeURIComponent(payload.id)}`;
+ const headers = { ...getHeaders(config), "kbn-xsrf": "true" };
+
+ /** PUT body — path carries skill id (see Kibana Agent Builder API). */
+ const updateBody = {
+ name: payload.name,
+ description: payload.description,
+ content: payload.content,
+ tool_ids: toolIds,
+ };
+
+ let res = await kibanaHttpFetch(skillsUrl, config, {
+ method: "POST",
+ headers,
+ body: JSON.stringify(payload),
+ });
+
+ if (res.ok) {
+ console.log(` Registered: ${payload.id}`);
+ return true;
+ }
+
+ const text = await res.text();
+ const alreadyExists =
+ res.status === 409 ||
+ (res.status === 400 && /already exists|duplicate/i.test(text)) ||
+ /already exists/i.test(text);
+
+ if (alreadyExists) {
+ console.log(` Already exists — updating (PUT): ${payload.id}`);
+ const putRes = await kibanaHttpFetch(skillByIdUrl, config, {
+ method: "PUT",
+ headers,
+ body: JSON.stringify(updateBody),
+ });
+ if (putRes.ok) {
+ console.log(` Updated: ${payload.id}`);
+ return true;
+ }
+ const putText = await putRes.text();
+ if (agentBuilderPutUnsupported(putRes.status, putText)) {
+ console.log(` PUT unsupported — deleting and re-creating: ${payload.id}`);
+ await kibanaHttpFetch(skillByIdUrl, config, { method: "DELETE", headers });
+ const recRes = await kibanaHttpFetch(skillsUrl, config, {
+ method: "POST",
+ headers,
+ body: JSON.stringify(payload),
+ });
+ if (recRes.ok) {
+ console.log(` Re-created: ${payload.id}`);
+ return true;
+ }
+ const recText = await recRes.text();
+ console.error(` Failed: ${recRes.status} ${recText.slice(0, 200)}`);
+ return false;
+ }
+ console.error(` Failed to update skill: ${putRes.status} ${putText.slice(0, 200)}`);
+ return false;
+ }
+
+ console.error(` Failed: ${res.status} ${text.slice(0, 200)}`);
+ return false;
+}
+
+/** True if object looks like an Agent Builder agent record (GET /api/agent_builder/agents/{id}). */
+function looksLikeAgentRecord(o) {
+ if (!o || typeof o !== "object") return false;
+ if (typeof o.id === "string") return true;
+ if (typeof o.name === "string") return true;
+ if (o.configuration != null && typeof o.configuration === "object") return true;
+ if (Array.isArray(o.skill_ids)) return true;
+ return false;
+}
+
+/** Normalize GET /api/agent_builder/agents/{id} JSON (envelope or raw agent). */
+function unwrapAgentPayload(data) {
+ if (!data || typeof data !== "object") return null;
+
+ const direct = looksLikeAgentRecord(data) ? data : null;
+ if (direct) return direct;
+
+ const nestedKeys = ["agent", "item", "attributes", "record", "result"];
+ for (const key of nestedKeys) {
+ const v = data[key];
+ if (looksLikeAgentRecord(v)) return v;
+ }
+
+ const d = data.data;
+ if (d != null && typeof d === "object") {
+ if (looksLikeAgentRecord(d)) return d;
+ if (Array.isArray(d) && d.length === 1 && looksLikeAgentRecord(d[0])) return d[0];
+ }
+
+ for (const v of Object.values(data)) {
+ if (looksLikeAgentRecord(v)) return v;
+ if (Array.isArray(v) && v.length && looksLikeAgentRecord(v[0])) return v[0];
+ if (v != null && typeof v === "object" && !Array.isArray(v)) {
+ for (const inner of Object.values(v)) {
+ if (looksLikeAgentRecord(inner)) return inner;
+ }
+ }
+ }
+
+ return null;
+}
+
+function buildAgentPutBody(agent, configuration) {
+ const payload = { configuration };
+ if (agent.name != null) payload.name = agent.name;
+ if (agent.description != null) payload.description = agent.description;
+ if (agent.avatar_color != null) payload.avatar_color = agent.avatar_color;
+ if (agent.avatar_symbol != null) payload.avatar_symbol = agent.avatar_symbol;
+ if (agent.labels != null) payload.labels = agent.labels;
+ if (agent.visibility != null) payload.visibility = agent.visibility;
+ return payload;
+}
+
+/**
+ * Build agent `configuration` from GET response (supports nested `configuration` or flat fields).
+ * Some Kibana versions expose skill_ids / workflow_ids / enable_elastic_capabilities at the top level.
+ */
+function extractAgentConfiguration(agent) {
+ const cfg = {
+ ...(agent.configuration && typeof agent.configuration === "object" ? agent.configuration : {}),
+ };
+ if (!Array.isArray(cfg.skill_ids) && Array.isArray(agent.skill_ids)) {
+ cfg.skill_ids = [...agent.skill_ids];
+ }
+ if (cfg.enable_elastic_capabilities === undefined && typeof agent.enable_elastic_capabilities === "boolean") {
+ cfg.enable_elastic_capabilities = agent.enable_elastic_capabilities;
+ }
+ if (!Array.isArray(cfg.workflow_ids) && Array.isArray(agent.workflow_ids)) {
+ cfg.workflow_ids = [...agent.workflow_ids];
+ }
+ if (!Array.isArray(cfg.skill_ids)) cfg.skill_ids = [];
+ if (!Array.isArray(cfg.workflow_ids)) cfg.workflow_ids = [];
+ return cfg;
+}
+
+/**
+ * Merge bundle skill IDs into the default Elastic AI Agent (PUT /api/agent_builder/agents/{id}).
+ * Requires agentBuilder:manageAgents on top of skill/tool registration privileges.
+ */
+async function attachSkillsToDefaultAgent(config, skillIds, agentId, dryRun) {
+ const basePath = getBasePath(config);
+ const agentUrl = `${basePath}/api/agent_builder/agents/${encodeURIComponent(agentId)}`;
+ const headers = { ...getHeaders(config), "kbn-xsrf": "true" };
+
+ if (dryRun) {
+ const merged = [...new Set(skillIds)];
+ const previewCfg = {
+ skill_ids: merged,
+ enable_elastic_capabilities: DEFAULT_AGENT_ENABLE_ELASTIC_CAPABILITIES === true,
+ workflow_ids: [],
+ };
+ console.log(`PUT ${agentUrl} (dry-run — no GET; preview configuration merge)`);
+ console.log(JSON.stringify({ configuration: previewCfg }, null, 2));
+ return true;
+ }
+
+ const getRes = await kibanaHttpFetch(agentUrl, config, { method: "GET", headers });
+ const { text: getText, parsed: payload } = await readJsonBody(getRes);
+
+ if (!getRes.ok) {
+ const snippet =
+ payload != null && typeof payload === "object" ? JSON.stringify(payload).slice(0, 280) : getText.slice(0, 280);
+ console.warn(` Could not GET agent "${agentId}" (${getRes.status}): ${snippet.slice(0, 220)}`);
+ console.warn(` Skipping default-agent skill attachment (needs read_onechat / agent read on agents).`);
+ return false;
+ }
+
+ const agent = unwrapAgentPayload(payload);
+ if (!agent || typeof agent !== "object") {
+ const keys =
+ payload != null && typeof payload === "object" ? Object.keys(payload).join(", ") : getText.slice(0, 80);
+ console.warn(
+ ` Unexpected GET agent response shape (keys/snippet: ${keys}) — skipping default-agent skill attachment`,
+ );
+ return false;
+ }
+
+ const cfg = extractAgentConfiguration(agent);
+ const existing = [...cfg.skill_ids];
+ const merged = [...new Set([...existing, ...skillIds])];
+ const missing = skillIds.filter((id) => !existing.includes(id));
+
+ const enableWasNotTrue = DEFAULT_AGENT_ENABLE_ELASTIC_CAPABILITIES && cfg.enable_elastic_capabilities !== true;
+ const needsSkillMerge = missing.length > 0;
+
+ if (!needsSkillMerge && !enableWasNotTrue) {
+ console.log(` Default agent "${agentId}" already includes all bundle skills and elastic capabilities are enabled`);
+ return true;
+ }
+
+ if (DEFAULT_AGENT_ENABLE_ELASTIC_CAPABILITIES) {
+ cfg.enable_elastic_capabilities = true;
+ }
+ cfg.skill_ids = merged;
+
+ const putBody = buildAgentPutBody(agent, cfg);
+
+ const reasons = [];
+ if (needsSkillMerge) reasons.push(`adding skills: ${missing.join(", ")}`);
+ if (enableWasNotTrue) reasons.push("enable_elastic_capabilities → true");
+
+ console.log(`Default agent: updating "${agentId}" (${reasons.join("; ")})`);
+
+ const putRes = await kibanaHttpFetch(agentUrl, config, {
+ method: "PUT",
+ headers,
+ body: JSON.stringify(putBody),
+ });
+
+ if (putRes.ok) {
+ console.log(` Updated agent "${agentId}": ${merged.length} skill_id(s) total`);
+ return true;
+ }
+
+ const { text: putErrRaw, parsed: putPayload } = await readJsonBody(putRes);
+ const errText =
+ putPayload != null && typeof putPayload === "object"
+ ? JSON.stringify(putPayload).slice(0, 450)
+ : putErrRaw.slice(0, 450);
+ console.error(` PUT agent "${agentId}" failed (${putRes.status}): ${errText.slice(0, 350)}`);
+ console.error(` (Requires privileges to update agents, e.g. agentBuilder / manage agents.)`);
+ return false;
+}
+
+async function cmdSkillsRegister(argv) {
+ let dryRun = false;
+ let skipDefaultAgent = false;
+ let defaultAgentId = DEFAULT_AGENT_ID;
+ const cli = {};
+ for (let i = 0; i < argv.length; i++) {
+ const a = argv[i];
+ if (a === "--dry-run") dryRun = true;
+ else if (a === "--skip-default-agent") skipDefaultAgent = true;
+ else if (a === "--default-agent-id") defaultAgentId = argv[++i];
+ else if (a === "--kibana-url") cli.kibanaUrl = argv[++i];
+ else if (a === "--username") cli.username = argv[++i];
+ else if (a === "--password") cli.password = argv[++i];
+ else if (a === "--api-key") {
+ cli.apiKeyFromCli = true;
+ cli.apiKey = argv[++i];
+ } else if (a === "--space-id") cli.spaceId = argv[++i];
+ else if (a === "--insecure") cli.insecure = true;
+ }
+
+ const config = getKibanaConfig(cli);
+ if (!validateConfig(config, { dryRun })) process.exit(1);
+ if (config.usingDefaults && !dryRun) warnUsingDefaults();
+
+ const defs = enrichSkillDefsWithToolIds(loadSkillDefs());
+ console.log(
+ `Loaded ${defs.length} skill definitions (tool_ids capped at ${MAX_SKILL_TOOL_IDS}; skill text first, then manifest fallback)`,
+ );
+
+ if (defs.length === 0) {
+ console.error(`No skills found under ${PLUGIN_ROOT} or ${SKILLS_DIR}`);
+ process.exit(1);
+ }
+
+ if (!dryRun) {
+ const st = await kibanaFetch(config, "/api/status");
+ if (!st.ok) {
+ console.error("Cannot reach Kibana:", st.status, st.data);
+ process.exit(1);
+ }
+ console.log("Connected to Kibana", st.data?.version?.number || "?");
+ }
+
+ let succeeded = 0,
+ failed = 0;
+ const succeededSkillIds = [];
+ for (const def of defs) {
+ const ok = await registerSkill(config, def, dryRun);
+ if (ok) {
+ succeeded++;
+ succeededSkillIds.push(def.id);
+ } else {
+ failed++;
+ }
+ }
+
+ if (!dryRun) {
+ console.log(`\nRegistration complete: ${succeeded} succeeded, ${failed} failed`);
+ }
+
+ if (!skipDefaultAgent && succeededSkillIds.length > 0) {
+ console.log("\nAttaching registered skills to default agent…");
+ await attachSkillsToDefaultAgent(config, succeededSkillIds, defaultAgentId, dryRun);
+ } else if (!skipDefaultAgent && succeededSkillIds.length === 0 && !dryRun) {
+ console.log("\nSkipping default-agent attachment (no skills registered successfully).");
+ }
+}
+
+/** Register tools, then workflows, then skills (skills attach `tool_ids` for tools POSTed in step 1). */
+async function cmdAllRegister(argv) {
+ console.log("=== 1/3 tools register ===\n");
+ await cmdToolsRegister(argv);
+ console.log("\n=== 2/3 workflows register ===\n");
+ await cmdWorkflowsRegister(argv);
+ console.log("\n=== 3/3 skills register (+ default agent) ===\n");
+ await cmdSkillsRegister(argv);
+ console.log("\n=== all register complete ===");
+}
+
+function printUsage() {
+ console.log(`
+Kibana Agent Builder (anomaly-detection)
+
+Commands:
+ test GET /api/status — verify URL and credentials
+ tools register POST all ES|QL tools to Agent Builder
+ workflows register POST all YAML workflow definitions to Kibana Workflows engine
+ skills register POST hub + skills/; PUT default agent configuration (skill_ids merge, enable_elastic_capabilities, workflow_ids)
+ all register Run tools register, workflows register, skills register (+ default agent)
+ jobs create-service-health Create baseline service issue-detection ML jobs + datafeeds
+
+Options (workflows/tools/skills/all register):
+ --dry-run Print JSON payloads only
+ --skip-default-agent Do not PUT skill_ids on the default Elastic AI Agent
+ --default-agent-id ID Agent id to update (default: from agent_builder_constants.json, usually elastic-ai-agent)
+ --kibana-url URL
+ --username USER
+ --password PASS
+ --api-key KEY
+ --space-id ID
+ --insecure Skip TLS verification (also KIBANA_INSECURE=true)
+
+Options (jobs create-service-health):
+ --prefix ID Job ID prefix (default: svc)
+ --metrics-index PAT Metrics index pattern (default: metrics-*)
+ --logs-index PAT Logs index pattern (default: logs-*)
+ --apm-index PAT APM index pattern (default: apm-*)
+ --bucket-span SPAN Bucket span (default: 15m)
+ --query-delay DELAY Datafeed query_delay (default: 120s)
+ --memory-limit SIZE model_memory_limit (default: 256mb)
+
+Environment variables match kibana-dashboards.js — see script header.
+`);
+}
+
+async function main() {
+ const argv = process.argv.slice(2);
+ if (argv.length === 0 || ["-h", "--help", "help"].includes(argv[0])) {
+ printUsage();
+ process.exit(argv.length === 0 ? 1 : 0);
+ }
+
+ const [cmd, sub] = argv;
+ if (cmd === "test") {
+ await cmdTest(getKibanaConfig({}));
+ return;
+ }
+ if (cmd === "tools" && sub === "register") {
+ await cmdToolsRegister(argv.slice(2));
+ return;
+ }
+ if (cmd === "workflows" && sub === "register") {
+ await cmdWorkflowsRegister(argv.slice(2));
+ return;
+ }
+ if (cmd === "skills" && sub === "register") {
+ await cmdSkillsRegister(argv.slice(2));
+ return;
+ }
+ if (cmd === "all" && sub === "register") {
+ await cmdAllRegister(argv.slice(2));
+ return;
+ }
+ if (cmd === "jobs" && sub === "create-service-health") {
+ await cmdJobsCreateServiceHealth(argv.slice(2));
+ return;
+ }
+
+ console.error(`Unknown command: ${cmd}${sub ? ` ${sub}` : ""}`);
+ printUsage();
+ process.exit(1);
+}
+
+main().catch((e) => {
+ console.error(e);
+ process.exit(1);
+});
diff --git a/skills/kibana/kibana-dashboards/SKILL.md b/skills/kibana/kibana-dashboards/SKILL.md
index 7472155..4a9e3aa 100644
--- a/skills/kibana/kibana-dashboards/SKILL.md
+++ b/skills/kibana/kibana-dashboards/SKILL.md
@@ -6,7 +6,7 @@ description: >
deployment.
metadata:
author: elastic
- version: 0.1.1
+ version: 0.1.2
---
# Kibana Dashboards and Visualizations
@@ -231,8 +231,8 @@ scrolling. Design for density—place primary KPIs and key trends above the fold
| `region_map` | Region/choropleth maps | Yes |
| `pie`, `treemap`, `mosaic`, `waffle` | Partition charts | Yes |
-> **Note:** To create donut charts, use `pie` with `donut_hole` set to `"s"`, `"m"`, or `"l"` (small, medium, large
-> hole). Use `"none"` for a solid pie.
+> **Note:** To create donut charts, use `pie` with `styling.donut_hole` set to `"s"`, `"m"`, or `"l"` (small, medium,
+> large hole). Use `"none"` for a solid pie. Example: `"styling": { "donut_hole": "m" }`.
### Dataset Types
@@ -325,7 +325,7 @@ For detailed schemas and all chart type options, see [Chart Types Reference](ref
{
"title": "Top Hosts",
"type": "xy",
- "axis": { "x": { "title": { "visible": false } }, "y": { "anchor": "start", "title": { "visible": false } } },
+ "axis": { "x": { "title": { "visible": false } }, "y": { "title": { "visible": false } } },
"layers": [
{
"type": "bar_horizontal",
@@ -345,7 +345,7 @@ For detailed schemas and all chart type options, see [Chart Types Reference](ref
"type": "xy",
"axis": {
"x": { "title": { "visible": false }, "scale": "temporal", "domain": { "type": "fit", "rounding": false } },
- "y": { "anchor": "start", "title": { "visible": false } }
+ "y": { "title": { "visible": false } }
},
"layers": [
{
diff --git a/skills/kibana/kibana-dashboards/assets/bar-chart-esql.json b/skills/kibana/kibana-dashboards/assets/bar-chart-esql.json
index 5e85649..51995b8 100644
--- a/skills/kibana/kibana-dashboards/assets/bar-chart-esql.json
+++ b/skills/kibana/kibana-dashboards/assets/bar-chart-esql.json
@@ -3,7 +3,7 @@
"type": "xy",
"axis": {
"x": { "title": { "visible": false } },
- "y": { "anchor": "start", "title": { "visible": false } }
+ "y": { "title": { "visible": false } }
},
"layers": [
{
diff --git a/skills/kibana/kibana-dashboards/assets/demo-dashboard.json b/skills/kibana/kibana-dashboards/assets/demo-dashboard.json
index 2da503c..c5eba97 100644
--- a/skills/kibana/kibana-dashboards/assets/demo-dashboard.json
+++ b/skills/kibana/kibana-dashboards/assets/demo-dashboard.json
@@ -91,7 +91,7 @@
"type": "xy",
"axis": {
"x": { "title": { "visible": false } },
- "y": { "anchor": "start", "title": { "visible": false } }
+ "y": { "title": { "visible": false } }
},
"layers": [
{
@@ -118,7 +118,7 @@
"type": "xy",
"axis": {
"x": { "title": { "visible": false } },
- "y": { "anchor": "start", "title": { "visible": false } }
+ "y": { "title": { "visible": false } }
},
"layers": [
{
@@ -195,7 +195,7 @@
"type": "xy",
"axis": {
"x": { "title": { "visible": false } },
- "y": { "anchor": "start", "title": { "visible": false } }
+ "y": { "title": { "visible": false } }
},
"layers": [
{
@@ -223,7 +223,7 @@
"type": "xy",
"axis": {
"x": { "title": { "visible": false } },
- "y": { "anchor": "start", "title": { "visible": false } }
+ "y": { "title": { "visible": false } }
},
"layers": [
{
diff --git a/skills/kibana/kibana-dashboards/assets/ecommerce-analytics-dashboard.json b/skills/kibana/kibana-dashboards/assets/ecommerce-analytics-dashboard.json
index 8a32f0a..c4eb1b5 100644
--- a/skills/kibana/kibana-dashboards/assets/ecommerce-analytics-dashboard.json
+++ b/skills/kibana/kibana-dashboards/assets/ecommerce-analytics-dashboard.json
@@ -123,7 +123,6 @@
}
},
"y": {
- "anchor": "start",
"title": {
"visible": false
}
@@ -170,7 +169,6 @@
}
},
"y": {
- "anchor": "start",
"title": {
"visible": false
}
@@ -217,7 +215,6 @@
}
},
"y": {
- "anchor": "start",
"title": {
"visible": false
}
@@ -265,7 +262,6 @@
}
},
"y": {
- "anchor": "start",
"title": {
"visible": false
}
@@ -313,7 +309,6 @@
}
},
"y": {
- "anchor": "start",
"title": {
"visible": false
}
@@ -453,7 +448,6 @@
}
},
"y": {
- "anchor": "start",
"title": {
"visible": false
}
@@ -500,7 +494,6 @@
}
},
"y": {
- "anchor": "start",
"title": {
"visible": false
}
diff --git a/skills/kibana/kibana-dashboards/assets/line-chart-timeseries.json b/skills/kibana/kibana-dashboards/assets/line-chart-timeseries.json
index c0886fc..53b5cfc 100644
--- a/skills/kibana/kibana-dashboards/assets/line-chart-timeseries.json
+++ b/skills/kibana/kibana-dashboards/assets/line-chart-timeseries.json
@@ -3,7 +3,7 @@
"type": "xy",
"axis": {
"x": { "title": { "visible": false }, "scale": "temporal", "domain": { "type": "fit", "rounding": false } },
- "y": { "anchor": "start", "title": { "visible": false } }
+ "y": { "title": { "visible": false } }
},
"layers": [
{
diff --git a/skills/kibana/kibana-dashboards/references/chart-types-reference.md b/skills/kibana/kibana-dashboards/references/chart-types-reference.md
index 1468262..3342ab9 100644
--- a/skills/kibana/kibana-dashboards/references/chart-types-reference.md
+++ b/skills/kibana/kibana-dashboards/references/chart-types-reference.md
@@ -11,8 +11,8 @@ Complete schema reference for each supported chart type via the Kibana dashboard
- `tag_cloud` — Tag/word cloud
- `data_table` — Data tables
- `region_map` — Region/choropleth maps
-- `pie`, `treemap`, `mosaic`, `waffle` — Partition charts (use `pie` with `donut_hole` for donuts: `"s"`, `"m"`, or
- `"l"`)
+- `pie`, `treemap`, `mosaic`, `waffle` — Partition charts (use `pie` with `styling.donut_hole` for donuts: `"s"`, `"m"`,
+ or `"l"`)
## DataView Aggregation Operations
@@ -323,8 +323,8 @@ For ES|QL, uses `metrics` and `rows` arrays. Each entry uses `{ column: "..." }`
Partition charts display parts of a whole. Uses a flat structure (no `layers`) with `metrics` for the slice sizes and
`group_by` for the rings or groupings. The schema is identical for all partition types—simply change `"type": "pie"` to
-`"treemap"`, `"mosaic"`, or `"waffle"`. To create a donut, use `"type": "pie"` with `"donut_hole"` set to `"s"`, `"m"`,
-or `"l"`.
+`"treemap"`, `"mosaic"`, or `"waffle"`. To create a donut, use `"type": "pie"` with `"styling": { "donut_hole": "m" }`.
+Valid `donut_hole` values are `"none"`, `"s"`, `"m"`, or `"l"`.
**ES|QL Example:**
diff --git a/skills/kibana/kibana-dashboards/references/dashboard-api-reference.md b/skills/kibana/kibana-dashboards/references/dashboard-api-reference.md
index 1d79229..8e9b7cb 100644
--- a/skills/kibana/kibana-dashboards/references/dashboard-api-reference.md
+++ b/skills/kibana/kibana-dashboards/references/dashboard-api-reference.md
@@ -369,7 +369,7 @@ node scripts/kibana-dashboards.js dashboard create dashboard.json
| Datatable structure | ES\|QL data_table requires `metrics` + `rows` arrays |
| XY chart fails | Put `data_source` inside each layer (for both dataView and ES\|QL) |
| Heatmap property names | Heatmap uses `x`, `y`, `metric` for axes and value |
-| XY axis config | Use `axis` (singular); `y` with `anchor: "start"` for left axis |
+| XY axis config | Use `axis` (singular); `y` for left axis, `y2` for right axis |
| ref_id panels missing | Prefer inline definitions (properties in `config`) over `ref_id` |
## Testing from Dev Tools
diff --git a/skills/kibana/kibana-dashboards/scripts/kibana-dashboards.js b/skills/kibana/kibana-dashboards/scripts/kibana-dashboards.js
index 08ce0d2..85a3124 100644
--- a/skills/kibana/kibana-dashboards/scripts/kibana-dashboards.js
+++ b/skills/kibana/kibana-dashboards/scripts/kibana-dashboards.js
@@ -143,7 +143,6 @@ function getHeaders(config) {
const headers = {
"Content-Type": "application/json",
"kbn-xsrf": "true",
- "x-elastic-internal-origin": "kibana",
"User-Agent": "elastic-agentic",
};
diff --git a/skills/observability/k8s-investigation/SKILL.md b/skills/observability/k8s-investigation/SKILL.md
new file mode 100644
index 0000000..33c5903
--- /dev/null
+++ b/skills/observability/k8s-investigation/SKILL.md
@@ -0,0 +1,465 @@
+---
+name: observability-k8s-investigation
+description: >
+ Investigate Kubernetes workload, node, and control-plane issues using OTel telemetry
+ (EDOT). Use when diagnosing pod failures (CrashLoopBackOff, OOMKilled, Error), node
+ pressure, resource exhaustion, image pull failures, admission rejections, autoscaling
+ anomalies, or correlating K8s state with application signals. OTel ingest path only
+ — the legacy ECS Kubernetes integration shape is out of scope.
+metadata:
+ author: elastic
+ version: 0.2.0
+---
+
+# Kubernetes Investigation
+
+Diagnose Kubernetes issues using OTel telemetry collected via EDOT (Elastic Distribution of OpenTelemetry) and the
+kube-stack collector. Correlate cluster state, pod runtime metrics, K8s events, application logs, and APM to identify
+root cause across the workload, node, and control-plane layers.
+
+## Scope
+
+**In scope:** OTel-receiver-namespaced indices (`metrics-kubeletstatsreceiver.otel-*`,
+`metrics-k8sclusterreceiver.otel-*`, `logs-k8seventsreceiver.otel-*`, `logs-k8sobjectsreceiver.otel-*`) and OTel
+semantic conventions (`k8s.pod.name`, `k8s.namespace.name`, `k8s.container.restarts`).
+
+**Out of scope:**
+
+- The legacy Elastic Agent Kubernetes integration (`metrics-kubernetes.*`, `logs-kubernetes.*`, `kubernetes.*` fields).
+ Being deprecated — do not author queries against these paths.
+- APM-layer analysis (service SLO breaches, transaction error rates, upstream dependency health). Different domain —
+ once a K8s root cause is ruled in or out, APM investigation continues outside this skill.
+- Cluster provisioning, capacity planning, cost optimization. Different domain.
+
+## Guidelines
+
+These apply to every investigation. When in doubt, re-read them before writing the synthesis.
+
+**Absence of evidence is not evidence. Do not confabulate from empty results.** If log queries return 0 rows, logs are
+likely not collected or the pod has no recent lines — this does _not_ mean "dependency unavailable" or any other
+specific failure mode. Report `no_logs_available` and weight remaining signals accordingly.
+
+**Empty dependency data ≠ upstream healthy.** Services without APM instrumentation (load generators, workers) emit no
+destination metrics. Report `insufficient_dependency_data`, not "upstreams OK."
+
+**Co-symptoms are not causes.** Two services degrading simultaneously usually share an upstream, not a causal link. Only
+attribute causation when (a) one service's degradation clearly precedes the other's, and (b) the delta is large (>5×
+error rate, >3× latency).
+
+**OOMKilled ≠ memory leak by default.** The limit may simply be undersized for the workload's working set. Compare
+against a 7-day baseline at the same hour-of-day before claiming a leak.
+
+**Error-termination ≠ application bug by default.** Check `k8s.pod.cpu_limit_utilization` first. CFS throttling driving
+liveness probe timeouts is the most common misdiagnosis in this space.
+
+**Average CPU hides throttling.** A pod can look healthy at 40–60% average `cpu_limit_utilization` while being throttled
+severely at p99. Linux enforces CPU limits in 100ms periods; bursty workloads hit quota mid-period and stall. Look at
+max and p95, not just average.
+
+**Restart count is boolean, not a counter.** `k8s.container.restarts` is pulled directly from the K8s API and may be
+pruned by the kubelet at any time, so the absolute value is unreliable. Treat it as `== 0` (no recent restarts) vs `> 0`
+(recently restarting); do not derive backoff timing or "linear vs exponential" patterns from it. Confirm the restart
+pattern via K8s `Killing` / `BackOff` events instead.
+
+**Prefer to report uncertainty over manufacturing confidence.** If the evidence is ambiguous, the synthesis should say
+so. Competing hypotheses are a valid output.
+
+## Indices and fields
+
+### Where to look
+
+| Signal | Index pattern | Use |
+| --------------------- | --------------------------------------------------- | ------------------------------------------------------------------- |
+| Pod/container runtime | `metrics-kubeletstatsreceiver.otel-*` | CPU, memory, network, filesystem. Utilization ratios. |
+| Cluster state | `metrics-k8sclusterreceiver.otel-*` | Restarts, phase, last-terminated reason, HPA, quota, node condition |
+| K8s events | `logs-k8seventsreceiver.otel-*` | Killing, BackOff, FailedScheduling, Evicted, image pull events |
+| K8s object snapshots | `logs-k8sobjectsreceiver.otel-*` | Deployment/service/configmap state over time |
+| Application logs | `logs-*.otel-*` | `body.text`, `severity_text`, filtered by `k8s.pod.name` |
+| APM | `traces-*.otel-*`, `metrics-service_*.otel-default` | Correlate via `service.name` + K8s resource attrs |
+| ML anomalies | `.ml-anomalies-*` | Memory-growth, restart-rate, throttle jobs (if configured) |
+
+### Key fields
+
+Flat OTel paths work in ES|QL. Prefer the flat form for readability; the nested `resource.attributes.*` form is for raw
+log documents only.
+
+| Field | Index | What it is |
+| ------------------------------------------------ | --------------------------- | ------------------------------------------------------- |
+| `k8s.pod.name` | all k8s | Pod name |
+| `k8s.namespace.name` | all k8s | Namespace |
+| `k8s.container.name` | all k8s | Container within pod |
+| `k8s.deployment.name` | k8sclusterreceiver + others | Parent deployment |
+| `k8s.pod.phase` | k8sclusterreceiver | Pending=1/Running=2/Succeeded=3/Failed=4/Unknown=5 |
+| `k8s.container.restarts` | k8sclusterreceiver | Total container restart count |
+| `k8s.container.status.last_terminated_reason` | k8sclusterreceiver | `OOMKilled`, `Error`, `Completed`, `ContainerCannotRun` |
+| `k8s.pod.status_reason` | k8sclusterreceiver | Pod-level reason (`Evicted`, `NodeLost`) |
+| `k8s.pod.memory_limit_utilization` | kubeletstatsreceiver | 0.0–1.0+ (can exceed 1 transiently before OOM) |
+| `k8s.pod.cpu_limit_utilization` | kubeletstatsreceiver | 0.0–N (frequently >1 under CFS throttling) |
+| `k8s.pod.memory.usage` / `.working_set` | kubeletstatsreceiver | Bytes |
+| `k8s.node.condition_memory_pressure` | k8sclusterreceiver | 1 = pressure, 0 = ok |
+| `k8s.node.condition_ready` | k8sclusterreceiver | 0 = NotReady |
+| `k8s.hpa.current_replicas` / `.desired_replicas` | k8sclusterreceiver | HPA state |
+| `attributes.k8s.event.reason` | k8seventsreceiver | Event reason (filter on this) |
+| `body.text` | k8seventsreceiver / logs | Event message / log message |
+| `k8s.object.name` | k8seventsreceiver | involvedObject name (log attribute, use flat form) |
+
+### Field availability
+
+Several fields above are off by default in stock kube-stack collectors and require explicit configuration. Verify
+presence before relying on them; if absent, fall back as noted and call out the substitution in the synthesis.
+
+| Field | Why it might be missing | Fall-back |
+| ------------------------------------------------------------ | ---------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------- |
+| `k8s.container.status.last_terminated_reason` | Optional metric in k8sclusterreceiver; gated behind `metrics_collected.metadata` config. | Infer from K8s `Killing` / `OOMKilling` events in `logs-k8seventsreceiver.otel-*` and exit codes in app logs. |
+| `k8s.pod.status_reason` | Same — optional metric on k8sclusterreceiver. | Infer from events: `Evicted`, `NodeLost`, `Preempted`. |
+| `k8s.pod.cpu_limit_utilization` / `memory_limit_utilization` | Only emitted when the pod has the corresponding limit set, and the kubeletstatsreceiver metric is enabled. | Compute manually as `k8s.pod.cpu.usage / ` from k8sclusterreceiver, or use absolute usage trending against a baseline. |
+| `k8s.node.condition_memory_pressure` | Gated behind k8sclusterreceiver `node_conditions_to_report` (default omits this). | Compare `k8s.node.memory.usage` against `k8s.node.allocatable_memory`, or look for `Evicted` events on the node. |
+
+If a fall-back is used, note it in the synthesis (e.g. `(via memory.usage; limit_utilization not collected)`) so the
+reader knows the signal is indirect.
+
+## ES|QL gotchas
+
+Before writing queries, know these. Each of them silently produces wrong answers rather than failing loudly.
+
+**`VALUES()` returns scalar for single distinct value, array for multiple.** Templating that assumes array shape (e.g.
+`| first`) extracts the first character of the string when scalar. Use `MV_FIRST(VALUES(...))` or handle both.
+
+**`PERCENTILE` does not work on OTel `histogram` type** (as of 8.15). For APM duration percentiles, use `AVG` on the
+`aggregate_metric_double` summary field (`AVG(transaction.duration.summary)` divides sum by value_count). For true
+percentiles, fall back to Kibana Query DSL.
+
+**`COUNT(agg_metric_double)` returns `value_count` (events), not doc count.** `SUM(field)` gives the sum component;
+`AVG(field)` gives sum/value_count. Do not use `SUM(transaction.duration.summary)` as an event-count proxy — it returns
+total duration.
+
+**K8s metrics use flat OTel field paths in ES|QL.** `k8s.pod.name`, not `resource.attributes.k8s.pod.name`. The nested
+form is for raw log documents.
+
+## Failure-mode taxonomy
+
+Vocabulary for classification, not a decision tree. Use the pivotal-signal column to recognize which mode you're looking
+at; use "Investigate" to know what else should corroborate.
+
+### Workload layer
+
+| Mode | Pivotal signal | Investigate |
+| ----------------------------------- | -------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| **OOMKilled** | `last_terminated_reason == "OOMKilled"` + `memory_limit_utilization → 1.0` | Monotonic rise (leak) vs. load-driven spike? Compare current trend to 7-day baseline. Check heap metrics (JVM, Go, Node) for GC pressure. |
+| **CPU throttling → Error exit** | `cpu_limit_utilization > 1.0` + `last_terminated_reason == "Error"` | Liveness/readiness probe timeouts from CFS throttling. Average CPU can look fine (40–60%) while p99 throttle is severe. Check probe timeouts vs observed startup/health latency. |
+| **Liveness probe misconfiguration** | Restarts without resource pressure; `initialDelaySeconds` < startup time | K8s events show `Unhealthy` / `Killing`. `kubectl logs --previous` typically shows healthy startup before kill. |
+| **CrashLoopBackOff (generic)** | `BackOff` events + rising `k8s.container.restarts` | Branch on `last_terminated_reason` — this is a meta-mode. OOMKilled → memory path; Error → logs + throttling; ContainerCannotRun → image/exec. |
+| **ImagePullBackOff** | K8s events `Failed` with image name + `429` or `not found` | Registry rate limit? Missing tag? Wrong imagePullSecret? Check recency of `Pulling`/`Pulled` events. |
+| **Stuck rollout** | New pods `Pending`/not-Ready > `progressDeadlineSeconds`; old pods still serving | Check `k8s.deployment.available` vs `.desired`. Admission rejection? Readiness probe failing on new pods? HPA not scaling? |
+| **Termination signal race** | Brief 5xx bursts correlated with rolling deploys | Endpoint removal races termination. New requests can hit the pod after SIGTERM starts. NGINX gotcha: `STOPSIGNAL SIGTERM` triggers _fast_ shutdown, not graceful — use `STOPSIGNAL SIGQUIT` for graceful drain. Check ingress 502 rate vs rollout timing. |
+
+### Node layer
+
+| Mode | Pivotal signal | Investigate |
+| ----------------------------------- | ----------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------- |
+| **Node NotReady cascade** | `k8s.node.condition_ready == 0` + mass `Evicted` events | Memory pressure? Disk pressure? Network partition from API server? Inspect kubelet logs, `k8s.node.condition_*` history. |
+| **Resource eviction** | `status_reason == "Evicted"` + `condition_memory_pressure == 1` on node | Node-level noisy neighbor. QoS order: BestEffort → Burstable → Guaranteed. Identify which pod drove node memory up. |
+| **Node affinity/selector conflict** | Mass unschedulable pods after label change | K8s events show `FailedScheduling`. Often triggered by cluster upgrades (e.g. `node-role.kubernetes.io/master` → `control-plane`). |
+
+### Control plane
+
+| Mode | Pivotal signal | Investigate |
+| ----------------------------- | ------------------------------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------- |
+| **etcd I/O cascade** | API server latency spike + cluster-wide kubelet heartbeat failures | Disk IOPS, fsync latency (must be <10ms). Cloud-burst-credit exhaustion is common. |
+| **Admission webhook block** | Mass `FailedCreate` across namespaces; deployments frozen | `failurePolicy:Fail` webhook pod crashed. Check webhook pod health + API server TCP connection cache (caches dead connections ~15 min). |
+| **Priority preemption storm** | Production pods terminating with `preempted-by` annotation | New `PriorityClass` with `globalDefault:true` caused cascade. Check `kube-scheduler` events. |
+| **PDB drain deadlock** | Node drain stuck indefinitely; HTTP 429 from Eviction API | PDB `minAvailable`/`maxUnavailable` too strict. No default drain timeout. Manual PDB deletion unblocks. |
+
+### Autoscaling & admission
+
+| Mode | Pivotal signal | Investigate |
+| ----------------------------- | ------------------------------------------------------------------ | -------------------------------------------------------------------------------------------------------------------------------------------------- |
+| **HPA unready-pod dampening** | Load rising, HPA not scaling; unready pods included in calculation | HPA averages CPU across all replicas including unready (0% contribution). Check `k8s.hpa.current_replicas` vs `.desired_replicas` + pod readiness. |
+| **Resource quota silent 403** | Deployment stuck at n-1/n; `FailedCreate` on ReplicaSet | Namespace quota exhausted (often CronJob accumulation). Check `k8s.resource_quota.used` vs `.hard_limit`. |
+
+### Networking
+
+| Mode | Pivotal signal | Investigate |
+| --------------------------- | -------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------- |
+| **StatefulSet split-brain** | Duplicate pod identities across partitioned nodes | Network partition + eviction timeout race. Two instances of same ordinal running. No fencing by default. |
+| **CoreDNS OOMKill** | CoreDNS restarts + cluster-wide DNS timeouts in app logs | Default CoreDNS memory (~170Mi) insufficient under query amplification (ndots:5, each external lookup → ~10 lookups). |
+
+### When classification is ambiguous
+
+Real incidents often match two modes. Examples:
+
+- OOMKilled pod with simultaneous CPU throttling — memory usually drives the kill, but verify by checking whether memory
+ or CPU hit limit first.
+- Stuck rollout with HPA dampening and resource quota near-exhaustion — both can freeze a deploy. Check which constraint
+ is binding.
+- Node NotReady with pods that were already crashing — the node issue may be incidental.
+
+When two modes fit, name both in the synthesis and say which one you believe is causal and why. Do not force a single
+hypothesis when the evidence supports two.
+
+## Signal interpretation
+
+### Memory
+
+- **Monotonic rise over 30–60 min** → leak. Check GC metrics for the language: JVM `jvm.gc.duration`, Go
+ `process.runtime.go.gc.pause_ns`, Node `v8js_gc_duration`. Rising GC frequency/pause with stable live-set is the
+ canonical leak signature.
+- **Diurnal / load-correlated spikes** → load-driven, not leak. Consider HPA tuning or limit increase.
+- **Hits 1.0, then restart** → OOMKilled confirmed. Exit code 137 (SIGKILL) in app logs consistent.
+
+### CPU
+
+- `cpu_limit_utilization > 1.0` sustained → CFS throttling. Node has spare CPU; the pod is quota-blocked.
+- Symptoms of throttling (not the throttle metric itself): liveness probe timeouts, p99 latency 4–16× p50, queue
+ backpressure upstream, Error-reason container terminations.
+- Average can look healthy while p95 is throttled. Do not trust average alone.
+
+### Restart patterns
+
+- `restarts > 0` recently → workload has been restarting. Don't read magnitude into the count (see _Restart count is
+ boolean_); confirm the pattern from K8s `Killing` / `BackOff` event timestamps in `logs-k8seventsreceiver.otel-*`.
+- Restarts correlated with memory pressure (`memory_limit_utilization → 1.0`) → OOMKilled path.
+- Restarts without memory/CPU pressure → probe misconfig, app bug, or startup dependency failure. Pull events for
+ `Unhealthy` and `Killing`.
+
+### Termination reasons
+
+- `OOMKilled` → memory path.
+- `Error` → non-zero exit. Check app logs; if empty/minimal, check CPU throttling before attributing to app logic.
+- `Completed` → ran to completion. Normal for Jobs/CronJobs/init containers; anomalous otherwise.
+- `ContainerCannotRun` → runtime/image/exec issue. Check image pull events.
+
+## Investigation flow
+
+> An investigation is not a checklist. The sections below describe a _typical_ arc — **compress, skip, or revisit them
+> based on what you find.** Terminate as soon as you have enough evidence to synthesize at a known confidence. Chasing
+> signals past the point of diminishing returns is a failure mode, not thoroughness.
+
+### Orient
+
+Resolve the target: `k8s.pod.name`, `k8s.namespace.name`, optionally `k8s.deployment.name` and `service.name`. If no
+time window is given, default to the last hour for pod-level investigations, last 2 hours for event correlation, last 6
+hours for ongoing/unresolved incidents.
+
+If the alert payload already tells you the failure mode (e.g., it fires specifically on `OOMKilled`), note that and skip
+classification; move to confirmation and baseline comparison.
+
+### Characterize
+
+Get the shape of the workload's recent behavior: restart count, termination reasons, phase, utilization. One or two
+queries usually suffice.
+
+```esql
+FROM metrics-k8sclusterreceiver.otel-*
+| WHERE k8s.pod.name == "" AND k8s.namespace.name == ""
+ AND @timestamp > NOW() - 1 hour
+| STATS restarts = MAX(k8s.container.restarts),
+ term_reasons = VALUES(k8s.container.status.last_terminated_reason),
+ phase = MAX(k8s.pod.phase)
+```
+
+```esql
+FROM metrics-kubeletstatsreceiver.otel-*
+| WHERE k8s.pod.name == "" AND @timestamp > NOW() - 15 minutes
+| STATS mem_pct = ROUND(MAX(k8s.pod.memory_limit_utilization) * 100, 1),
+ cpu_pct = ROUND(MAX(k8s.pod.cpu_limit_utilization) * 100, 1)
+```
+
+### Classify
+
+Use the taxonomy. The pivotal signal should match; the "Investigate" column tells you what corroboration to seek.
+
+When two modes fit, note both and proceed with the one that has the stronger pivotal signal. You may revise during
+corroboration.
+
+### Corroborate
+
+Pull the evidence your classification predicts you'll find. Typical sources:
+
+**K8s events** for the namespace and window:
+
+```esql
+FROM logs-k8seventsreceiver.otel-*
+| WHERE k8s.namespace.name == ""
+ AND @timestamp > NOW() - 2 hours
+ AND attributes.k8s.event.reason IN (
+ "BackOff", "Killing", "Unhealthy", "Failed",
+ "FailedScheduling", "Evicted", "SuccessfulRescale",
+ "Pulling", "Pulled", "Started", "Created"
+ )
+| SORT @timestamp DESC
+| KEEP @timestamp, attributes.k8s.event.reason, body.text, k8s.object.name
+| LIMIT 30
+```
+
+**Application logs** if available — look at the 200 most recent lines before the termination timestamp. If absent, flag
+`no_logs_available`; do not invent a log pattern.
+
+**APM** if the pod runs an instrumented service — resolve `service.name` from pod resource attributes for later
+correlation. SLO / latency / error-rate analysis itself is APM-layer work and out of scope for this skill.
+
+**Baseline comparison** — for utilization-based findings, compare current values to 7-day-prior at the same hour-of-day.
+"High memory" is meaningful only relative to what's normal for this workload.
+
+### Check for upstream cause (conditional)
+
+Only pursue if the symptom pattern suggests it. Threshold: upstream error rate >5× baseline _or_ latency >3× baseline,
+AND degradation started before the symptom on the target service. Co-symptoms do not establish causation.
+
+If `metrics-service_destination.1m.otel-default` has no rows for the service, report `insufficient_dependency_data` —
+not "upstreams healthy."
+
+### Check for recent change (conditional)
+
+`SuccessfulCreate` / `Pulled` events in the last 2 hours often correlate with deploys. `logs-k8sobjectsreceiver.otel-*`
+shows configmap/secret/deployment spec changes. A change within 15 minutes of the symptom onset is a strong correlation,
+but still a correlation — verify it plausibly explains the mode you've classified.
+
+### Synthesize and stop
+
+Synthesize as soon as you have enough evidence to support a hypothesis at known confidence. You do not need to complete
+every section above — investigation terminates when either:
+
+- You have a high-confidence hypothesis with corroboration, or
+- You have a low/medium-confidence hypothesis and further queries are unlikely to change the picture (e.g., logs are
+ unavailable, APM isn't instrumented, no recent changes found).
+
+## Synthesis
+
+Default structure:
+
+```text
+HYPOTHESIS (confidence: high | medium | low)
+
+
+EVIDENCE
+-
+-
+-
+
+CONFIDENCE NOTE
+
+
+RECOMMENDED NEXT STEPS
+1.
+2.
+
+DOWNSTREAM IMPACT
+
+```
+
+**When two hypotheses are live:** replace HYPOTHESIS with COMPETING HYPOTHESES; list both, say which you lean toward and
+why, and list the evidence that would disambiguate them.
+
+**When no incident is found** (symptom resolved, or alert appears spurious): say so directly.
+`ALERT FIRED BUT SYSTEM APPEARS HEALTHY` is a valid output. List what you checked and what you didn't find.
+
+### Confidence calibration
+
+Start at **high** and downgrade based on what's missing:
+
+- Downgrade to **medium** if: primary signal is clear but corroboration is missing (no logs, no APM, no baseline
+ comparison possible). Or: two modes fit and you can't disambiguate.
+- Downgrade to **low** if: only a single signal supports the hypothesis, signals conflict, or the mode requires evidence
+ you couldn't fetch.
+
+Never return **high** when application log data was absent and the hypothesis depends on application behavior. Absence
+of evidence does not corroborate a hypothesis.
+
+## Query recipes
+
+### Most-restarting pods in a namespace
+
+```esql
+FROM metrics-k8sclusterreceiver.otel-*
+| WHERE k8s.namespace.name == "" AND @timestamp > NOW() - 1 hour
+| STATS restarts = MAX(k8s.container.restarts) BY k8s.pod.name, k8s.container.status.last_terminated_reason
+| WHERE restarts > 0
+| SORT restarts DESC
+| LIMIT 20
+```
+
+### CPU throttling check for a pod
+
+```esql
+FROM metrics-kubeletstatsreceiver.otel-*
+| WHERE k8s.pod.name == "" AND @timestamp > NOW() - 30 minutes
+| STATS max_cpu_ratio = ROUND(MAX(k8s.pod.cpu_limit_utilization), 2),
+ avg_cpu_ratio = ROUND(AVG(k8s.pod.cpu_limit_utilization), 2),
+ max_cpu_cores = ROUND(MAX(k8s.pod.cpu.usage), 3)
+```
+
+Sustained ratio >1.0 = throttling. Transient >1.0 with avg <0.5 is usually benign burst.
+
+### Nodes under memory pressure (right now)
+
+```esql
+FROM metrics-k8sclusterreceiver.otel-*
+| WHERE @timestamp > NOW() - 15 minutes AND k8s.node.condition_memory_pressure == 1
+| STATS ts = MAX(@timestamp) BY k8s.node.name
+| SORT ts DESC
+```
+
+### Admission denials (webhook or quota) last hour
+
+```esql
+FROM logs-k8seventsreceiver.otel-*
+| WHERE @timestamp > NOW() - 1 hour
+ AND (attributes.k8s.event.reason == "FailedCreate"
+ OR body.text LIKE "*admission webhook*"
+ OR body.text LIKE "*exceeded quota*")
+| SORT @timestamp DESC
+| KEEP @timestamp, k8s.namespace.name, attributes.k8s.event.reason, body.text
+| LIMIT 30
+```
+
+### Firing K8s alerts
+
+```text
+GET /api/alerting/rules/_find?search=k8s&search_fields=tags&filter=alert.attributes.executionStatus.status:active
+```
+
+## Examples
+
+### "Why is my pod CrashLoopBackOff-ing?"
+
+Characterize first: get restart count, termination reason, memory and CPU utilization.
+
+- If `last_terminated_reason == "OOMKilled"` and memory utilization hit 1.0 → memory path. Corroborate with 7-day
+ baseline: monotonic rise over days = leak; spiky = load-driven. Check GC metrics if language is known.
+- If `last_terminated_reason == "Error"` and `cpu_limit_utilization > 1.0` → CPU throttling path. Corroborate with
+ liveness probe config (initialDelaySeconds, timeoutSeconds) and K8s events for `Unhealthy`.
+- If `last_terminated_reason == "Error"` and CPU is fine → application-logic path. Pull recent logs before termination.
+- If `last_terminated_reason == "ContainerCannotRun"` → image/exec path. Check K8s events for `Failed` pull events.
+
+Synthesize with appropriate confidence. If logs were unavailable on the Error path, downgrade to medium and say so.
+
+### "Is my rollout stuck?"
+
+Authoritative signal: `k8s.deployment.available < k8s.deployment.desired` for > 10 minutes.
+
+Diagnose the constraint:
+
+- K8s events on the new ReplicaSet: `FailedCreate` → admission rejection (quota, webhook, PSP). `FailedScheduling` → no
+ node fits.
+- New-pod utilization: all at 0% memory → never started (image pull failure); high CPU with low memory → slow startup
+ hitting readiness probe.
+- HPA state: stable `current_replicas < desired_replicas` under load → unready-pod dampening.
+
+### "Alert fired but everything looks healthy"
+
+Possible and worth naming explicitly. Check:
+
+- Has the symptom resolved? Compare current utilization/restart rate to the alert trigger point.
+- Was the alert a transient spike that's already decayed?
+- Is the alert tuned appropriately (e.g., too-short evaluation window)?
+
+Output: `ALERT FIRED BUT SYSTEM APPEARS HEALTHY` with what you checked. Recommend alert tuning if the pattern is
+recurrent.
+
+## Related
+
+- **Workflow:** `K8s CrashLoopBackOff Investigation` — alert-triggered automated version of the pod-level path above.
+ Runs deterministic ESQL + branches; this skill provides the interpretation layer the workflow lacks.
+- **Forge genome library:** 16 K8s failure scenarios (OOMKill cascade, CPU throttling, probe misconfig, node NotReady,
+ admission webhook block, etc.) validating this skill's coverage.