From 1a2e8d91a4e75d389621c69b105bcfeaeae34568 Mon Sep 17 00:00:00 2001 From: slorello89 Date: Wed, 26 Aug 2026 09:59:54 -0400 Subject: [PATCH] Fix redis_enterprise_prometheus dashboard queries and collect two missing metrics Datadog reported several default-dashboard widgets returning no data. Verified each against a live Redis Enterprise 8.0.10 /v2 endpoint. Check: - Collect node_uname_info and node_config. Both were already documented in metadata.csv but absent from the metric map, so they never emitted. node_uname_info carries the per-node `nodename` label, which is the internal hostname the Node dashboard needs; node_config carries `rs_version`. Dashboards: - Database List: database_syncer_config -> db_config. The former is a config placeholder that only exists on Active-Active clusters and sits in the opt-in REDIS2.REPLICATION group, so the widget was empty by default. - endpoint_ingress / endpoint_egress: add the missing `.count` suffix. These are Prometheus counters and OpenMetrics v2 emits them as `.count`. - Connections: compute from the `.total` gauges and subtract endpoint_proxy_disconnections.total, instead of differencing monotonic deltas. - Remove double rate division in Database Input/Output, Shard Process CPU and Proxy Threads CLI Session, which wrapped `.as_rate()` queries in per_second()/derivative(). - Un-swap the ingress/egress series labels and fix the note text describing them. - Node Latency: rebuild as a calculated field over the endpoint_*_requests_latency_histogram sum/count pairs (microseconds -> ms). It previously queried rdse.node_avg_latency, a V1 metric under the wrong namespace. - Cluster Nodes: group by `nodename` rather than the non-existent `internal-hostname` tag. - Remove Node Network Traffic; both rdse.node_{in,e}gress_bytes_median are V1-only with no V2 equivalent. - Drop `title` from note widgets, folding it into the content as a markdown heading. The Dashboards API rejects `title` on notes, which made every one of these dashboard assets fail to import. Co-Authored-By: Claude Opus 5 --- redis_enterprise_prometheus/CHANGELOG.md | 18 +++ ...s_enterprise-prometheus_active-active.json | 3 +- .../redis_enterprise-prometheus_database.json | 18 ++- .../redis_enterprise-prometheus_node.json | 110 +++++++----------- .../redis_enterprise-prometheus_overview.json | 57 +++++---- ...s_enterprise-prometheus_proxy-threads.json | 2 +- .../redis_enterprise-prometheus_shard.json | 4 +- .../redis_enterprise_prometheus/__about__.py | 2 +- .../redis_enterprise_prometheus/metrics.py | 2 + redis_enterprise_prometheus/tests/support.py | 2 + 10 files changed, 109 insertions(+), 109 deletions(-) diff --git a/redis_enterprise_prometheus/CHANGELOG.md b/redis_enterprise_prometheus/CHANGELOG.md index 535e54ec8..470454faa 100644 --- a/redis_enterprise_prometheus/CHANGELOG.md +++ b/redis_enterprise_prometheus/CHANGELOG.md @@ -1,5 +1,23 @@ # CHANGELOG - Redis Enterprise Prometheus +## 1.2.0 / 2026-08-26 + +***Added***: + +* Collect `rdse2.node_uname_info` and `rdse2.node_config`. Both were already documented in `metadata.csv` but were missing from the check's metric map, so they were never emitted. `node_uname_info` carries the per-node `nodename` label (the internal hostname) and `node_config` carries `rs_version`. [#3136](https://github.com/DataDog/integrations-extras/pull/3136) + +***Fixed***: + +* Point the Database List widgets at `rdse2.db_config` instead of `rdse2.database_syncer_config`. The latter is a configuration-label placeholder that only exists on Active-Active deployments and lives in the opt-in `REDIS2.REPLICATION` group, so the widget was empty on a default install. [#3136](https://github.com/DataDog/integrations-extras/pull/3136) +* Add the missing `.count` suffix to the `rdse2.endpoint_ingress` / `rdse2.endpoint_egress` queries in the Database Input/Output widget. These are Prometheus counters, which the OpenMetrics v2 check emits as `.count`. [#3136](https://github.com/DataDog/integrations-extras/pull/3136) +* Compute current connections from the `.total` gauges and subtract `rdse2.endpoint_proxy_disconnections.total`, rather than differencing two monotonic `.count` deltas. [#3136](https://github.com/DataDog/integrations-extras/pull/3136) +* Stop dividing by time twice in the Database Input/Output, Shard Process CPU, and Proxy Threads CLI Session widgets, which wrapped an already-rated (`.as_rate()`) query in `per_second()` / `derivative()`. [#3136](https://github.com/DataDog/integrations-extras/pull/3136) +* Un-swap the ingress and egress series labels (and the accompanying note text) in the Database Input/Output widget. [#3136](https://github.com/DataDog/integrations-extras/pull/3136) +* Rebuild the Node Latency widget as a calculated field over the `endpoint_*_requests_latency_histogram` sum/count pairs. It previously queried `rdse.node_avg_latency`, a V1-only metric under the wrong namespace. [#3136](https://github.com/DataDog/integrations-extras/pull/3136) +* Group the Cluster Nodes widget by `nodename` from `rdse2.node_uname_info` instead of the non-existent `internal-hostname` tag. [#3136](https://github.com/DataDog/integrations-extras/pull/3136) +* Remove the Node Network Traffic widget; both `rdse.node_egress_bytes_median` and `rdse.node_ingress_bytes_median` are V1-only metrics with no V2 equivalent. [#3136](https://github.com/DataDog/integrations-extras/pull/3136) +* Remove `title` from `note` widgets, folding it into the note content as a markdown heading. The Dashboards API rejects `title` on notes, which made the dashboard assets fail to import. [#3136](https://github.com/DataDog/integrations-extras/pull/3136) + ## 1.1.0 / 2026-07-29 ***Deprecated***: diff --git a/redis_enterprise_prometheus/assets/dashboards/redis_enterprise-prometheus_active-active.json b/redis_enterprise_prometheus/assets/dashboards/redis_enterprise-prometheus_active-active.json index b64574868..ae9ed480d 100644 --- a/redis_enterprise_prometheus/assets/dashboards/redis_enterprise-prometheus_active-active.json +++ b/redis_enterprise_prometheus/assets/dashboards/redis_enterprise-prometheus_active-active.json @@ -231,8 +231,7 @@ "id": 6950044907570933, "definition": { "type": "note", - "title": "About This Active Active Dashboard", - "content": "The Redis Enterprise Shard Dashboard is used to visualize metrics at the shard level. In Redis Enterprise Software, the shard is the lowest single level of abstraction tracked by metrics. Key metrics such as memory usage, connection count, cpu utilization are all tracked by this dashboard.\n \nPlease note, in order to view the metrics in this dashboard you must enable `REDIS2.REPLICATION` in the `extra_metrics` section of the integration configuration.", + "content": "## About This Active Active Dashboard\n\nThe Redis Enterprise Shard Dashboard is used to visualize metrics at the shard level. In Redis Enterprise Software, the shard is the lowest single level of abstraction tracked by metrics. Key metrics such as memory usage, connection count, cpu utilization are all tracked by this dashboard.\n \nPlease note, in order to view the metrics in this dashboard you must enable `REDIS2.REPLICATION` in the `extra_metrics` section of the integration configuration.", "background_color": "white", "font_size": "14", "text_align": "left", diff --git a/redis_enterprise_prometheus/assets/dashboards/redis_enterprise-prometheus_database.json b/redis_enterprise_prometheus/assets/dashboards/redis_enterprise-prometheus_database.json index b5cbaa57c..c9032a809 100644 --- a/redis_enterprise_prometheus/assets/dashboards/redis_enterprise-prometheus_database.json +++ b/redis_enterprise_prometheus/assets/dashboards/redis_enterprise-prometheus_database.json @@ -15,8 +15,7 @@ "id": 6906627780487202, "definition": { "type": "note", - "title": "About The Database Dashboard", - "content": "The Database dashboard provides a visualization of metrics at the Redis Enterprise database level. In Redis Enterprise, a database is an abstraction over a collection of Redis Shards fronted by a single endpoint (or several when using the OSS cluster API). Key metrics such as memory usage, connections and latency are observable at the Database level.", + "content": "## About The Database Dashboard\n\nThe Database dashboard provides a visualization of metrics at the Redis Enterprise database level. In Redis Enterprise, a database is an abstraction over a collection of Redis Shards fronted by a single endpoint (or several when using the OSS cluster API). Key metrics such as memory usage, connections and latency are observable at the Database level.", "background_color": "vivid_purple", "font_size": "16", "text_align": "left", @@ -120,7 +119,7 @@ { "data_source": "metrics", "name": "query1", - "query": "avg:rdse2.database_syncer_config{*} by {cluster,db_name,db}", + "query": "avg:rdse2.db_config{*} by {cluster,db_name,db}", "aggregator": "avg" } ], @@ -285,19 +284,26 @@ { "data_source": "metrics", "name": "query1", - "query": "sum:rdse2.endpoint_client_connections{$cluster,$database}", + "query": "sum:rdse2.endpoint_client_connections.total{$cluster,$database}", "aggregator": "last" }, { "data_source": "metrics", "name": "query2", - "query": "sum:rdse2.endpoint_client_disconnections{$cluster,$database}", + "query": "sum:rdse2.endpoint_client_disconnections.total{$cluster,$database}", + "aggregator": "last" + }, + { + "data_source": "metrics", + "name": "query3", + "query": "sum:rdse2.endpoint_proxy_disconnections.total{$cluster,$database}", "aggregator": "last" } ], "formulas": [ { - "formula": "query1 - query2" + "formula": "query1 - query2 - query3", + "alias": "current connections" } ] } diff --git a/redis_enterprise_prometheus/assets/dashboards/redis_enterprise-prometheus_node.json b/redis_enterprise_prometheus/assets/dashboards/redis_enterprise-prometheus_node.json index c7f405fc5..1451f3036 100644 --- a/redis_enterprise_prometheus/assets/dashboards/redis_enterprise-prometheus_node.json +++ b/redis_enterprise_prometheus/assets/dashboards/redis_enterprise-prometheus_node.json @@ -15,8 +15,7 @@ "id": 6906627780487202, "definition": { "type": "note", - "title": "About The Node Dashboard", - "content": "This Dashboard provides visualization of metrics at the Redis Enterprise Node level. In Redis Enterprise, a node is a single server with the Redis Enterprise Software installed on it. Memory usage, cpu usage, as well as key shard details per node are all visualized in the Node dashboard.", + "content": "## About The Node Dashboard\n\nThis Dashboard provides visualization of metrics at the Redis Enterprise Node level. In Redis Enterprise, a node is a single server with the Redis Enterprise Software installed on it. Memory usage, cpu usage, as well as key shard details per node are all visualized in the Node dashboard.", "background_color": "vivid_purple", "font_size": "16", "text_align": "left", @@ -115,7 +114,7 @@ { "data_source": "metrics", "name": "query1", - "query": "avg:rdse2.node_os_info{$cluster, $node} by {internal-hostname,node}", + "query": "avg:rdse2.node_uname_info{$cluster, $node} by {node,nodename}", "aggregator": "avg" } ], @@ -207,9 +206,6 @@ "title": "Node Latency", "title_size": "16", "title_align": "left", - "time": { - "hide_incomplete_cost_data": true - }, "type": "query_value", "requests": [ { @@ -217,14 +213,51 @@ "queries": [ { "data_source": "metrics", + "aggregator": "sum", "name": "query1", - "query": "avg:rdse.node_avg_latency{$cluster, $node}", - "aggregator": "avg" + "query": "sum:rdse2.endpoint_read_requests_latency_histogram.sum{$cluster, $node}.as_rate()" + }, + { + "data_source": "metrics", + "aggregator": "sum", + "name": "query2", + "query": "sum:rdse2.endpoint_read_requests_latency_histogram.count{$cluster, $node}.as_rate()" + }, + { + "data_source": "metrics", + "aggregator": "sum", + "name": "query3", + "query": "sum:rdse2.endpoint_write_requests_latency_histogram.sum{$cluster, $node}.as_rate()" + }, + { + "data_source": "metrics", + "aggregator": "sum", + "name": "query4", + "query": "sum:rdse2.endpoint_write_requests_latency_histogram.count{$cluster, $node}.as_rate()" + }, + { + "data_source": "metrics", + "aggregator": "sum", + "name": "query5", + "query": "sum:rdse2.endpoint_other_requests_latency_histogram.sum{$cluster, $node}.as_rate()" + }, + { + "data_source": "metrics", + "aggregator": "sum", + "name": "query6", + "query": "sum:rdse2.endpoint_other_requests_latency_histogram.count{$cluster, $node}.as_rate()" } ], "formulas": [ { - "formula": "query1" + "formula": "(query1 + query3 + query5) / (query2 + query4 + query6) / 1000", + "alias": "avg latency", + "number_format": { + "unit": { + "type": "canonical_unit", + "unit_name": "millisecond" + } + } } ] } @@ -709,63 +742,6 @@ "type": "group", "layout_type": "ordered", "widgets": [ - { - "id": 7788477609382880, - "definition": { - "title": "Node Network Traffic", - "title_size": "16", - "title_align": "left", - "show_legend": true, - "legend_layout": "auto", - "legend_columns": [ - "avg", - "min", - "max", - "value", - "sum" - ], - "type": "timeseries", - "requests": [ - { - "formulas": [ - { - "alias": "egress", - "formula": "query1" - }, - { - "alias": "ingress", - "formula": "query2" - } - ], - "queries": [ - { - "data_source": "metrics", - "name": "query1", - "query": "avg:rdse.node_egress_bytes_median{$cluster, $node}" - }, - { - "data_source": "metrics", - "name": "query2", - "query": "avg:rdse.node_ingress_bytes_median{$cluster, $node}" - } - ], - "response_format": "timeseries", - "style": { - "palette": "blue", - "line_type": "solid", - "line_width": "normal" - }, - "display_type": "line" - } - ] - }, - "layout": { - "x": 0, - "y": 0, - "width": 12, - "height": 4 - } - }, { "id": 4265268039928674, "definition": { @@ -1000,4 +976,4 @@ "layout_type": "ordered", "notify_list": [], "reflow_type": "fixed" -} \ No newline at end of file +} diff --git a/redis_enterprise_prometheus/assets/dashboards/redis_enterprise-prometheus_overview.json b/redis_enterprise_prometheus/assets/dashboards/redis_enterprise-prometheus_overview.json index 930480c84..6f4b617cf 100644 --- a/redis_enterprise_prometheus/assets/dashboards/redis_enterprise-prometheus_overview.json +++ b/redis_enterprise_prometheus/assets/dashboards/redis_enterprise-prometheus_overview.json @@ -15,8 +15,7 @@ "id": 6544225029855812, "definition": { "type": "note", - "title": "About This Overview Dashboard", - "content": "The Redis Enterprise Overview Dashboard provides a visualization of key Redis metrics at the cluster level. A cluster in Redis Enterprise is a collection of nodes with Redis Enterprise Software installed on them which may contain a number of Redis Enterprise Databases, which may each contain one or more shards. Metrics such as throughput, used memory, key counts, node counts, and latency are all captured at the top cluster level in this dashboard.", + "content": "## About This Overview Dashboard\n\nThe Redis Enterprise Overview Dashboard provides a visualization of key Redis metrics at the cluster level. A cluster in Redis Enterprise is a collection of nodes with Redis Enterprise Software installed on them which may contain a number of Redis Enterprise Databases, which may each contain one or more shards. Metrics such as throughput, used memory, key counts, node counts, and latency are all captured at the top cluster level in this dashboard.", "background_color": "white", "font_size": "14", "text_align": "left", @@ -46,7 +45,7 @@ { "data_source": "metrics", "name": "query1", - "query": "avg:rdse2.database_syncer_config{$cluster} by {cluster,db_name}", + "query": "avg:rdse2.db_config{$cluster} by {cluster,db_name}", "aggregator": "avg" } ], @@ -90,8 +89,7 @@ "id": 7032118608350450, "definition": { "type": "note", - "title": "About This Integration", - "content": "This integration makes it possible to:\n\n- Collect and display metrics not available in the admin console\n\n- Set up automatic alerts for node or cluster events\n\n- Display these metrics alongside data from other systems", + "content": "## About This Integration\n\nThis integration makes it possible to:\n\n- Collect and display metrics not available in the admin console\n\n- Set up automatic alerts for node or cluster events\n\n- Display these metrics alongside data from other systems", "background_color": "white", "font_size": "14", "text_align": "left", @@ -272,8 +270,7 @@ "id": 1508854872890636, "definition": { "type": "note", - "title": "About the Latency Section", - "content": "Latency is the length of time it takes for the system to respond.", + "content": "## About the Latency Section\n\nLatency is the length of time it takes for the system to respond.", "background_color": "orange", "font_size": "14", "text_align": "left", @@ -408,8 +405,7 @@ "id": 5205778284461508, "definition": { "type": "note", - "title": "About the Memory Section", - "content": "Memory limits how much data Redis can store.\n", + "content": "## About the Memory Section\n\nMemory limits how much data Redis can store.\n", "background_color": "blue", "font_size": "14", "text_align": "left", @@ -503,8 +499,7 @@ "id": 4698654642100834, "definition": { "type": "note", - "title": "About the Keys Section", - "content": "Keys represent individual pieces of data in Redis. Here we take a look at the cumulative Redis Keyspace for your cluster\n\n", + "content": "## About the Keys Section\n\nKeys represent individual pieces of data in Redis. Here we take a look at the cumulative Redis Keyspace for your cluster\n\n", "background_color": "green", "font_size": "14", "text_align": "left", @@ -592,8 +587,7 @@ "id": 3606284556444836, "definition": { "type": "note", - "title": "About the Requests Section", - "content": "Requests represent read, write, and all other requests. In this section we look at cluster level throughput metrics", + "content": "## About the Requests Section\n\nRequests represent read, write, and all other requests. In this section we look at cluster level throughput metrics", "background_color": "yellow", "font_size": "14", "text_align": "left", @@ -717,8 +711,7 @@ "id": 6536426319322986, "definition": { "type": "note", - "title": "About the Database Section", - "content": "The input, and/or output, of a database is the measure of how much information is being processed.", + "content": "## About the Database Section\n\nThe input, and/or output, of a database is the measure of how much information is being processed.", "background_color": "pink", "font_size": "14", "text_align": "left", @@ -755,36 +748,36 @@ { "formulas": [ { - "alias": "egress", + "alias": "ingress", "number_format": { "unit": { "type": "canonical_unit", "unit_name": "byte" } }, - "formula": "per_second(query1)" + "formula": "query1" }, { - "alias": "ingress", + "alias": "egress", "number_format": { "unit": { "type": "canonical_unit", "unit_name": "byte" } }, - "formula": "per_second(query2)" + "formula": "query2" } ], "queries": [ { "data_source": "metrics", "name": "query1", - "query": "avg:rdse2.endpoint_ingress{$cluster} by {db}.as_count()" + "query": "sum:rdse2.endpoint_ingress.count{$cluster} by {db}.as_rate()" }, { "data_source": "metrics", "name": "query2", - "query": "avg:rdse2.endpoint_egress{$cluster} by {db}.as_count()" + "query": "sum:rdse2.endpoint_egress.count{$cluster} by {db}.as_rate()" } ], "response_format": "timeseries", @@ -807,9 +800,8 @@ { "id": 457640250423346, "definition": { - "title":"Ingress", "type": "note", - "content": "- ingress - Rate of outgoing network traffic from the DB\n- egress - Rate of incoming network traffic to DB", + "content": "## Ingress\n\n- ingress - Rate of incoming network traffic to the DB\n- egress - Rate of outgoing network traffic from the DB", "background_color": "vivid_pink", "font_size": "12", "text_align": "left", @@ -840,19 +832,26 @@ { "data_source": "metrics", "name": "query1", - "query": "sum:rdse2.endpoint_client_connections{*}", + "query": "sum:rdse2.endpoint_client_connections.total{*}", "aggregator": "avg" }, { "data_source": "metrics", "name": "query2", - "query": "sum:rdse2.endpoint_client_disconnections{*}", + "query": "sum:rdse2.endpoint_client_disconnections.total{*}", + "aggregator": "avg" + }, + { + "data_source": "metrics", + "name": "query3", + "query": "sum:rdse2.endpoint_proxy_disconnections.total{*}", "aggregator": "avg" } ], "formulas": [ { - "formula": "query1 - query2" + "formula": "query1 - query2 - query3", + "alias": "current connections" } ] } @@ -895,8 +894,7 @@ "id": 5309930525296140, "definition": { "type": "note", - "title": "About the Shard Section", - "content": "The individual partitions of a clustered database and the keys they contain can tells us if they are unbalanced.", + "content": "## About the Shard Section\n\nThe individual partitions of a clustered database and the keys they contain can tells us if they are unbalanced.", "background_color": "pink", "font_size": "14", "text_align": "left", @@ -983,8 +981,7 @@ "id": 5424811187083390, "definition": { "type": "note", - "title": "Shard Keys", - "content": "- expired - Rate keys expired in DB\n- evicted - Rate of key evictions from DB\n- trimmed - The number of keys that were trimmed in the current or last resharding process", + "content": "## Shard Keys\n\n- expired - Rate keys expired in DB\n- evicted - Rate of key evictions from DB\n- trimmed - The number of keys that were trimmed in the current or last resharding process", "background_color": "vivid_pink", "font_size": "12", "text_align": "left", diff --git a/redis_enterprise_prometheus/assets/dashboards/redis_enterprise-prometheus_proxy-threads.json b/redis_enterprise_prometheus/assets/dashboards/redis_enterprise-prometheus_proxy-threads.json index 013d25d6a..06497a516 100644 --- a/redis_enterprise_prometheus/assets/dashboards/redis_enterprise-prometheus_proxy-threads.json +++ b/redis_enterprise_prometheus/assets/dashboards/redis_enterprise-prometheus_proxy-threads.json @@ -540,7 +540,7 @@ "unit_name": "fraction" } }, - "formula": "derivative(query2)" + "formula": "query2" } ], "queries": [ diff --git a/redis_enterprise_prometheus/assets/dashboards/redis_enterprise-prometheus_shard.json b/redis_enterprise_prometheus/assets/dashboards/redis_enterprise-prometheus_shard.json index 600b6d4cb..341dd2be5 100644 --- a/redis_enterprise_prometheus/assets/dashboards/redis_enterprise-prometheus_shard.json +++ b/redis_enterprise_prometheus/assets/dashboards/redis_enterprise-prometheus_shard.json @@ -927,7 +927,7 @@ } }, "alias": "system", - "formula": "per_second(query1)" + "formula": "query1" }, { "alias": "user", @@ -937,7 +937,7 @@ "unit_name": "fraction" } }, - "formula": "per_second(query2)" + "formula": "query2" } ], "queries": [ diff --git a/redis_enterprise_prometheus/datadog_checks/redis_enterprise_prometheus/__about__.py b/redis_enterprise_prometheus/datadog_checks/redis_enterprise_prometheus/__about__.py index 6849410aa..c68196d1c 100644 --- a/redis_enterprise_prometheus/datadog_checks/redis_enterprise_prometheus/__about__.py +++ b/redis_enterprise_prometheus/datadog_checks/redis_enterprise_prometheus/__about__.py @@ -1 +1 @@ -__version__ = "1.1.0" +__version__ = "1.2.0" diff --git a/redis_enterprise_prometheus/datadog_checks/redis_enterprise_prometheus/metrics.py b/redis_enterprise_prometheus/datadog_checks/redis_enterprise_prometheus/metrics.py index 3c5890c2d..ec37ca424 100644 --- a/redis_enterprise_prometheus/datadog_checks/redis_enterprise_prometheus/metrics.py +++ b/redis_enterprise_prometheus/datadog_checks/redis_enterprise_prometheus/metrics.py @@ -142,9 +142,11 @@ } REDIS_INFO = { + "node_config": "node_config", "node_disk_info": "node_disk_info", "node_dmi_info": "node_dmi_info", "node_os_info": "node_os_info", + "node_uname_info": "node_uname_info", } ### END DEFAULT diff --git a/redis_enterprise_prometheus/tests/support.py b/redis_enterprise_prometheus/tests/support.py index 25a4785fc..daae284a0 100644 --- a/redis_enterprise_prometheus/tests/support.py +++ b/redis_enterprise_prometheus/tests/support.py @@ -197,9 +197,11 @@ "rdse2.redis_server_used_memory", ], "REDIS2.INFO": [ + "rdse2.node_config", "rdse2.node_dmi_info", "rdse2.node_os_info", "rdse2.node_disk_info", + "rdse2.node_uname_info", ], # END DEFAULT "REDIS2.REPLICATION": [