fix(o11y): vendor the etcd dashboard so its job variable stops posing as the cluster picker (#351)

This commit is contained in:
bo0tzz
2026-09-23 10:32:02 -04:00
committed by GitHub
parent dc64062dd5
commit b8ec0354b7
3 changed files with 9 additions and 3 deletions
+5 -1
View File
@@ -87,7 +87,7 @@ run = "markdownlint-cli2 --fix 'docs/**/*.md' 'README.md'"
description = "Auto-fix markdown lint issues where possible"
[tasks."o11y:vendor"]
description = "Fetch upstream boards that need a rewrite before the sync-job sees them (CloudNativePG: its `cluster` means the Postgres cluster; the fleet keeps that under `pg_cluster`)"
description = "Fetch upstream boards that need a rewrite before the sync-job sees them (CloudNativePG: its `cluster` means the Postgres cluster, kept as `pg_cluster`; etcd: its `cluster` variable holds job names, kept as `job`)"
run = """
mkdir -p {{config_root}}/o11y/vendor
curl -fsSL https://raw.githubusercontent.com/cloudnative-pg/grafana-dashboards/cluster-v0.0.5/charts/cluster/grafana-dashboard.json \\
@@ -97,6 +97,10 @@ curl -fsSL https://raw.githubusercontent.com/cloudnative-pg/grafana-dashboards/c
| sd '"name":\\s*"cluster"' '"name": "pg_cluster"' \\
| sd -s '\\\\bcluster\\\\b=' '\\\\bpg_cluster\\\\b=' \\
> {{config_root}}/o11y/vendor/cloudnativepg.json
curl -fsSL https://raw.githubusercontent.com/monitoring-mixins/website/master/assets/etcd/dashboards/etcd.json \\
| sd '\\$cluster\\b' '$$job' \\
| sd '"name":\\s*"cluster"' '"name": "job"' \\
> {{config_root}}/o11y/vendor/etcd.json
"""
[tasks."o11y:render"]
+1 -1
View File
@@ -10,7 +10,7 @@ Cluster-generic dashboards every tenant used to ship in its own bundle, provided
mise run //:o11y:render # render manifests/dashboards.yaml locally for review (git-ignored)
```
Both tasks first run `o11y:vendor`, which fetches boards that need a rewrite the sync-job cannot express and writes them to `vendor/` as local sources. Today that is CloudNativePG only: upstream uses `cluster` to mean the Postgres cluster, while the fleet keeps that name under `pg_cluster` and reserves `cluster` for the Kubernetes cluster, so the vendor step renames the label and variable before the sync-job adds the fleet's `$cluster`. A cluster's CNPG series must carry `pg_cluster` for the board to list its databases; the shipping guide's identity-label section shows the relabel rule, which o11y, azad and harbor apply.
Both tasks first run `o11y:vendor`, which fetches boards that need a rewrite the sync-job cannot express and writes them to `vendor/` as local sources. Two boards need it. CloudNativePG uses `cluster` to mean the Postgres cluster, while the fleet keeps that name under `pg_cluster` and reserves `cluster` for the Kubernetes cluster, so the vendor step renames the label and variable before the sync-job adds the fleet's `$cluster`. The etcd mixin names its variable `cluster` but fills it from the `job` label and filters on `job="$cluster"`; left alone, the sync-job takes that variable for the fleet's and rewrites the filter to `cluster=~"$cluster"`, so the picker offers job names and every panel is empty. The vendor step renames that variable to `job`. A cluster's CNPG series must carry `pg_cluster` for the board to list its databases; the shipping guide's identity-label section shows the relabel rule, which o11y, azad and harbor apply.
The set is everything at least two clusters run: the dotdc Kubernetes views and system boards, the kube-prometheus mixin boards (kubelet, scheduler, controller manager, proxy, API server, compute resources, networking, persistent volumes, node exporter USE method), Node Exporter Full, etcd, vmagent, Cilium and Hubble, Flux, Spegel, Envoy Gateway and CloudNativePG. Windows, AIX, macOS, Prometheus, Alertmanager and Grafana-overview boards from the mixin bundle are disabled. Documents are sorted by name so an unchanged upstream renders byte-identical: the artifact's content layer keeps its digest across publishes, and only the manifest's revision and timestamp annotations change, so Flux's re-apply after a publish with no upstream change is a no-op.
+3 -1
View File
@@ -37,6 +37,8 @@ dashboards:
clusterMetric: spegel_mirror_requests_total
cloudnativepg:
clusterMetric: cnpg_collector_up
etcd:
clusterMetric: etcd_server_has_leader
prometheus:
enabled: false
prometheus-remote-write:
@@ -75,7 +77,7 @@ dashboards:
- url: https://raw.githubusercontent.com/fluxcd/flux2-monitoring-example/7ab65dc8b90f/monitoring/configs/dashboards/control-plane.json
- url: https://raw.githubusercontent.com/spegel-org/spegel/v0.7.4/charts/spegel/monitoring/grafana-dashboard.json
- url: https://raw.githubusercontent.com/prometheus-operator/kube-prometheus/main/manifests/grafana-dashboardDefinitions.yaml
- url: https://raw.githubusercontent.com/monitoring-mixins/website/master/assets/etcd/dashboards/etcd.json
- url: /config/vendor/etcd.json
- url: /config/vendor/cloudnativepg.json
- url: https://grafana.com/api/dashboards/24459/revisions/3/download
- url: https://grafana.com/api/dashboards/24457/revisions/4/download