From b8ec0354b75847d0013885f04e99da592959d8ad Mon Sep 17 00:00:00 2001 From: bo0tzz Date: Wed, 23 Sep 2026 16:32:02 +0200 Subject: [PATCH] fix(o11y): vendor the etcd dashboard so its job variable stops posing as the cluster picker (#351) --- .mise/config.toml | 6 +++++- o11y/README.md | 2 +- o11y/sync-job.yaml | 4 +++- 3 files changed, 9 insertions(+), 3 deletions(-) diff --git a/.mise/config.toml b/.mise/config.toml index 23e8326..4772fbb 100644 --- a/.mise/config.toml +++ b/.mise/config.toml @@ -87,7 +87,7 @@ run = "markdownlint-cli2 --fix 'docs/**/*.md' 'README.md'" description = "Auto-fix markdown lint issues where possible" [tasks."o11y:vendor"] -description = "Fetch upstream boards that need a rewrite before the sync-job sees them (CloudNativePG: its `cluster` means the Postgres cluster; the fleet keeps that under `pg_cluster`)" +description = "Fetch upstream boards that need a rewrite before the sync-job sees them (CloudNativePG: its `cluster` means the Postgres cluster, kept as `pg_cluster`; etcd: its `cluster` variable holds job names, kept as `job`)" run = """ mkdir -p {{config_root}}/o11y/vendor curl -fsSL https://raw.githubusercontent.com/cloudnative-pg/grafana-dashboards/cluster-v0.0.5/charts/cluster/grafana-dashboard.json \\ @@ -97,6 +97,10 @@ curl -fsSL https://raw.githubusercontent.com/cloudnative-pg/grafana-dashboards/c | sd '"name":\\s*"cluster"' '"name": "pg_cluster"' \\ | sd -s '\\\\bcluster\\\\b=' '\\\\bpg_cluster\\\\b=' \\ > {{config_root}}/o11y/vendor/cloudnativepg.json +curl -fsSL https://raw.githubusercontent.com/monitoring-mixins/website/master/assets/etcd/dashboards/etcd.json \\ + | sd '\\$cluster\\b' '$$job' \\ + | sd '"name":\\s*"cluster"' '"name": "job"' \\ + > {{config_root}}/o11y/vendor/etcd.json """ [tasks."o11y:render"] diff --git a/o11y/README.md b/o11y/README.md index cd556ac..a598186 100644 --- a/o11y/README.md +++ b/o11y/README.md @@ -10,7 +10,7 @@ Cluster-generic dashboards every tenant used to ship in its own bundle, provided mise run //:o11y:render # render manifests/dashboards.yaml locally for review (git-ignored) ``` -Both tasks first run `o11y:vendor`, which fetches boards that need a rewrite the sync-job cannot express and writes them to `vendor/` as local sources. Today that is CloudNativePG only: upstream uses `cluster` to mean the Postgres cluster, while the fleet keeps that name under `pg_cluster` and reserves `cluster` for the Kubernetes cluster, so the vendor step renames the label and variable before the sync-job adds the fleet's `$cluster`. A cluster's CNPG series must carry `pg_cluster` for the board to list its databases; the shipping guide's identity-label section shows the relabel rule, which o11y, azad and harbor apply. +Both tasks first run `o11y:vendor`, which fetches boards that need a rewrite the sync-job cannot express and writes them to `vendor/` as local sources. Two boards need it. CloudNativePG uses `cluster` to mean the Postgres cluster, while the fleet keeps that name under `pg_cluster` and reserves `cluster` for the Kubernetes cluster, so the vendor step renames the label and variable before the sync-job adds the fleet's `$cluster`. The etcd mixin names its variable `cluster` but fills it from the `job` label and filters on `job="$cluster"`; left alone, the sync-job takes that variable for the fleet's and rewrites the filter to `cluster=~"$cluster"`, so the picker offers job names and every panel is empty. The vendor step renames that variable to `job`. A cluster's CNPG series must carry `pg_cluster` for the board to list its databases; the shipping guide's identity-label section shows the relabel rule, which o11y, azad and harbor apply. The set is everything at least two clusters run: the dotdc Kubernetes views and system boards, the kube-prometheus mixin boards (kubelet, scheduler, controller manager, proxy, API server, compute resources, networking, persistent volumes, node exporter USE method), Node Exporter Full, etcd, vmagent, Cilium and Hubble, Flux, Spegel, Envoy Gateway and CloudNativePG. Windows, AIX, macOS, Prometheus, Alertmanager and Grafana-overview boards from the mixin bundle are disabled. Documents are sorted by name so an unchanged upstream renders byte-identical: the artifact's content layer keeps its digest across publishes, and only the manifest's revision and timestamp annotations change, so Flux's re-apply after a publish with no upstream change is a no-op. diff --git a/o11y/sync-job.yaml b/o11y/sync-job.yaml index 6bab5e4..5e3083a 100644 --- a/o11y/sync-job.yaml +++ b/o11y/sync-job.yaml @@ -37,6 +37,8 @@ dashboards: clusterMetric: spegel_mirror_requests_total cloudnativepg: clusterMetric: cnpg_collector_up + etcd: + clusterMetric: etcd_server_has_leader prometheus: enabled: false prometheus-remote-write: @@ -75,7 +77,7 @@ dashboards: - url: https://raw.githubusercontent.com/fluxcd/flux2-monitoring-example/7ab65dc8b90f/monitoring/configs/dashboards/control-plane.json - url: https://raw.githubusercontent.com/spegel-org/spegel/v0.7.4/charts/spegel/monitoring/grafana-dashboard.json - url: https://raw.githubusercontent.com/prometheus-operator/kube-prometheus/main/manifests/grafana-dashboardDefinitions.yaml - - url: https://raw.githubusercontent.com/monitoring-mixins/website/master/assets/etcd/dashboards/etcd.json + - url: /config/vendor/etcd.json - url: /config/vendor/cloudnativepg.json - url: https://grafana.com/api/dashboards/24459/revisions/3/download - url: https://grafana.com/api/dashboards/24457/revisions/4/download