mirror of
https://github.com/immich-app/yucca-o11y.git
synced 2026-09-30 13:23:23 +08:00
ci(o11y): render the shared dashboards in CI instead of committing them (#325)
Signed-off-by: Devin Buhl <devin@buhl.casa>
This commit is contained in:
@@ -1,8 +1,9 @@
|
||||
---
|
||||
# yaml-language-server: $schema=https://json.schemastore.org/github-workflow.json
|
||||
# Validates the shared dashboards render on PRs and publishes o11y/manifests
|
||||
# from main as a signed OCI artifact (ghcr.io/<repo>/o11y-manifests:main), the
|
||||
# same shape the tenant bundles use. See o11y/README.md.
|
||||
# Renders the shared dashboards with the mise tasks (nothing generated is
|
||||
# committed), validates the bundle on PRs, and publishes it from main as a signed
|
||||
# OCI artifact (ghcr.io/<repo>/o11y-manifests:main), the same shape the tenant
|
||||
# bundles use. See o11y/README.md.
|
||||
name: o11y
|
||||
|
||||
on:
|
||||
@@ -48,8 +49,8 @@ jobs:
|
||||
install_args: --locked yq sd kustomize
|
||||
cache: false
|
||||
|
||||
- name: Committed render is current
|
||||
run: mise run //:o11y:check
|
||||
- name: Render
|
||||
run: mise run //:o11y:render
|
||||
|
||||
- name: Bundle builds
|
||||
run: kustomize build o11y/manifests > /dev/null
|
||||
@@ -74,9 +75,12 @@ jobs:
|
||||
- name: Set up mise
|
||||
uses: jdx/mise-action@c2a87611a18de5b3828c5652fe268e992400cb5c # v4.3.0
|
||||
with:
|
||||
install_args: --locked jq flux2 github:sigstore/cosign
|
||||
install_args: --locked yq sd jq flux2 github:sigstore/cosign
|
||||
cache: false
|
||||
|
||||
- name: Render
|
||||
run: mise run //:o11y:render
|
||||
|
||||
- name: Log in to GHCR
|
||||
uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4.6.0
|
||||
with:
|
||||
|
||||
@@ -41,3 +41,7 @@ tmp/
|
||||
mise.local.toml
|
||||
|
||||
.claude/
|
||||
|
||||
# Rendered by `mise run //:o11y:render`; CI renders and publishes, nothing generated is committed
|
||||
o11y/manifests/dashboards.yaml
|
||||
o11y/vendor/
|
||||
|
||||
+4
-7
@@ -41,7 +41,8 @@ sync_job = "docker run --rm -e OUTPUT=- -e CONFIG=/config/sync-job.yaml -e NAMES
|
||||
# sync-job leaves them free, and Grafana would otherwise open a shared board
|
||||
# on the default (tenant-scoped) datasource and show only this cluster.
|
||||
pin_fleet = "yq '(select(.kind == \"GrafanaDashboard\") | .spec.json) |= (fromjson | .templating.list[] |= (select(.type == \"datasource\") |= (.regex = \"/^VictoriaMetrics Fleet$/\" | .current = {\"text\": \"VictoriaMetrics Fleet\", \"value\": \"VictoriaMetricsFleet\"})) | tojson)'"
|
||||
# The sync-job emits the kube-prometheus bundle in map order; sort documents so re-renders diff cleanly.
|
||||
# The sync-job emits the kube-prometheus bundle in map order; sort documents so an unchanged
|
||||
# upstream renders byte-identical and the published artifact keeps its digest.
|
||||
sort_docs = "yq eval-all '[.] | sort_by(.metadata.name) | .[] | splitDoc'"
|
||||
tg = "op run {% if get_env(name='OP_SERVICE_ACCOUNT_TOKEN', default='') == '' %}--account 'team-futo.1password.com' {% endif %}'--env-file={{config_root}}/deployment/.env' -- terragrunt"
|
||||
|
||||
@@ -88,6 +89,7 @@ description = "Auto-fix markdown lint issues where possible"
|
||||
[tasks."o11y:vendor"]
|
||||
description = "Fetch upstream boards that need a rewrite before the sync-job sees them (CloudNativePG: its `cluster` means the Postgres cluster; the fleet keeps that under `pg_cluster`)"
|
||||
run = """
|
||||
mkdir -p {{config_root}}/o11y/vendor
|
||||
curl -fsSL https://raw.githubusercontent.com/cloudnative-pg/grafana-dashboards/cluster-v0.0.5/charts/cluster/grafana-dashboard.json \\
|
||||
| sd '\\$cluster\\b' '$$pg_cluster' \\
|
||||
| sd '\\bcluster(\\s*(?:=~|!~|!=|=)\\s*)' 'pg_cluster$1' \\
|
||||
@@ -100,12 +102,7 @@ curl -fsSL https://raw.githubusercontent.com/cloudnative-pg/grafana-dashboards/c
|
||||
[tasks."o11y:render"]
|
||||
depends = ["o11y:vendor"]
|
||||
run = "{{vars.sync_job}} | {{vars.pin_fleet}} | {{vars.sort_docs}} > {{config_root}}/o11y/manifests/dashboards.yaml"
|
||||
description = "Re-render o11y/manifests/dashboards.yaml from the upstream sources in o11y/sync-job.yaml"
|
||||
|
||||
[tasks."o11y:check"]
|
||||
depends = ["o11y:vendor"]
|
||||
run = "{{vars.sync_job}} | {{vars.pin_fleet}} | {{vars.sort_docs}} | diff -u {{config_root}}/o11y/manifests/dashboards.yaml - && echo 'dashboards.yaml is up to date'"
|
||||
description = "Fail if o11y/manifests/dashboards.yaml differs from a fresh render"
|
||||
description = "Render o11y/manifests/dashboards.yaml (git-ignored) from the upstream sources in o11y/sync-job.yaml"
|
||||
|
||||
[settings]
|
||||
experimental = true
|
||||
|
||||
+4
-5
@@ -4,19 +4,18 @@ Cluster-generic dashboards every tenant used to ship in its own bundle, provided
|
||||
|
||||
## How it is built
|
||||
|
||||
`sync-job.yaml` is a config for the [VictoriaMetrics sync-job](https://github.com/VictoriaMetrics/helm-charts/tree/master/hack/sync-job), the same tool the VictoriaMetrics chart runs in-cluster. Run in generate mode it fetches each upstream dashboard listed under `sources`, renames the cluster label, adds a `cluster=~"$cluster"` filter to selectors and `by (cluster)` to aggregations, builds the `$cluster` variable from the dashboard's `clusterMetric`, points fixed datasource references at `VictoriaMetricsFleet`, and emits `GrafanaDashboard` CRs into `manifests/dashboards.yaml`. A `yq` step then pins every datasource-type variable to the Fleet datasource, which the sync-job leaves free. The output is committed. On every push to `main` that touches it, `.github/workflows/o11y.yml` publishes `manifests/` as the signed OCI artifact `ghcr.io/immich-app/yucca-o11y/o11y-manifests:main` (`flux push artifact` plus keyless cosign), the same shape and naming the tenant bundles use; pull requests run `o11y:check` so a stale render cannot merge. The central cluster consumes it through `kubernetes/apps/base/tenants/shared/bundle.yaml`, alongside the `Shared` folder in `manifests/folder.yaml`.
|
||||
`sync-job.yaml` is a config for the [VictoriaMetrics sync-job](https://github.com/VictoriaMetrics/helm-charts/tree/master/hack/sync-job), the same tool the VictoriaMetrics chart runs in-cluster. Run in generate mode it fetches each upstream dashboard listed under `sources`, renames the cluster label, adds a `cluster=~"$cluster"` filter to selectors and `by (cluster)` to aggregations, builds the `$cluster` variable from the dashboard's `clusterMetric`, points fixed datasource references at `VictoriaMetricsFleet`, and emits `GrafanaDashboard` CRs into `manifests/dashboards.yaml`. A `yq` step then pins every datasource-type variable to the Fleet datasource, which the sync-job leaves free. Nothing generated is committed: `manifests/dashboards.yaml` and `vendor/` are git-ignored. `.github/workflows/o11y.yml` runs the same mise tasks in CI: pull requests render and `kustomize build` the bundle, and every push to `main` that touches `o11y/` renders again and publishes `manifests/` as the signed OCI artifact `ghcr.io/immich-app/yucca-o11y/o11y-manifests:main` (`flux push artifact` plus keyless cosign), the same shape and naming the tenant bundles use. Upstream changes reach the clusters on the next publish, so re-run the workflow by hand to pick them up without a config change. The central cluster consumes it through `kubernetes/apps/base/tenants/shared/bundle.yaml`, alongside the `Shared` folder in `manifests/folder.yaml`.
|
||||
|
||||
```fish
|
||||
mise run //:o11y:render # refresh manifests/dashboards.yaml from upstream
|
||||
mise run //:o11y:check # fail if the committed render is stale
|
||||
mise run //:o11y:render # render manifests/dashboards.yaml locally for review (git-ignored)
|
||||
```
|
||||
|
||||
Both tasks first run `o11y:vendor`, which fetches boards that need a rewrite the sync-job cannot express and writes them to `vendor/` as local sources. Today that is CloudNativePG only: upstream uses `cluster` to mean the Postgres cluster, while the fleet keeps that name under `pg_cluster` and reserves `cluster` for the Kubernetes cluster, so the vendor step renames the label and variable before the sync-job adds the fleet's `$cluster`. A cluster's CNPG series must carry `pg_cluster` for the board to list its databases; the shipping guide's identity-label section shows the relabel rule, which o11y, azad and harbor apply.
|
||||
|
||||
The set is everything at least two clusters run: the dotdc Kubernetes views and system boards, the kube-prometheus mixin boards (kubelet, scheduler, controller manager, proxy, API server, compute resources, networking, persistent volumes, node exporter USE method), Node Exporter Full, etcd, vmagent, Cilium and Hubble, Flux, Spegel, Envoy Gateway and CloudNativePG. Windows, AIX, macOS, Prometheus, Alertmanager and Grafana-overview boards from the mixin bundle are disabled. Documents are sorted by name so re-renders diff cleanly.
|
||||
The set is everything at least two clusters run: the dotdc Kubernetes views and system boards, the kube-prometheus mixin boards (kubelet, scheduler, controller manager, proxy, API server, compute resources, networking, persistent volumes, node exporter USE method), Node Exporter Full, etcd, vmagent, Cilium and Hubble, Flux, Spegel, Envoy Gateway and CloudNativePG. Windows, AIX, macOS, Prometheus, Alertmanager and Grafana-overview boards from the mixin bundle are disabled. Documents are sorted by name so an unchanged upstream renders byte-identical and the published artifact keeps its digest.
|
||||
|
||||
Upstream sources are pinned where the upstream moves (Cilium by release tag, Flux by commit) and tracked at `master` where the VictoriaMetrics chart does the same.
|
||||
|
||||
## Adding a dashboard
|
||||
|
||||
Add its URL under `sources`, and if the upstream board has no `cluster` variable, set `clusterMetric` for it under `dashboards` keyed by the slug of its title (`Kubernetes / Views / Pods` becomes `kubernetes-views-pods`). Run the render task, review the diff, commit. Keep the board out of the VictoriaMetrics chart's own `defaultDashboards.sources` if it is one the chart also imports, so the two do not fight over the same Grafana uid.
|
||||
Add its URL under `sources`, and if the upstream board has no `cluster` variable, set `clusterMetric` for it under `dashboards` keyed by the slug of its title (`Kubernetes / Views / Pods` becomes `kubernetes-views-pods`). Run the render task and review the output locally, then commit the config change; CI renders and publishes. Keep the board out of the VictoriaMetrics chart's own `defaultDashboards.sources` if it is one the chart also imports, so the two do not fight over the same Grafana uid.
|
||||
|
||||
File diff suppressed because one or more lines are too long
Vendored
-9347
File diff suppressed because it is too large
Load Diff
Reference in New Issue
Block a user