feat(o11y): self-select vmauth proxy and Fleet datasource (#195)

Signed-off-by: Devin Buhl <devin@buhl.casa>
This commit is contained in:
Devin Buhl
2026-08-06 11:50:07 -04:00
committed by GitHub
parent d168175360
commit f71208bb6a
6 changed files with 125 additions and 55 deletions
@@ -0,0 +1,71 @@
---
# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/grafana.integreatly.org/grafanaalertrulegroup_v1beta1.json
# Fleet-wide rules: everything here queries the unscoped VictoriaMetrics
# Fleet datasource on purpose. Rules about this cluster's own health belong
# in alerts-o11y.yaml, which rides the (scoped) default datasource.
apiVersion: grafana.integreatly.org/v1beta1
kind: GrafanaAlertRuleGroup
metadata:
name: o11y-fleet
spec:
instanceSelector:
matchLabels:
dashboards: grafana
folderRef: o11y
interval: 1m
rules:
- uid: o11y-cluster-silent
title: o11y cluster stopped reporting
condition: B
for: 15m
labels:
project: o11y
severity: critical
annotations:
summary: Cluster {{ $labels.cluster }} reported metrics within the last day but has sent nothing recently; its remote write path may be down.
execErrState: Error
noDataState: OK
data:
- refId: A
datasourceUid: VictoriaMetricsFleet
queryType: ""
relativeTimeRange:
from: 600
to: 0
model:
refId: A
datasource:
type: prometheus
uid: VictoriaMetricsFleet
expr: count by (cluster) (up offset 1d) unless count by (cluster) (up)
instant: true
range: false
intervalMs: 1000
maxDataPoints: 43200
- refId: B
datasourceUid: __expr__
queryType: ""
relativeTimeRange:
from: 600
to: 0
model:
refId: B
type: threshold
datasource:
type: __expr__
uid: __expr__
expression: A
intervalMs: 1000
maxDataPoints: 43200
conditions:
- type: query
evaluator:
type: gt
params: [0]
operator:
type: and
query:
params: [A]
reducer:
type: last
params: []
@@ -232,61 +232,6 @@ spec:
reducer:
type: last
params: []
- uid: o11y-cluster-silent
title: o11y cluster stopped reporting
condition: B
for: 15m
labels:
project: o11y
severity: critical
annotations:
summary: Cluster {{ $labels.cluster }} reported metrics within the last day but has sent nothing recently; its remote write path may be down.
execErrState: Error
noDataState: OK
data:
- refId: A
datasourceUid: VictoriaMetrics
queryType: ""
relativeTimeRange:
from: 600
to: 0
model:
refId: A
datasource:
type: prometheus
uid: VictoriaMetrics
expr: count by (cluster) (up offset 1d) unless count by (cluster) (up)
instant: true
range: false
intervalMs: 1000
maxDataPoints: 43200
- refId: B
datasourceUid: __expr__
queryType: ""
relativeTimeRange:
from: 600
to: 0
model:
refId: B
type: threshold
datasource:
type: __expr__
uid: __expr__
expression: A
intervalMs: 1000
maxDataPoints: 43200
conditions:
- type: query
evaluator:
type: gt
params: [0]
operator:
type: and
query:
params: [A]
reducer:
type: last
params: []
- uid: o11y-node-not-ready
title: o11y node not ready
condition: B
@@ -0,0 +1,19 @@
---
# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/grafana.integreatly.org/grafanadatasource_v1beta1.json
apiVersion: grafana.integreatly.org/v1beta1
kind: GrafanaDatasource
metadata:
name: victoria-metrics-fleet
spec:
instanceSelector:
matchLabels:
dashboards: grafana
datasource:
# Unscoped view of the whole multi-tenant store: explicit opt-in for
# cross-cluster dashboards (the yucca folder) and fleet-wide alerts.
name: VictoriaMetrics Fleet
uid: VictoriaMetricsFleet
type: prometheus
access: proxy
isDefault: false
url: http://vmselect-victoria-metrics.o11y.svc.cluster.local:8481/select/0/prometheus
@@ -5,9 +5,11 @@ resources:
- ./externalsecret.yaml
- ./grafana.yaml
- ./datasource.yaml
- ./datasource-fleet.yaml
- ./contactpoint-discord.yaml
- ./contactpoint-rootly.yaml
- ./notificationpolicy.yaml
- ./servicemonitor.yaml
- ./alerts-o11y.yaml
- ./alerts-fleet.yaml
- ./dashboards
@@ -4,3 +4,4 @@ kind: Kustomization
resources:
- ./vmuser-remote-clusters.yaml
- ./vmauth-mesh-unauth.yaml
- ./vmauth-self-select.yaml
@@ -0,0 +1,32 @@
---
# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/operator.victoriametrics.com/vmauth_v1beta1.json
apiVersion: operator.victoriametrics.com/v1beta1
kind: VMAuth
metadata:
name: self-select
spec:
# In-cluster only: backs Grafana's default datasource. The appended
# extra_label is enforced by vmselect on every request (queries and
# label/series metadata alike), pinning anything that queries through
# here to this cluster's own series. Cross-cluster reads go through the
# explicit VictoriaMetrics Fleet datasource instead.
port: "8427"
replicaCount: 2
podDisruptionBudget:
minAvailable: 1
topologySpreadConstraints:
- maxSkew: 1
topologyKey: kubernetes.io/hostname
whenUnsatisfiable: ScheduleAnyway
labelSelector:
matchLabels:
app.kubernetes.io/name: vmauth
app.kubernetes.io/instance: self-select
selectAllByDefault: false
unauthorizedUserAccessSpec:
targetRefs:
- static:
url: http://vmselect-victoria-metrics.o11y.svc.cluster.local:8481
paths:
- /select/0/.*
target_path_suffix: "?extra_label=cluster=${CLUSTER_NAME}"