fix(flux): make the database health check actually apply (#614)

This commit is contained in:
Antoine Lecompte
2026-09-01 09:57:06 -04:00
committed by GitHub
parent 2044b5095e
commit ec696ed762
@@ -10,12 +10,11 @@ spec:
- name: cnpg-operator
- name: cnpg-plugin-barman
- name: openebs
# Gate dependents on the database SERVING, not on Helm succeeding. CNPG holds
# the Cluster's Ready condition False while any instance is short (e.g. a lost
# node strands an instance on its node-local PV), which times out the Helm
# release and — through dependsOn — freezes the image rollout of every app
# behind it, even though the primary is up and taking queries. readyInstances
# is the field that tracks "someone can serve"; the Ready condition is not.
# Gate on serving, not on Helm succeeding: CNPG holds Ready False whenever an
# instance is short, which froze every dependent app's images for 18h.
# Threshold 2 — one stranded instance is survivable, a lone primary is not.
# wait MUST stay false: it health-checks the whole inventory and ignores
# healthChecks, which puts the stuck HelmRelease back in the way.
healthChecks:
- apiVersion: postgresql.cnpg.io/v1
kind: Cluster
@@ -24,13 +23,13 @@ spec:
healthCheckExprs:
- apiVersion: postgresql.cnpg.io/v1
kind: Cluster
current: status.readyInstances >= 1
current: status.readyInstances >= 2
interval: 1h
retryInterval: 2m
timeout: 10m
path: ./kubernetes/apps/base/yucca-database
prune: true
wait: true
wait: false
sourceRef:
kind: GitRepository
name: ${MANIFEST_SOURCE:=flux-system}