feat(cnpg): make better (#496)

This commit is contained in:
Antoine Lecompte
2026-08-19 15:27:09 -04:00
committed by GitHub
parent 569c17ff63
commit 045491d437
34 changed files with 586 additions and 18 deletions
+14
View File
@@ -46,6 +46,20 @@ radosgw-admin user create \
--max-buckets=100
```
## Service accounts
| UID | Purpose | Buckets | Caps |
|---|---|---|---|
| `svc-yucca-restic` | michael's restic object store (one bucket per repository) | 100 | — |
| `metrics-worker` | yucca-metrics-worker usage scraping via the RGW admin API | 0 | `buckets=read;usage=read;metadata=read;users=read` |
| `svc-yucca-db-backup` | CNPG (yucca-database) WAL archiving + base backups via the Barman Cloud plugin | 1 | — |
All three are created by `rgw.yml` with predetermined, TF-minted keys (see
[secrets.md](secrets.md)). `svc-yucca-db-backup` never needs a pre-created
bucket: barman creates `yucca-db-backups` on first use, and `--max-buckets=1`
caps the user there. The k8s side consumes the keys plus the RGW cert from the
TF-provisioned `yucca-db-backup-s3` Secret.
## Self-signed certificate handling
The cluster uses a self-signed wildcard certificate. Every client must either
+6 -1
View File
@@ -72,6 +72,8 @@ credentials without waiting for post-bootstrap capture):
|---|---|---|
| `SIETCH_CEPH_S3_SVC_YUCCA_RESTIC_ACCESS_KEY` | `password` | `vault_s3_restic_access_key` -> `ceph_rgw_s3_user_access_key` |
| `SIETCH_CEPH_S3_SVC_YUCCA_RESTIC_SECRET_KEY` | `password` | `vault_s3_restic_secret_key` -> `ceph_rgw_s3_user_secret_key` |
| `SIETCH_CEPH_S3_SVC_YUCCA_DB_BACKUP_ACCESS_KEY` | `password` | `vault_db_backup_access_key` -> `ceph_rgw_db_backup_user_access_key` (also read by the talos stack into the `yucca-db-backup-s3` Secret for CNPG barman) |
| `SIETCH_CEPH_S3_SVC_YUCCA_DB_BACKUP_SECRET_KEY` | `password` | `vault_db_backup_secret_key` -> `ceph_rgw_db_backup_user_secret_key` (same dual consumption) |
**Disaster-recovery items** (populated by `mise run capture` after
deploy -- stored in 1P for recovery if the bootstrap node's filesystem
@@ -84,7 +86,10 @@ is lost):
| `<CLUSTER>_CEPH_CLIENT_ADMIN_KEYRING` | `password` (concealed) | `/etc/ceph/ceph.client.admin.keyring` on bootstrap |
Items are created on the first `mise run capture`; later runs overwrite them
only if the content changed.
only if the content changed. `<CLUSTER>_CEPH_RGW_TLS_CERT` is no longer
DR-only: the talos stack reads it into the `yucca-db-backup-s3` Secret as the
CA bundle CNPG's barman plugin verifies the RGW endpoint with — run the
capture before the talos apply on a fresh cluster.
Item names are derived in `tf/shared/modules/ceph-cluster/main.tf`
(`local.secret_prefix`). Hardcoded `CEPH` (not `role_in_hostname`) so every
@@ -388,6 +388,15 @@ ceph_rgw_metrics_user_access_key: "{{ vault_metrics_worker_access_key }}"
ceph_rgw_metrics_user_secret_key: "{{ vault_metrics_worker_secret_key }}"
ceph_rgw_metrics_user_caps: "buckets=read;usage=read;metadata=read;users=read"
# CNPG database-backup S3 user: the yucca-database cluster's Barman Cloud
# plugin archives WALs/base backups to its own bucket. Keys TF-minted in 1P
# (SPICE_CEPH_S3_SVC_YUCCA_DB_BACKUP_*); rgw.yml (Step 14.6) creates the user
# with max-buckets=1. Mirror of the sietch definition.
ceph_rgw_db_backup_user_uid: svc-yucca-db-backup
ceph_rgw_db_backup_user_display_name: "yucca/db-backup service account (CNPG barman)"
ceph_rgw_db_backup_user_access_key: "{{ vault_db_backup_access_key }}"
ceph_rgw_db_backup_user_secret_key: "{{ vault_db_backup_secret_key }}"
# === Monitoring Stack ===
ceph_prometheus_port: 9095
ceph_grafana_port: 3000
@@ -112,6 +112,15 @@ ceph_rgw_metrics_user_access_key: "{{ vault_metrics_worker_access_key }}"
ceph_rgw_metrics_user_secret_key: "{{ vault_metrics_worker_secret_key }}"
ceph_rgw_metrics_user_caps: "buckets=read;usage=read;metadata=read;users=read"
# CNPG database-backup S3 user: the yucca-database cluster's Barman Cloud
# plugin archives WALs/base backups to its own bucket. Keys TF-minted in 1P
# (SIETCH_CEPH_S3_SVC_YUCCA_DB_BACKUP_{ACCESS,SECRET}_KEY); rgw.yml (Step 14.6)
# creates the user with max-buckets=1.
ceph_rgw_db_backup_user_uid: svc-yucca-db-backup
ceph_rgw_db_backup_user_display_name: "yucca/db-backup service account (CNPG barman)"
ceph_rgw_db_backup_user_access_key: "{{ vault_db_backup_access_key }}"
ceph_rgw_db_backup_user_secret_key: "{{ vault_db_backup_secret_key }}"
# --- RGW DNS + TLS ---
# Virtual-hosted S3 support: setting rgw_dns_name tells RGW to strip this
# suffix from the Host header and treat the remainder as the bucket name.
@@ -899,6 +899,36 @@
changed_when: true
no_log: true
# --- Step 14.6: CNPG database-backup S3 user ---
#
# A dedicated S3 user the yucca-database CNPG cluster (Barman Cloud plugin)
# archives WALs and base backups with. Keys are TF-minted in 1P
# (<CLUSTER>_CEPH_S3_SVC_YUCCA_DB_BACKUP_{ACCESS,SECRET}_KEY) and passed here so
# the consumer is pre-configured with matching credentials. max-buckets=1: barman
# creates its single bucket on first use; this user can never create another.
- name: Check if db-backup RGW user exists
ansible.builtin.command: "radosgw-admin user info --uid={{ ceph_rgw_db_backup_user_uid }}"
register: db_backup_user_check
when: inventory_hostname in groups['ceph_bootstrap']
changed_when: false
failed_when: false
- name: Create db-backup S3 user with predetermined keys for {{ ceph_rgw_db_backup_user_uid }}
ansible.builtin.command: >
radosgw-admin user create
--uid={{ ceph_rgw_db_backup_user_uid }}
--display-name='{{ ceph_rgw_db_backup_user_display_name }}'
--access-key='{{ ceph_rgw_db_backup_user_access_key }}'
--secret-key='{{ ceph_rgw_db_backup_user_secret_key }}'
--max-buckets=1
register: db_backup_user_create
when:
- inventory_hostname in groups['ceph_bootstrap']
- db_backup_user_check.rc != 0
changed_when: true
no_log: true
- name: Ensure metrics-worker RGW user has read-only admin caps
ansible.builtin.shell: |
set -o pipefail
@@ -4,14 +4,25 @@ metadata:
name: {{ .Values.clusterName }}
spec:
instances: {{ .Values.instances }}
imageName: {{ .Values.imageName }}
{{- with .Values.resources }}
resources: {{- toYaml . | nindent 4 }}
{{- end }}
{{- with .Values.affinity }}
affinity: {{- toYaml . | nindent 4 }}
{{- end }}
storage:
size: {{ .Values.storage }}
{{- with .Values.storageClass }}
storageClass: {{ . }}
{{- end }}
{{- if .Values.backup.enabled }}
plugins:
- name: barman-cloud.cloudnative-pg.io
isWALArchiver: true
parameters:
barmanObjectName: {{ .Values.clusterName }}
{{- end }}
bootstrap:
initdb:
database: {{ .Values.database }}
@@ -0,0 +1,25 @@
{{- if .Values.backup.enabled }}
apiVersion: barmancloud.cnpg.io/v1
kind: ObjectStore
metadata:
name: {{ .Values.clusterName }}
spec:
retentionPolicy: {{ .Values.backup.retentionPolicy }}
configuration:
destinationPath: s3://{{ required "backup.bucket is required" .Values.backup.bucket }}/
endpointURL: {{ required "backup.endpointURL is required" .Values.backup.endpointURL }}
endpointCA:
name: {{ required "backup.secretName is required" .Values.backup.secretName }}
key: CA_CERT
s3Credentials:
accessKeyId:
name: {{ .Values.backup.secretName }}
key: ACCESS_KEY_ID
secretAccessKey:
name: {{ .Values.backup.secretName }}
key: ACCESS_SECRET_KEY
wal:
compression: gzip
data:
compression: gzip
{{- end }}
@@ -0,0 +1,15 @@
{{- if .Values.backup.enabled }}
apiVersion: postgresql.cnpg.io/v1
kind: ScheduledBackup
metadata:
name: {{ .Values.clusterName }}-daily
spec:
schedule: {{ .Values.backup.schedule | quote }}
immediate: true
backupOwnerReference: self
cluster:
name: {{ .Values.clusterName }}
method: plugin
pluginConfiguration:
name: barman-cloud.cloudnative-pg.io
{{- end }}
+19
View File
@@ -7,6 +7,25 @@ storage: 1Gi
storageClass: ""
database: yucca
owner: yucca
# Pinned operand image: the operator's compiled-in default changes (and can
# jump Postgres majors) on operator upgrades, so never leave this implicit.
imageName: ghcr.io/cloudnative-pg/postgresql:17.2
# Per-instance postgres pod resources (CNPG spec.resources). Empty = BestEffort
# (dev default); real clusters set this via the HelmRelease.
resources: {}
# CNPG spec.affinity. Real clusters set podAntiAffinityType: required — the
# default "preferred" can co-schedule two instances, and local PVs
# (WaitForFirstConsumer) then pin that co-location permanently.
affinity: {}
# Barman Cloud plugin backups (WAL archiving + daily base backup) to an
# S3-compatible object store. Requires the plugin-barman-cloud release in
# cnpg-system, so dev (no cert-manager, no plugin) keeps this off.
backup:
enabled: false
bucket: ""
endpointURL: ""
# Secret holding ACCESS_KEY_ID / ACCESS_SECRET_KEY / CA_CERT (the RGW
# endpoints use self-signed TLS and barman cannot skip verification).
secretName: ""
retentionPolicy: 30d
schedule: "0 0 2 * * *"
@@ -10,5 +10,5 @@ spec:
mediaType: application/vnd.cncf.helm.chart.content.v1.tar+gzip
operation: copy
ref:
tag: 0.23.0
tag: 0.29.0
url: oci://ghcr.io/cloudnative-pg/charts/cloudnative-pg
@@ -0,0 +1,20 @@
---
# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/helm.toolkit.fluxcd.io/helmrelease_v2.json
apiVersion: helm.toolkit.fluxcd.io/v2
kind: HelmRelease
metadata:
name: cnpg-plugin-barman
spec:
interval: 1h
chartRef:
kind: OCIRepository
name: cnpg-plugin-barman
install:
crds: CreateReplace
remediation:
retries: 3
upgrade:
crds: CreateReplace
cleanupOnFail: true
remediation:
retries: 3
@@ -0,0 +1,6 @@
---
apiVersion: kustomize.config.k8s.io/v1beta1
kind: Kustomization
resources:
- ./ocirepository.yaml
- ./helmrelease.yaml
@@ -0,0 +1,14 @@
---
# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/source.toolkit.fluxcd.io/ocirepository_v1.json
apiVersion: source.toolkit.fluxcd.io/v1
kind: OCIRepository
metadata:
name: cnpg-plugin-barman
spec:
interval: 15m
layerSelector:
mediaType: application/vnd.cncf.helm.chart.content.v1.tar+gzip
operation: copy
ref:
tag: 0.7.1
url: oci://ghcr.io/cloudnative-pg/charts/plugin-barman-cloud
@@ -23,22 +23,33 @@ spec:
cleanupOnFail: true
remediation:
retries: 3
# HA across the 3 control-planes: CNPG runs a primary + 2 streaming replicas
# and spreads them one-per-node via its default pod anti-affinity. This base
# is consumed only by the real clusters (staging/production) — local dev uses
# the separate dev-mirror tree, which keeps the chart's 1-instance default.
# TODO: point backups at object storage; consider synchronous replication.
values:
instances: 3
storage: 10Gi
imageName: ghcr.io/cloudnative-pg/postgresql:17.2
# One instance per node, enforced: with "preferred" (the CNPG default) a
# cordoned/full node lets two instances co-schedule, and the local PVs
# (WaitForFirstConsumer) then pin that co-location permanently.
affinity:
podAntiAffinityType: required
# Generous requests, huge memory limit (no CPU limit — CFS throttling hurts
# postgres tail latency more than contention does).
resources:
requests: { cpu: 500m, memory: 1Gi }
limits: { memory: 8Gi }
# Node-local NVMe (no replicated block storage). CNPG provides HA at the app
# layer: primary + 2 streaming replicas, spread one-per-node by its default
# anti-affinity, each binding a local PV on its node (WaitForFirstConsumer).
# Per-cluster class: staging = openebs-spare-disk (overlay SC), prod =
# openebs-hostpath (chart localpv class) — set in each cluster-settings.
# layer: primary + 2 streaming replicas, each binding a local PV on its node
# (WaitForFirstConsumer). Per-cluster class: staging = openebs-spare-disk
# (overlay SC), prod = openebs-hostpath (chart localpv class) — set in each
# cluster-settings.
storageClass: ${DB_STORAGE_CLASS}
# PITR to the region's Ceph RGW via the Barman Cloud plugin: continuous WAL
# archiving + a daily base backup, 30d retention (chart default). Barman
# creates the bucket on first use (the RGW user is capped at max-buckets=1);
# the secret is TF-provisioned (talos stack) with the RGW user's keys and
# the self-signed RGW cert as CA_CERT. TODO: second copy on OVH S3.
backup:
enabled: true
bucket: yucca-db-backups
endpointURL: ${S3_ENDPOINT}
secretName: yucca-db-backup-s3
@@ -10,5 +10,5 @@ spec:
mediaType: application/vnd.cncf.helm.chart.content.v1.tar+gzip
operation: copy
ref:
tag: 0.23.0
tag: 0.29.0
url: oci://ghcr.io/cloudnative-pg/charts/cloudnative-pg
@@ -9,6 +9,9 @@ spec:
chart:
spec:
chart: charts/platform/cnpg-cluster
# Repackage on every git revision — the in-repo charts keep a static
# version, so the default ChartVersion strategy never ships template edits.
reconcileStrategy: Revision
sourceRef:
kind: GitRepository
name: yucca
@@ -0,0 +1,31 @@
# Barman Cloud CNPG-I plugin on father — CHERRY-PICKED from
# components/infra/cnpg-system for the same reason as cnpg.yaml (the full infra
# component stays off). WAL archiving + object-store backups for yucca-database;
# gRPC TLS certificates come from cert-manager, hence the extra dependency.
# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/kustomize.toolkit.fluxcd.io/kustomization_v1.json
---
apiVersion: kustomize.toolkit.fluxcd.io/v1
kind: Kustomization
metadata:
name: cnpg-plugin-barman
namespace: flux-system
spec:
dependsOn:
- name: cnpg-operator
- name: cert-manager
healthChecks:
- apiVersion: helm.toolkit.fluxcd.io/v2
kind: HelmRelease
name: cnpg-plugin-barman
namespace: cnpg-system
interval: 1h
retryInterval: 2m
timeout: 10m
path: ./kubernetes/apps/base/cnpg-plugin-barman
prune: true
wait: true
sourceRef:
kind: GitRepository
name: ${MANIFEST_SOURCE:=flux-system}
namespace: flux-system
targetNamespace: cnpg-system
@@ -17,4 +17,5 @@ resources:
- ./envoy.yaml
- ./openebs.yaml
- ./cnpg.yaml
- ./cnpg-plugin-barman.yaml
- ./spegel.yaml
@@ -8,6 +8,7 @@ metadata:
spec:
dependsOn:
- name: cnpg-operator
- name: cnpg-plugin-barman
- name: openebs
healthChecks:
- apiVersion: helm.toolkit.fluxcd.io/v2
@@ -0,0 +1,30 @@
# The Barman Cloud CNPG-I plugin (WAL archiving + object-store backups for the
# yucca-database Cluster). Lives next to the operator in cnpg-system; its
# gRPC TLS certificates come from cert-manager, hence the extra dependency.
# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/kustomize.toolkit.fluxcd.io/kustomization_v1.json
---
apiVersion: kustomize.toolkit.fluxcd.io/v1
kind: Kustomization
metadata:
name: cnpg-plugin-barman
namespace: flux-system
spec:
dependsOn:
- name: cnpg-operator
- name: cert-manager
healthChecks:
- apiVersion: helm.toolkit.fluxcd.io/v2
kind: HelmRelease
name: cnpg-plugin-barman
namespace: cnpg-system
interval: 1h
retryInterval: 2m
timeout: 10m
path: ./kubernetes/apps/base/cnpg-plugin-barman
prune: true
wait: true
sourceRef:
kind: GitRepository
name: ${MANIFEST_SOURCE:=flux-system}
namespace: flux-system
targetNamespace: cnpg-system
@@ -4,3 +4,4 @@ kind: Kustomization
resources:
- ./namespace.yaml
- ./cloudnative-pg.yaml
- ./cnpg-plugin-barman.yaml
+1
View File
@@ -158,6 +158,7 @@ Conventions:
| `michael.yaml` | 5xx ratio, RGW backend pool ejection, storage-op failures, unknown storage cluster, p99 TTFB, outage |
| `yucca-services.yaml` | API 5xx ratio, zero-replica outage of any yucca deployment |
| `backup-health.yaml` | metering pipeline stale, fleet-wide backup staleness (systemic only) |
| `database.yaml` | CNPG (yucca-db) backups: WAL archiving stuck, base backup failed/stale, exporter scrape gone |
| `kubernetes.yaml` | flux reconciliation, cert-manager expiry/readiness, node not ready, crashloops, PVC fill 90% warning / 95% critical (father+luke) |
| `cilium.yaml` | agent daemonset, BGP control-plane sessions (k8s side of the fabric peering) |
| `fabric.yaml` | transit BGP per-carrier (critical; peer IPs pinned from `fabric.tf`), all-transits-down, other BGP sessions, chassis alarms, interface errors, exporter/sFlow liveness |
+212
View File
@@ -0,0 +1,212 @@
# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/grafana.integreatly.org/grafanaalertrulegroup_v1beta1.json
# CNPG control-plane database backups (Barman Cloud plugin → region RGW):
# continuous WAL archiving + a daily 02:00 base backup, so PITR only holds if
# both keep moving. Sources: the cnpg_collector_* gauges every instance manager
# exposes on :9187 (scraped by the yucca-database VMPodScrape); backup
# timestamps mirror the Cluster status the plugin maintains. All three
# instances report the same status values, hence the max by (cluster).
apiVersion: grafana.integreatly.org/v1beta1
kind: GrafanaAlertRuleGroup
metadata:
name: yucca-database
spec:
folderRef: yucca
instanceSelector:
matchLabels:
dashboards: grafana
interval: 1m
rules:
# WAL segments stuck in ready state = archiving to the object store is
# broken and the PITR window stops advancing. A busy checkpoint can queue a
# couple briefly; a sustained queue cannot happen while archiving works.
- uid: cnpg-wal-archiving-stuck
title: CnpgWalArchivingStuck
condition: B
for: 30m
noDataState: OK
execErrState: Error
isPaused: false
labels:
severity: critical
annotations:
summary: database WAL archiving is stuck
description: >-
{{ $values.A.Value }} WAL segments on {{ $labels.cluster }} have been
waiting to archive for 30m — the yucca-db PITR window is not
advancing. Check the barman plugin sidecar logs and the
yucca-db-backup-s3 credentials/RGW reachability.
data:
- refId: A
relativeTimeRange:
from: 600
to: 0
datasourceUid: VictoriaMetrics
model:
refId: A
instant: true
range: false
intervalMs: 300000
maxDataPoints: 43200
expr: >-
max by (cluster)
(last_over_time(cnpg_collector_pg_wal_archive_status{value="ready",
cluster=~"father|luke"}[5m])) > 2
- refId: B
relativeTimeRange:
from: 600
to: 0
datasourceUid: __expr__
model:
refId: B
type: threshold
expression: A
conditions:
- evaluator:
params: [0]
type: gt
# A failure newer than the last success. Positive-while-firing: the value
# is how many seconds the failure postdates the last good backup, and it
# keeps growing until the next success clears it. Also covers a cluster
# whose very first backup failed (last available = 0).
- uid: cnpg-backup-failed
title: CnpgBackupFailed
condition: B
for: 15m
noDataState: OK
execErrState: Error
isPaused: false
labels:
severity: warning
annotations:
summary: database base backup failed
description: >-
The latest yucca-db base backup on {{ $labels.cluster }} failed and
no success has followed. WAL archiving may still be fine; the next
scheduled attempt is 02:00. kubectl get backups -n yucca for the
error.
data:
- refId: A
relativeTimeRange:
from: 600
to: 0
datasourceUid: VictoriaMetrics
model:
refId: A
instant: true
range: false
intervalMs: 300000
maxDataPoints: 43200
expr: >-
max by (cluster)
(last_over_time(cnpg_collector_last_failed_backup_timestamp{cluster=~"father|luke"}[5m]))
- max by (cluster)
(last_over_time(cnpg_collector_last_available_backup_timestamp{cluster=~"father|luke"}[5m]))
> 0
- refId: B
relativeTimeRange:
from: 600
to: 0
datasourceUid: __expr__
model:
refId: B
type: threshold
expression: A
conditions:
- evaluator:
params: [0]
type: gt
# 26h = the daily schedule plus slack for a slow run. The > 0 guard keeps a
# freshly bootstrapped cluster (no backup yet, timestamp 0) from firing;
# cnpg-backup-failed owns that phase.
- uid: cnpg-backup-stale
title: CnpgBackupStale
condition: B
for: 30m
noDataState: OK
execErrState: Error
isPaused: false
labels:
severity: critical
annotations:
summary: no recent database base backup
description: >-
yucca-db on {{ $labels.cluster }} has had no successful base backup
for over 26h (schedule is daily 02:00) — restore points are aging
out against the 30d retention.
data:
- refId: A
relativeTimeRange:
from: 600
to: 0
datasourceUid: VictoriaMetrics
model:
refId: A
instant: true
range: false
intervalMs: 300000
maxDataPoints: 43200
expr: >-
(time() - max by (cluster)
(last_over_time(cnpg_collector_last_available_backup_timestamp{cluster=~"father|luke"}[5m]))
> 93600) and max by (cluster)
(last_over_time(cnpg_collector_last_available_backup_timestamp{cluster=~"father|luke"}[5m]))
> 0
- refId: B
relativeTimeRange:
from: 600
to: 0
datasourceUid: __expr__
model:
refId: B
type: threshold
expression: A
conditions:
- evaluator:
params: [0]
type: gt
# The three rules above are noDataState: OK, so they go blind if the :9187
# scrape breaks (netpol regression, VMPodScrape drift). Guarded absence per
# the house pattern; scoped tighter than telemetry.yaml's per-cluster rule,
# which only catches a cluster-wide remote-write stop.
- uid: cnpg-metrics-stale
title: CnpgMetricsStale
condition: B
for: 20m
noDataState: OK
execErrState: Error
isPaused: false
labels:
severity: warning
annotations:
summary: database metrics stopped
description: >-
cnpg_collector_up disappeared — the yucca-db exporter scrape is
broken, and the database backup alerts are blind until it returns.
data:
- refId: A
relativeTimeRange:
from: 600
to: 0
datasourceUid: VictoriaMetrics
model:
refId: A
instant: true
range: false
intervalMs: 300000
maxDataPoints: 43200
expr: >-
absent(last_over_time(cnpg_collector_up[10m])) and on()
count(count_over_time(cnpg_collector_up[24h])) > 0
- refId: B
relativeTimeRange:
from: 600
to: 0
datasourceUid: __expr__
model:
refId: B
type: threshold
expression: A
conditions:
- evaluator:
params: [0]
type: gt
+7
View File
@@ -54,6 +54,13 @@ export TF_VAR_yucca_rgw_secret_access_key="op://yucca_tf_staging/SIETCH_CEPH_S3_
export TF_VAR_sietch_metrics_worker_access_key="op://yucca_tf_staging/SIETCH_METRICS_WORKER_ACCESS_KEY/password"
export TF_VAR_sietch_metrics_worker_secret_key="op://yucca_tf_staging/SIETCH_METRICS_WORKER_SECRET_KEY/password"
# CNPG database backups → sietch RGW (svc-yucca-db-backup, TF-minted by the
# ceph stack). The cert is the DR item `mise run capture` snapshots from the
# bootstrap node's /etc/ceph/rgw-ssl.crt; barman needs it as a CA bundle.
export TF_VAR_sietch_db_backup_access_key="op://yucca_tf_staging/SIETCH_CEPH_S3_SVC_YUCCA_DB_BACKUP_ACCESS_KEY/password"
export TF_VAR_sietch_db_backup_secret_key="op://yucca_tf_staging/SIETCH_CEPH_S3_SVC_YUCCA_DB_BACKUP_SECRET_KEY/password"
export TF_VAR_sietch_rgw_tls_cert="op://yucca_tf_staging/SIETCH_CEPH_RGW_TLS_CERT/password"
# vmagent + collector → o11y staging vmauth bearer token.
export TF_VAR_vmauth_remote_write_password="op://shared_tf_staging/VICTORIAMETRICS_VMAUTH_PASSWORD/password"
+7
View File
@@ -71,5 +71,12 @@ export TF_VAR_yucca_rgw_secret_access_key="op://yucca_tf_prod/SPICE_CEPH_S3_SVC_
export TF_VAR_spice_metrics_worker_access_key="op://yucca_tf_prod/SPICE_METRICS_WORKER_ACCESS_KEY/password"
export TF_VAR_spice_metrics_worker_secret_key="op://yucca_tf_prod/SPICE_METRICS_WORKER_SECRET_KEY/password"
# CNPG database backups → spice RGW (svc-yucca-db-backup, TF-minted by the
# ceph stack). The cert is the DR item `mise run capture` snapshots from the
# bootstrap node's /etc/ceph/rgw-ssl.crt; barman needs it as a CA bundle.
export TF_VAR_spice_db_backup_access_key="op://yucca_tf_prod/SPICE_CEPH_S3_SVC_YUCCA_DB_BACKUP_ACCESS_KEY/password"
export TF_VAR_spice_db_backup_secret_key="op://yucca_tf_prod/SPICE_CEPH_S3_SVC_YUCCA_DB_BACKUP_SECRET_KEY/password"
export TF_VAR_spice_rgw_tls_cert="op://yucca_tf_prod/SPICE_CEPH_RGW_TLS_CERT/password"
# vmagent/logs remote-write bearer for o11y prod vmauth.
export TF_VAR_vmauth_remote_write_password="op://shared_tf_prod/O11Y_VICTORIAMETRICS_VMAUTH_PASSWORD/password"
+5 -3
View File
@@ -32,13 +32,15 @@ locals {
# Per-role generated-password length. ops is the break-glass account typed by
# hand at the KVM/console, so keep it short; dashboard/grafana are web logins
# (paste-friendly) and stay long. The metrics-worker RGW keys follow the
# AWS/RGW key shape (20-char access id, 40-char secret). Roles not listed use
# the default.
# (paste-friendly) and stay long. The metrics-worker and db-backup RGW keys
# follow the AWS/RGW key shape (20-char access id, 40-char secret). Roles not
# listed use the default.
ceph_password_length = {
ops = 16
metrics_worker_access = 20
metrics_worker_secret = 40
db_backup_access = 20
db_backup_secret = 40
}
ceph_password_default_length = 32
@@ -217,6 +217,29 @@ resource "kubernetes_secret_v1" "yucca_metrics_rgw" {
}
}
# yucca-database backups: the spice RGW svc-yucca-db-backup S3 keys for the
# CNPG Barman Cloud plugin, plus the RGW's self-signed cert as the CA bundle
# (barman cannot skip TLS verification). The cert comes from the DR item that
# `mise run capture` snapshots off the ceph bootstrap node.
resource "kubernetes_secret_v1" "yucca_db_backup_s3" {
metadata {
name = "yucca-db-backup-s3"
namespace = kubernetes_namespace_v1.yucca.metadata[0].name
}
data = {
ACCESS_KEY_ID = var.spice_db_backup_access_key
ACCESS_SECRET_KEY = var.spice_db_backup_secret_key
CA_CERT = var.spice_rgw_tls_cert
}
lifecycle {
precondition {
condition = length(var.spice_db_backup_access_key) > 0 && length(var.spice_db_backup_secret_key) > 0 && length(var.spice_rgw_tls_cert) > 0
error_message = "spice db-backup RGW keys or TLS cert are empty — run applies through tf/op-run.sh with OP_ENV_FILE=tf/.env.prod."
}
}
}
# ─── Observability Secret (namespace: observability) ─────────────────────────
# Bearer token vmagent + the logs collector present to o11y's prod vmauth.
resource "kubernetes_secret_v1" "vmagent_remote_write" {
@@ -214,6 +214,25 @@ variable "spice_metrics_worker_secret_key" {
default = ""
}
variable "spice_db_backup_access_key" {
description = "Spice RGW (S3) access key for CNPG database backups (svc-yucca-db-backup, TF-minted SPICE_CEPH_S3_SVC_YUCCA_DB_BACKUP_ACCESS_KEY)."
type = string
default = ""
}
variable "spice_db_backup_secret_key" {
description = "Spice RGW (S3) secret key for CNPG database backups (svc-yucca-db-backup)."
type = string
sensitive = true
default = ""
}
variable "spice_rgw_tls_cert" {
description = "Spice RGW self-signed TLS certificate PEM (DR item SPICE_CEPH_RGW_TLS_CERT, snapshotted by `mise run capture`); CA bundle for the CNPG barman ObjectStore."
type = string
default = ""
}
variable "vmauth_remote_write_password" {
description = "o11y prod vmauth bearer token for vmagent/logs remote-write (shared_tf_prod/O11Y_VICTORIAMETRICS_VMAUTH_PASSWORD)."
type = string
+5 -3
View File
@@ -28,13 +28,15 @@ locals {
# Per-role generated-password length. ops is the break-glass account typed by
# hand at the KVM/console, so keep it short; dashboard/grafana are web logins
# (paste-friendly) and stay long. The metrics-worker RGW keys follow the
# AWS/RGW key shape (20-char access id, 40-char secret). Roles not listed use
# the default.
# (paste-friendly) and stay long. The metrics-worker and db-backup RGW keys
# follow the AWS/RGW key shape (20-char access id, 40-char secret). Roles not
# listed use the default.
ceph_password_length = {
ops = 16
metrics_worker_access = 20
metrics_worker_secret = 40
db_backup_access = 20
db_backup_secret = 40
}
ceph_password_default_length = 32
@@ -220,6 +220,23 @@ resource "kubernetes_secret_v1" "yucca_metrics_rgw" {
}
}
# yucca-database backups: the sietch RGW svc-yucca-db-backup S3 keys for the
# CNPG Barman Cloud plugin, plus the RGW's self-signed cert as the CA bundle
# (barman cannot skip TLS verification). The cert comes from the DR item that
# `mise run capture` snapshots off the ceph bootstrap node.
resource "kubernetes_secret_v1" "yucca_db_backup_s3" {
count = local.provision_secrets ? 1 : 0
metadata {
name = "yucca-db-backup-s3"
namespace = kubernetes_namespace_v1.yucca[0].metadata[0].name
}
data = {
ACCESS_KEY_ID = var.sietch_db_backup_access_key
ACCESS_SECRET_KEY = var.sietch_db_backup_secret_key
CA_CERT = var.sietch_rgw_tls_cert
}
}
# ─── Observability Secret (namespace: observability) ────────────────────
# Bearer token vmagent + vlagent present to o11y's vmauth for remote-write.
resource "kubernetes_secret_v1" "vmagent_remote_write" {
@@ -116,6 +116,25 @@ variable "sietch_metrics_worker_secret_key" {
default = ""
}
variable "sietch_db_backup_access_key" {
description = "Sietch RGW (S3) access key for CNPG database backups (svc-yucca-db-backup, TF-minted SIETCH_CEPH_S3_SVC_YUCCA_DB_BACKUP_ACCESS_KEY)."
type = string
default = ""
}
variable "sietch_db_backup_secret_key" {
description = "Sietch RGW (S3) secret key for CNPG database backups (svc-yucca-db-backup)."
type = string
sensitive = true
default = ""
}
variable "sietch_rgw_tls_cert" {
description = "Sietch RGW self-signed TLS certificate PEM (DR item SIETCH_CEPH_RGW_TLS_CERT, snapshotted by `mise run capture`); CA bundle for the CNPG barman ObjectStore."
type = string
default = ""
}
# Bearer token vmagent uses to remote-write metrics to o11y's vmauth. This is
# the shared VICTORIAMETRICS_VMAUTH_PASSWORD from the shared_tf_staging vault (the
# `remote-clusters` VMUser authenticates remote clusters with it).
+2
View File
@@ -75,6 +75,8 @@ locals {
grafana = "${local.secret_prefix}_GRAFANA_PASSWORD"
s3_restic_access = "${local.secret_prefix}_S3_SVC_YUCCA_RESTIC_ACCESS_KEY"
s3_restic_secret = "${local.secret_prefix}_S3_SVC_YUCCA_RESTIC_SECRET_KEY"
db_backup_access = "${local.secret_prefix}_S3_SVC_YUCCA_DB_BACKUP_ACCESS_KEY"
db_backup_secret = "${local.secret_prefix}_S3_SVC_YUCCA_DB_BACKUP_SECRET_KEY"
# RGW admin (read-only) keys for the metrics worker. Titled <CLUSTER>_
# METRICS_WORKER_* (no _CEPH infix) to match the metrics-worker consumer's
# 1P contract, which is named by cluster, not by the ceph subsystem.
@@ -14,6 +14,8 @@ vault_ceph_dashboard_password: op://${vault}/${secrets.dashboard}/password
vault_grafana_admin_password: op://${vault}/${secrets.grafana}/password
vault_s3_restic_access_key: op://${vault}/${secrets.s3_restic_access}/password
vault_s3_restic_secret_key: op://${vault}/${secrets.s3_restic_secret}/password
vault_db_backup_access_key: op://${vault}/${secrets.db_backup_access}/password
vault_db_backup_secret_key: op://${vault}/${secrets.db_backup_secret}/password
vault_metrics_worker_access_key: op://${vault}/${secrets.metrics_worker_access}/password
vault_metrics_worker_secret_key: op://${vault}/${secrets.metrics_worker_secret}/password
%{ if contains(keys(secrets), "alertmanager_webhook") ~}