feat(ceph): move user creation inside terraform (#541)

This commit is contained in:
Antoine Lecompte
2026-08-25 06:32:50 -07:00
committed by GitHub
parent c78ddee105
commit 8392f4f379
20 changed files with 459 additions and 153 deletions
+25
View File
@@ -27,6 +27,31 @@ provider "registry.opentofu.org/1password/onepassword" {
]
}
provider "registry.opentofu.org/fitbeard/radosgw" {
version = "1.6.1"
constraints = "~> 1.6"
hashes = [
"h1:JYsrM9bWyiVMYNLnBB/wmRDBkyv+axvM3tsG8z9D5A8=",
"h1:cN8l4svIr0CdvWGP7OYgPILaChyHBi8GsuWOHUsGpZE=",
"h1:ccJenZcN0DGSSxduQ3BVBybVxI4fCwHHK/vCCT5njaU=",
"h1:igFuURHMzTI7rPIMZjp4QLvulLXAXszxs7njr6oPmZ0=",
"zh:005d66c5061ae2e7ee65715531483bce347227abf39632b04ec1fc04139f5446",
"zh:053445eb5d79f7702973e7e0d0bafba80949f51a8e87ec7187d588dc2c9ad8a6",
"zh:0552f576d6c5ec38f1a721dadb3c10a1e8775c8a6e11354fb32be2a00733bc4a",
"zh:15b8c468dbb4b62c2ef7c488b77b711992ea667e0167fad8b6d85061d1b92f9d",
"zh:2a7af61b6754464fc36e37651852449702e2704c54af3305d2461b353e577f49",
"zh:3248f6c21ef37c0505ed05269c73ceda972382faad8811b196e2670e6884fc7f",
"zh:6c62cd59903a201605c9319b6b1e4e25e7a2ea8eb43ecaf717e4ca17b284a891",
"zh:6fda43406bfb49dc68b30ebc3286ad6ea0c0c052045bcd0dd5447e9df389d22c",
"zh:8f30b2b469932b13e4780f253d903c0423a4f9d81b9a93543c8b1e590e33d476",
"zh:a9ee6ef2483545c9ca3b3f78eb58d37b2b78e225cf88b42d7b5456c4ed48bf95",
"zh:b4f08e0f0a72659d9fbcc3d84e1b0c829dae0da0edaa501e4f7604d0453c8343",
"zh:bac0f40534a1d07e49bf7b4c3edf68fd8f4cc5df254b945f4902fdd04864b070",
"zh:ef470b212eb3dc9bd8489e8ae6f5bfba2fff310336646bb531a20bbf1daedd01",
"zh:f809ab383cca0a5f83072981c64208cbd7fa67e986a86ee02dd2c82333221e32",
]
}
provider "registry.opentofu.org/hashicorp/local" {
version = "2.9.0"
constraints = "~> 2.5"
@@ -28,6 +28,10 @@ clusters = {
ansible_ssh_key = "~/.ssh/id_ed25519_spice"
vault = "yucca_tf_prod"
provision_profile = null
manage_rgw_users = true
# 0 = no bucket limit (RGW semantics): michael mints one bucket per restic
# repository, so prod must not cap them.
rgw_restic_max_buckets = 0
# SPICE_CEPH_ALERTMANAGER_WEBHOOK_URL is provisioned out of band (Zulip
# incoming webhook) and referenced, never generated. See secrets.tf.
alertmanager_webhook = true
@@ -0,0 +1,130 @@
# RGW service users, managed over the live RGW admin API via the radosgw
# provider. Key VALUES keep living in 1P: TF pushes the same predetermined
# keys to RGW that the ansible create-if-missing steps used to, so every
# consumer keeps reading the same items. Mirrors the staging stack's rgw-users.tf.
#
# Per-cluster opt-in (manage_rgw_users in clusters.auto.tfvars) because the
# provider needs two things that exist only after a converge:
# 1. the svc-yucca-terraform admin user on the RGW (rgw.yml Step 14.7),
# authenticated with the <CLUSTER>_CEPH_TF_ADMIN_* keys this stack mints;
# 2. network reach to the RFC1918 S3 endpoint — infra.yml joins NetBird for
# the ceph stack; local plans need the operator's own overlay connection.
#
# Bootstrap sequence for a new cluster, flag OFF:
# apply (mints the TF_ADMIN_* items) → ceph converge (rgw.yml Step 14.7
# creates the admin user; the service users no longer come from ansible) →
# flip manage_rgw_users → apply creates the service users. sietch and spice
# predate this file: their ansible-created users were adopted out-of-band
# with `terragrunt import` (2026-08), so their first flagged plan was a
# no-op.
#
# Decommissioning a managed cluster: destroy (or state-rm) its rgw resources
# BEFORE dropping the entry from clusters -- the provider instance must outlive
# the resources it destroys (tofu warns about the shared for_each at plan).
locals {
rgw_managed_clusters = { for k, c in var.clusters : k => c if c.manage_rgw_users }
rgw_users = {
restic = {
user_id = "svc-yucca-restic"
display_name = "yucca/restic service account"
max_buckets = null
access_role = "s3_restic_access"
secret_role = "s3_restic_secret"
}
metrics_worker = {
user_id = "metrics-worker"
display_name = "yucca/metrics-worker RGW admin (read-only)"
max_buckets = 0
access_role = "metrics_worker_access"
secret_role = "metrics_worker_secret"
}
db_backup = {
user_id = "svc-yucca-db-backup"
display_name = "yucca/db-backup service account (CNPG barman)"
max_buckets = 1
access_role = "db_backup_access"
secret_role = "db_backup_secret"
}
}
rgw_cluster_users = merge([
for cname, c in local.rgw_managed_clusters : {
for uname, u in local.rgw_users :
"${cname}.${uname}" => merge(u, {
cluster = cname
max_buckets = coalesce(u.max_buckets, c.rgw_restic_max_buckets)
})
}
]...)
# Every key this file reads from 1P: the provider's own admin credential plus
# each service user's key pair. Data lookups (not the onepassword_item
# resources in secrets.tf) so out-of-band roles (s3_restic_*) and TF-managed
# roles resolve uniformly.
rgw_key_roles = concat(
["tf_admin_access", "tf_admin_secret"],
flatten([for u in local.rgw_users : [u.access_role, u.secret_role]]),
)
rgw_key_lookups = merge([
for cname, c in local.rgw_managed_clusters : {
for role in local.rgw_key_roles :
"${cname}.${role}" => {
vault = coalesce(c.vault, "Yucca")
title = module.cluster[cname].secrets[role]
}
}
]...)
}
data "onepassword_item" "rgw_key" {
for_each = local.rgw_key_lookups
vault = data.onepassword_vault.target[each.value.vault].uuid
title = each.value.title
}
provider "radosgw" {
alias = "cluster"
for_each = local.rgw_managed_clusters
endpoint = "https://s3.${each.value.domain}"
access_key = data.onepassword_item.rgw_key["${each.key}.tf_admin_access"].password
secret_key = data.onepassword_item.rgw_key["${each.key}.tf_admin_secret"].password
# The RGW frontend serves the self-signed cert from rgw.yml Step 11.6; there
# is no CA to pin (the dashboard's RGW client skips verification the same way).
tls_insecure_skip_verify = true
}
resource "radosgw_iam_user" "svc" {
for_each = local.rgw_cluster_users
provider = radosgw.cluster[each.value.cluster]
user_id = each.value.user_id
display_name = each.value.display_name
max_buckets = each.value.max_buckets
}
resource "radosgw_iam_access_key" "svc" {
for_each = local.rgw_cluster_users
provider = radosgw.cluster[each.value.cluster]
user_id = radosgw_iam_user.svc[each.key].user_id
access_key = data.onepassword_item.rgw_key["${each.value.cluster}.${each.value.access_role}"].password
secret_key = data.onepassword_item.rgw_key["${each.value.cluster}.${each.value.secret_role}"].password
}
# Read-only admin caps for the usage/bucket/user stats scrape.
resource "radosgw_iam_user_caps" "metrics_worker" {
for_each = local.rgw_managed_clusters
provider = radosgw.cluster[each.key]
user_id = radosgw_iam_user.svc["${each.key}.metrics_worker"].user_id
caps = [
{ type = "buckets", perm = "read" },
{ type = "metadata", perm = "read" },
{ type = "usage", perm = "read" },
{ type = "users", perm = "read" },
]
}
+5 -3
View File
@@ -32,15 +32,17 @@ locals {
# Per-role generated-password length. ops is the break-glass account typed by
# hand at the KVM/console, so keep it short; dashboard/grafana are web logins
# (paste-friendly) and stay long. The metrics-worker and db-backup RGW keys
# follow the AWS/RGW key shape (20-char access id, 40-char secret). Roles not
# listed use the default.
# (paste-friendly) and stay long. The metrics-worker, db-backup and tf-admin
# RGW keys follow the AWS/RGW key shape (20-char access id, 40-char secret).
# Roles not listed use the default.
ceph_password_length = {
ops = 16
metrics_worker_access = 20
metrics_worker_secret = 40
db_backup_access = 20
db_backup_secret = 40
tf_admin_access = 20
tf_admin_secret = 40
}
ceph_password_default_length = 32
@@ -10,6 +10,14 @@ variable "clusters" {
ansible_ssh_key = string
vault = optional(string, "Yucca")
provision_profile = optional(string)
# Opt-in: manage the RGW service users via the radosgw provider. Flip only
# after the cluster is converged (svc-yucca-terraform must exist on the
# RGW). See rgw-users.tf for the bootstrap sequence.
manage_rgw_users = optional(bool, false)
# svc-yucca-restic bucket cap (RGW semantics: 0 = no limit, negative
# disables creation). michael creates one bucket per repository, so prod
# runs unlimited; the 100 default matches the historical staging value.
rgw_restic_max_buckets = optional(number, 100)
# Reference an out-of-band <CLUSTER>_CEPH_ALERTMANAGER_WEBHOOK_URL item so
# the cluster's alertmanager gets a real receiver. See the ceph-cluster
# module's alertmanager_webhook variable.
@@ -16,5 +16,11 @@ terraform {
source = "1Password/onepassword"
version = "~> 2.1"
}
# Manages the RGW service users over the admin API (rgw-users.tf). Only
# instantiated for clusters with manage_rgw_users = true.
radosgw = {
source = "fitbeard/radosgw"
version = "~> 1.6"
}
}
}
+25
View File
@@ -27,6 +27,31 @@ provider "registry.opentofu.org/1password/onepassword" {
]
}
provider "registry.opentofu.org/fitbeard/radosgw" {
version = "1.6.1"
constraints = "~> 1.6"
hashes = [
"h1:JYsrM9bWyiVMYNLnBB/wmRDBkyv+axvM3tsG8z9D5A8=",
"h1:cN8l4svIr0CdvWGP7OYgPILaChyHBi8GsuWOHUsGpZE=",
"h1:ccJenZcN0DGSSxduQ3BVBybVxI4fCwHHK/vCCT5njaU=",
"h1:igFuURHMzTI7rPIMZjp4QLvulLXAXszxs7njr6oPmZ0=",
"zh:005d66c5061ae2e7ee65715531483bce347227abf39632b04ec1fc04139f5446",
"zh:053445eb5d79f7702973e7e0d0bafba80949f51a8e87ec7187d588dc2c9ad8a6",
"zh:0552f576d6c5ec38f1a721dadb3c10a1e8775c8a6e11354fb32be2a00733bc4a",
"zh:15b8c468dbb4b62c2ef7c488b77b711992ea667e0167fad8b6d85061d1b92f9d",
"zh:2a7af61b6754464fc36e37651852449702e2704c54af3305d2461b353e577f49",
"zh:3248f6c21ef37c0505ed05269c73ceda972382faad8811b196e2670e6884fc7f",
"zh:6c62cd59903a201605c9319b6b1e4e25e7a2ea8eb43ecaf717e4ca17b284a891",
"zh:6fda43406bfb49dc68b30ebc3286ad6ea0c0c052045bcd0dd5447e9df389d22c",
"zh:8f30b2b469932b13e4780f253d903c0423a4f9d81b9a93543c8b1e590e33d476",
"zh:a9ee6ef2483545c9ca3b3f78eb58d37b2b78e225cf88b42d7b5456c4ed48bf95",
"zh:b4f08e0f0a72659d9fbcc3d84e1b0c829dae0da0edaa501e4f7604d0453c8343",
"zh:bac0f40534a1d07e49bf7b4c3edf68fd8f4cc5df254b945f4902fdd04864b070",
"zh:ef470b212eb3dc9bd8489e8ae6f5bfba2fff310336646bb531a20bbf1daedd01",
"zh:f809ab383cca0a5f83072981c64208cbd7fa67e986a86ee02dd2c82333221e32",
]
}
provider "registry.opentofu.org/hashicorp/local" {
version = "2.9.0"
constraints = "~> 2.5"
@@ -23,6 +23,7 @@ clusters = {
ansible_ssh_key = "~/.ssh/id_ed25519_sietch"
vault = "yucca_tf_staging"
provision_profile = "debian-live"
manage_rgw_users = true
hosts = [
{ name = "laurel", bond_ip = "10.10.10.90", bootstrap = true },
{ name = "lawson", bond_ip = "10.10.10.91" },
@@ -0,0 +1,130 @@
# RGW service users, managed over the live RGW admin API via the radosgw
# provider. Key VALUES keep living in 1P: TF pushes the same predetermined
# keys to RGW that the ansible create-if-missing steps used to, so every
# consumer keeps reading the same items. Mirrors the prod stack's rgw-users.tf.
#
# Per-cluster opt-in (manage_rgw_users in clusters.auto.tfvars) because the
# provider needs two things that exist only after a converge:
# 1. the svc-yucca-terraform admin user on the RGW (rgw.yml Step 14.7),
# authenticated with the <CLUSTER>_CEPH_TF_ADMIN_* keys this stack mints;
# 2. network reach to the RFC1918 S3 endpoint — infra.yml joins NetBird for
# the ceph stack; local plans need the operator's own overlay connection.
#
# Bootstrap sequence for a new cluster, flag OFF:
# apply (mints the TF_ADMIN_* items) → ceph converge (rgw.yml Step 14.7
# creates the admin user; the service users no longer come from ansible) →
# flip manage_rgw_users → apply creates the service users. sietch and spice
# predate this file: their ansible-created users were adopted out-of-band
# with `terragrunt import` (2026-08), so their first flagged plan was a
# no-op.
#
# Decommissioning a managed cluster: destroy (or state-rm) its rgw resources
# BEFORE dropping the entry from clusters -- the provider instance must outlive
# the resources it destroys (tofu warns about the shared for_each at plan).
locals {
rgw_managed_clusters = { for k, c in var.clusters : k => c if c.manage_rgw_users }
rgw_users = {
restic = {
user_id = "svc-yucca-restic"
display_name = "yucca/restic service account"
max_buckets = null
access_role = "s3_restic_access"
secret_role = "s3_restic_secret"
}
metrics_worker = {
user_id = "metrics-worker"
display_name = "yucca/metrics-worker RGW admin (read-only)"
max_buckets = 0
access_role = "metrics_worker_access"
secret_role = "metrics_worker_secret"
}
db_backup = {
user_id = "svc-yucca-db-backup"
display_name = "yucca/db-backup service account (CNPG barman)"
max_buckets = 1
access_role = "db_backup_access"
secret_role = "db_backup_secret"
}
}
rgw_cluster_users = merge([
for cname, c in local.rgw_managed_clusters : {
for uname, u in local.rgw_users :
"${cname}.${uname}" => merge(u, {
cluster = cname
max_buckets = coalesce(u.max_buckets, c.rgw_restic_max_buckets)
})
}
]...)
# Every key this file reads from 1P: the provider's own admin credential plus
# each service user's key pair. Data lookups (not the onepassword_item
# resources in secrets.tf) so out-of-band roles (s3_restic_*) and TF-managed
# roles resolve uniformly.
rgw_key_roles = concat(
["tf_admin_access", "tf_admin_secret"],
flatten([for u in local.rgw_users : [u.access_role, u.secret_role]]),
)
rgw_key_lookups = merge([
for cname, c in local.rgw_managed_clusters : {
for role in local.rgw_key_roles :
"${cname}.${role}" => {
vault = coalesce(c.vault, "Yucca")
title = module.cluster[cname].secrets[role]
}
}
]...)
}
data "onepassword_item" "rgw_key" {
for_each = local.rgw_key_lookups
vault = data.onepassword_vault.target[each.value.vault].uuid
title = each.value.title
}
provider "radosgw" {
alias = "cluster"
for_each = local.rgw_managed_clusters
endpoint = "https://s3.${each.value.domain}"
access_key = data.onepassword_item.rgw_key["${each.key}.tf_admin_access"].password
secret_key = data.onepassword_item.rgw_key["${each.key}.tf_admin_secret"].password
# The RGW frontend serves the self-signed cert from rgw.yml Step 11.6; there
# is no CA to pin (the dashboard's RGW client skips verification the same way).
tls_insecure_skip_verify = true
}
resource "radosgw_iam_user" "svc" {
for_each = local.rgw_cluster_users
provider = radosgw.cluster[each.value.cluster]
user_id = each.value.user_id
display_name = each.value.display_name
max_buckets = each.value.max_buckets
}
resource "radosgw_iam_access_key" "svc" {
for_each = local.rgw_cluster_users
provider = radosgw.cluster[each.value.cluster]
user_id = radosgw_iam_user.svc[each.key].user_id
access_key = data.onepassword_item.rgw_key["${each.value.cluster}.${each.value.access_role}"].password
secret_key = data.onepassword_item.rgw_key["${each.value.cluster}.${each.value.secret_role}"].password
}
# Read-only admin caps for the usage/bucket/user stats scrape.
resource "radosgw_iam_user_caps" "metrics_worker" {
for_each = local.rgw_managed_clusters
provider = radosgw.cluster[each.key]
user_id = radosgw_iam_user.svc["${each.key}.metrics_worker"].user_id
caps = [
{ type = "buckets", perm = "read" },
{ type = "metadata", perm = "read" },
{ type = "usage", perm = "read" },
{ type = "users", perm = "read" },
]
}
+5 -3
View File
@@ -28,15 +28,17 @@ locals {
# Per-role generated-password length. ops is the break-glass account typed by
# hand at the KVM/console, so keep it short; dashboard/grafana are web logins
# (paste-friendly) and stay long. The metrics-worker and db-backup RGW keys
# follow the AWS/RGW key shape (20-char access id, 40-char secret). Roles not
# listed use the default.
# (paste-friendly) and stay long. The metrics-worker, db-backup and tf-admin
# RGW keys follow the AWS/RGW key shape (20-char access id, 40-char secret).
# Roles not listed use the default.
ceph_password_length = {
ops = 16
metrics_worker_access = 20
metrics_worker_secret = 40
db_backup_access = 20
db_backup_secret = 40
tf_admin_access = 20
tf_admin_secret = 40
}
ceph_password_default_length = 32
@@ -10,6 +10,14 @@ variable "clusters" {
ansible_ssh_key = string
vault = optional(string, "Yucca")
provision_profile = optional(string)
# Opt-in: manage the RGW service users via the radosgw provider. Flip only
# after the cluster is converged (svc-yucca-terraform must exist on the
# RGW). See rgw-users.tf for the bootstrap sequence.
manage_rgw_users = optional(bool, false)
# svc-yucca-restic bucket cap (RGW semantics: 0 = no limit, negative
# disables creation). michael creates one bucket per repository, so prod
# runs unlimited; the 100 default matches the historical staging value.
rgw_restic_max_buckets = optional(number, 100)
# -> group_vars/all/ceph-config.generated.yml (`ceph_config_cluster`).
# Empty for sietch today; the schema mirrors prod/htz-fsn1.
ceph_config = optional(map(map(string)), {})
@@ -17,5 +17,11 @@ terraform {
source = "1Password/onepassword"
version = "~> 2.1"
}
# Manages the RGW service users over the admin API (rgw-users.tf). Only
# instantiated for clusters with manage_rgw_users = true.
radosgw = {
source = "fitbeard/radosgw"
version = "~> 1.6"
}
}
}
+5
View File
@@ -82,6 +82,11 @@ locals {
# 1P contract, which is named by cluster, not by the ceph subsystem.
metrics_worker_access = "${upper(var.cluster_name)}_METRICS_WORKER_ACCESS_KEY"
metrics_worker_secret = "${upper(var.cluster_name)}_METRICS_WORKER_SECRET_KEY"
# RGW admin (write) keys for the radosgw terraform provider (the ceph
# stacks' rgw-users.tf). The svc-yucca-terraform user itself is bootstrapped
# by ansible (rgw.yml) with these keys -- TF cannot create its own admin.
tf_admin_access = "${local.secret_prefix}_TF_ADMIN_ACCESS_KEY"
tf_admin_secret = "${local.secret_prefix}_TF_ADMIN_SECRET_KEY"
},
# Alertmanager receiver URL. Opt-in per cluster, and provisioned OUT OF BAND:
# the value is an externally-issued webhook (Zulip/Opsgenie/etc), not a
@@ -14,10 +14,8 @@ vault_ceph_dashboard_password: op://${vault}/${secrets.dashboard}/password
vault_grafana_admin_password: op://${vault}/${secrets.grafana}/password
vault_s3_restic_access_key: op://${vault}/${secrets.s3_restic_access}/password
vault_s3_restic_secret_key: op://${vault}/${secrets.s3_restic_secret}/password
vault_db_backup_access_key: op://${vault}/${secrets.db_backup_access}/password
vault_db_backup_secret_key: op://${vault}/${secrets.db_backup_secret}/password
vault_metrics_worker_access_key: op://${vault}/${secrets.metrics_worker_access}/password
vault_metrics_worker_secret_key: op://${vault}/${secrets.metrics_worker_secret}/password
vault_tf_admin_access_key: op://${vault}/${secrets.tf_admin_access}/password
vault_tf_admin_secret_key: op://${vault}/${secrets.tf_admin_secret}/password
%{ if contains(keys(secrets), "alertmanager_webhook") ~}
vault_alertmanager_webhook_url: op://${vault}/${secrets.alertmanager_webhook}/password
%{ endif ~}