From 76e4e69f0de95e65a860295e288bfdba66ab1db9 Mon Sep 17 00:00:00 2001 From: Devin Buhl Date: Wed, 2 Sep 2026 09:08:17 -0400 Subject: [PATCH] feat(deployment): guard servers from destroy and stage production talos reboots (#269) * feat(deployment): guard servers from destroy and stage production talos reboots Add prevent_destroy to the OVH control-plane instances, the bare-metal workers, and the Talos machine secrets so a plan that would replace or delete one fails instead of applying. Add a talos apply_mode variable. Production applies with staged_if_needing_reboot: a machine-config change that needs a reboot is written to the node but stays inactive until an operator reboots it, one node at a time. Staging keeps auto so the reboot path is exercised there first. The provider has no try-style auto-revert mode, and the per-node applies are unordered, so this is the only lever that stops a bad config from rebooting every control plane at once. Document both behaviours in the bootstrap guide. Signed-off-by: Devin Buhl * chore(rootly): refresh provider lock hashes Signed-off-by: Devin Buhl --------- Signed-off-by: Devin Buhl --- deployment/modules/ovh/account/controlplane.tf | 3 ++- deployment/modules/ovh/account/workers.tf | 1 + deployment/modules/rootly/cluster/.terraform.lock.hcl | 1 + deployment/modules/talos/cluster/controlplane.tf | 1 + deployment/modules/talos/cluster/main.tf | 6 +++++- deployment/modules/talos/cluster/terragrunt.hcl | 10 ++++++---- deployment/modules/talos/cluster/variables.tf | 11 +++++++++++ deployment/modules/talos/cluster/workers.tf | 1 + docs/01-bootstrap-guide.md | 5 +++++ 9 files changed, 33 insertions(+), 6 deletions(-) diff --git a/deployment/modules/ovh/account/controlplane.tf b/deployment/modules/ovh/account/controlplane.tf index b330925..c59378e 100644 --- a/deployment/modules/ovh/account/controlplane.tf +++ b/deployment/modules/ovh/account/controlplane.tf @@ -38,6 +38,7 @@ resource "ovh_cloud_project_instance" "controlplane" { lifecycle { # boot_from is consumed at create only; Talos upgrades go via machine.install.image. - ignore_changes = [boot_from] + ignore_changes = [boot_from] + prevent_destroy = true } } diff --git a/deployment/modules/ovh/account/workers.tf b/deployment/modules/ovh/account/workers.tf index ec220c8..ae6847c 100644 --- a/deployment/modules/ovh/account/workers.tf +++ b/deployment/modules/ovh/account/workers.tf @@ -79,5 +79,6 @@ resource "ovh_dedicated_server" "worker" { efi_bootloader_path, customizations, ] + prevent_destroy = true } } diff --git a/deployment/modules/rootly/cluster/.terraform.lock.hcl b/deployment/modules/rootly/cluster/.terraform.lock.hcl index 33ec9bf..cec9c32 100644 --- a/deployment/modules/rootly/cluster/.terraform.lock.hcl +++ b/deployment/modules/rootly/cluster/.terraform.lock.hcl @@ -51,5 +51,6 @@ provider "registry.opentofu.org/rootlyhq/rootly" { "zh:cecfff40cc8bf4a6fcad9ce57b9d64e28c85071b29d8fb4affbb25048924caf7", "zh:d358d1bb4a91503e4bc2e0b1e044253aecf93203e3c1454d03b7cfdeb040d52a", "zh:d7f076da3d4fcad6fafecc287b2b2d1755ee25eaadc9b9e100a6560b16b44d3a", + "zh:f569b65999264a9416862bca5cd2a6177d94ccb0424f3a4ef424428912b9cb3c", ] } diff --git a/deployment/modules/talos/cluster/controlplane.tf b/deployment/modules/talos/cluster/controlplane.tf index f326e6a..6d1cbec 100644 --- a/deployment/modules/talos/cluster/controlplane.tf +++ b/deployment/modules/talos/cluster/controlplane.tf @@ -12,6 +12,7 @@ resource "talos_machine_configuration_apply" "controlplane" { client_configuration = talos_machine_secrets.this.client_configuration machine_configuration_input = data.talos_machine_configuration.controlplane.machine_configuration node = local.controlplane_endpoint_ips[each.key] + apply_mode = var.apply_mode on_destroy = { reboot = true diff --git a/deployment/modules/talos/cluster/main.tf b/deployment/modules/talos/cluster/main.tf index 36f8705..2821dd9 100644 --- a/deployment/modules/talos/cluster/main.tf +++ b/deployment/modules/talos/cluster/main.tf @@ -25,7 +25,11 @@ locals { operator_endpoint = "https://${local.bootstrap_node.private_ip}:6443" } -resource "talos_machine_secrets" "this" {} +resource "talos_machine_secrets" "this" { + lifecycle { + prevent_destroy = true + } +} data "talos_client_configuration" "this" { cluster_name = local.cluster_name diff --git a/deployment/modules/talos/cluster/terragrunt.hcl b/deployment/modules/talos/cluster/terragrunt.hcl index 2c14df7..5237f8e 100644 --- a/deployment/modules/talos/cluster/terragrunt.hcl +++ b/deployment/modules/talos/cluster/terragrunt.hcl @@ -73,16 +73,18 @@ dependency "netbird_cluster" { } inputs = { - kubernetes_version = "1.36.2" - controlplane_nodes = dependency.ovh.outputs.controlplane_nodes - worker_nodes = dependency.ovh.outputs.worker_nodes - private_network_cidr = dependency.ovh.outputs.private_network_cidr + kubernetes_version = "1.36.2" + controlplane_nodes = dependency.ovh.outputs.controlplane_nodes + worker_nodes = dependency.ovh.outputs.worker_nodes + private_network_cidr = dependency.ovh.outputs.private_network_cidr talos_installer_images = dependency.ovh.outputs.talos_installer_images worker_data_disk_match = local.worker_data_disk_match worker_data_disk2_match = local.worker_data_disk2_match worker_nics = local.worker_nics netbird_setup_key = dependency.netbird_cluster.outputs.talos_setup_key mesh_dns_zone = dependency.netbird_cluster.outputs.mesh_dns_zone + # CI applies on merge; production never reboots unattended (see variables.tf). + apply_mode = local.env == "production" ? "staged_if_needing_reboot" : "auto" } generate "backend" { diff --git a/deployment/modules/talos/cluster/variables.tf b/deployment/modules/talos/cluster/variables.tf index 43fa01c..9cb5b21 100644 --- a/deployment/modules/talos/cluster/variables.tf +++ b/deployment/modules/talos/cluster/variables.tf @@ -92,6 +92,17 @@ variable "worker_nics" { })) } +# How machine-config changes reach the nodes. "auto" lets the provider reboot a +# node whenever the change needs it: every node at once, since the applies are +# not ordered. "staged_if_needing_reboot" dry-runs first and stages any +# reboot-requiring change instead, so the node keeps its running config until an +# operator reboots it; the dry-run needs the node reachable at plan time and +# falls back to "auto" when it isn't. +variable "apply_mode" { + type = string + default = "auto" +} + # True only during initial bring-up of a brand-new env, before the Netbird # extension has registered any node. Drop back to false once each node is on # the netbird mesh, so future applies go via the vRack and the ingress firewall diff --git a/deployment/modules/talos/cluster/workers.tf b/deployment/modules/talos/cluster/workers.tf index 85079d8..c8341be 100644 --- a/deployment/modules/talos/cluster/workers.tf +++ b/deployment/modules/talos/cluster/workers.tf @@ -12,6 +12,7 @@ resource "talos_machine_configuration_apply" "worker" { client_configuration = talos_machine_secrets.this.client_configuration machine_configuration_input = data.talos_machine_configuration.worker.machine_configuration node = local.worker_endpoint_ips[each.key] + apply_mode = var.apply_mode on_destroy = { reboot = true diff --git a/docs/01-bootstrap-guide.md b/docs/01-bootstrap-guide.md index 6e267e1..e147150 100644 --- a/docs/01-bootstrap-guide.md +++ b/docs/01-bootstrap-guide.md @@ -117,6 +117,11 @@ talosctl -n 10.150.200.10 health # any CP private IP; the talosconfig also lis | Lint docs | `mise run //:md:lint` | | Fetch kubeconfig / talosconfig | `mise run //deployment/modules/talos/cluster:kubeconfig` / `mise run //deployment/modules/talos/cluster:talosconfig` (see [Cluster access](#cluster-access)) | +## Apply guardrails + +* **Production never reboots unattended.** The production Talos module applies with `staged_if_needing_reboot`: a machine-config change that needs a reboot is written to the node but stays inactive until the node reboots (the plan shows `resolved_apply_mode = "staged"` for those nodes). Roll them yourself, one at a time, checking `talosctl health` between nodes and doing the elected NetBird routing peer last (rebooting it drops the vRack route until another peer is elected). Staging applies with `auto`, so it reboots on its own and proves the change first. +* **Servers cannot be destroyed by an apply.** The OVH control-plane instances, the bare-metal workers, and the Talos machine secrets carry `prevent_destroy`; a plan that would replace or delete one fails instead. Tearing one down deliberately means removing that lifecycle rule in the same change. + ## Tooling * **OpenTofu** + **Terragrunt** for IaC; **mise** drives tool versions and task wrappers. Version pins live in `.mise/config.toml` and each module's lock file.