mirror of
https://github.com/immich-app/yucca-o11y.git
synced 2026-09-30 13:23:23 +08:00
feat(deployment): guard servers from destroy and stage production talos reboots (#269)
* feat(deployment): guard servers from destroy and stage production talos reboots Add prevent_destroy to the OVH control-plane instances, the bare-metal workers, and the Talos machine secrets so a plan that would replace or delete one fails instead of applying. Add a talos apply_mode variable. Production applies with staged_if_needing_reboot: a machine-config change that needs a reboot is written to the node but stays inactive until an operator reboots it, one node at a time. Staging keeps auto so the reboot path is exercised there first. The provider has no try-style auto-revert mode, and the per-node applies are unordered, so this is the only lever that stops a bad config from rebooting every control plane at once. Document both behaviours in the bootstrap guide. Signed-off-by: Devin Buhl <devin@buhl.casa> * chore(rootly): refresh provider lock hashes Signed-off-by: Devin Buhl <devin@buhl.casa> --------- Signed-off-by: Devin Buhl <devin@buhl.casa>
This commit is contained in:
@@ -38,6 +38,7 @@ resource "ovh_cloud_project_instance" "controlplane" {
|
||||
|
||||
lifecycle {
|
||||
# boot_from is consumed at create only; Talos upgrades go via machine.install.image.
|
||||
ignore_changes = [boot_from]
|
||||
ignore_changes = [boot_from]
|
||||
prevent_destroy = true
|
||||
}
|
||||
}
|
||||
|
||||
@@ -79,5 +79,6 @@ resource "ovh_dedicated_server" "worker" {
|
||||
efi_bootloader_path,
|
||||
customizations,
|
||||
]
|
||||
prevent_destroy = true
|
||||
}
|
||||
}
|
||||
|
||||
@@ -51,5 +51,6 @@ provider "registry.opentofu.org/rootlyhq/rootly" {
|
||||
"zh:cecfff40cc8bf4a6fcad9ce57b9d64e28c85071b29d8fb4affbb25048924caf7",
|
||||
"zh:d358d1bb4a91503e4bc2e0b1e044253aecf93203e3c1454d03b7cfdeb040d52a",
|
||||
"zh:d7f076da3d4fcad6fafecc287b2b2d1755ee25eaadc9b9e100a6560b16b44d3a",
|
||||
"zh:f569b65999264a9416862bca5cd2a6177d94ccb0424f3a4ef424428912b9cb3c",
|
||||
]
|
||||
}
|
||||
|
||||
@@ -12,6 +12,7 @@ resource "talos_machine_configuration_apply" "controlplane" {
|
||||
client_configuration = talos_machine_secrets.this.client_configuration
|
||||
machine_configuration_input = data.talos_machine_configuration.controlplane.machine_configuration
|
||||
node = local.controlplane_endpoint_ips[each.key]
|
||||
apply_mode = var.apply_mode
|
||||
|
||||
on_destroy = {
|
||||
reboot = true
|
||||
|
||||
@@ -25,7 +25,11 @@ locals {
|
||||
operator_endpoint = "https://${local.bootstrap_node.private_ip}:6443"
|
||||
}
|
||||
|
||||
resource "talos_machine_secrets" "this" {}
|
||||
resource "talos_machine_secrets" "this" {
|
||||
lifecycle {
|
||||
prevent_destroy = true
|
||||
}
|
||||
}
|
||||
|
||||
data "talos_client_configuration" "this" {
|
||||
cluster_name = local.cluster_name
|
||||
|
||||
@@ -73,16 +73,18 @@ dependency "netbird_cluster" {
|
||||
}
|
||||
|
||||
inputs = {
|
||||
kubernetes_version = "1.36.2"
|
||||
controlplane_nodes = dependency.ovh.outputs.controlplane_nodes
|
||||
worker_nodes = dependency.ovh.outputs.worker_nodes
|
||||
private_network_cidr = dependency.ovh.outputs.private_network_cidr
|
||||
kubernetes_version = "1.36.2"
|
||||
controlplane_nodes = dependency.ovh.outputs.controlplane_nodes
|
||||
worker_nodes = dependency.ovh.outputs.worker_nodes
|
||||
private_network_cidr = dependency.ovh.outputs.private_network_cidr
|
||||
talos_installer_images = dependency.ovh.outputs.talos_installer_images
|
||||
worker_data_disk_match = local.worker_data_disk_match
|
||||
worker_data_disk2_match = local.worker_data_disk2_match
|
||||
worker_nics = local.worker_nics
|
||||
netbird_setup_key = dependency.netbird_cluster.outputs.talos_setup_key
|
||||
mesh_dns_zone = dependency.netbird_cluster.outputs.mesh_dns_zone
|
||||
# CI applies on merge; production never reboots unattended (see variables.tf).
|
||||
apply_mode = local.env == "production" ? "staged_if_needing_reboot" : "auto"
|
||||
}
|
||||
|
||||
generate "backend" {
|
||||
|
||||
@@ -92,6 +92,17 @@ variable "worker_nics" {
|
||||
}))
|
||||
}
|
||||
|
||||
# How machine-config changes reach the nodes. "auto" lets the provider reboot a
|
||||
# node whenever the change needs it: every node at once, since the applies are
|
||||
# not ordered. "staged_if_needing_reboot" dry-runs first and stages any
|
||||
# reboot-requiring change instead, so the node keeps its running config until an
|
||||
# operator reboots it; the dry-run needs the node reachable at plan time and
|
||||
# falls back to "auto" when it isn't.
|
||||
variable "apply_mode" {
|
||||
type = string
|
||||
default = "auto"
|
||||
}
|
||||
|
||||
# True only during initial bring-up of a brand-new env, before the Netbird
|
||||
# extension has registered any node. Drop back to false once each node is on
|
||||
# the netbird mesh, so future applies go via the vRack and the ingress firewall
|
||||
|
||||
@@ -12,6 +12,7 @@ resource "talos_machine_configuration_apply" "worker" {
|
||||
client_configuration = talos_machine_secrets.this.client_configuration
|
||||
machine_configuration_input = data.talos_machine_configuration.worker.machine_configuration
|
||||
node = local.worker_endpoint_ips[each.key]
|
||||
apply_mode = var.apply_mode
|
||||
|
||||
on_destroy = {
|
||||
reboot = true
|
||||
|
||||
@@ -117,6 +117,11 @@ talosctl -n 10.150.200.10 health # any CP private IP; the talosconfig also lis
|
||||
| Lint docs | `mise run //:md:lint` |
|
||||
| Fetch kubeconfig / talosconfig | `mise run //deployment/modules/talos/cluster:kubeconfig` / `mise run //deployment/modules/talos/cluster:talosconfig` (see [Cluster access](#cluster-access)) |
|
||||
|
||||
## Apply guardrails
|
||||
|
||||
* **Production never reboots unattended.** The production Talos module applies with `staged_if_needing_reboot`: a machine-config change that needs a reboot is written to the node but stays inactive until the node reboots (the plan shows `resolved_apply_mode = "staged"` for those nodes). Roll them yourself, one at a time, checking `talosctl health` between nodes and doing the elected NetBird routing peer last (rebooting it drops the vRack route until another peer is elected). Staging applies with `auto`, so it reboots on its own and proves the change first.
|
||||
* **Servers cannot be destroyed by an apply.** The OVH control-plane instances, the bare-metal workers, and the Talos machine secrets carry `prevent_destroy`; a plan that would replace or delete one fails instead. Tearing one down deliberately means removing that lifecycle rule in the same change.
|
||||
|
||||
## Tooling
|
||||
|
||||
* **OpenTofu** + **Terragrunt** for IaC; **mise** drives tool versions and task wrappers. Version pins live in `.mise/config.toml` and each module's lock file.
|
||||
|
||||
Reference in New Issue
Block a user