From 1f9b84bb1d294f29a60187f494733615f4d39508 Mon Sep 17 00:00:00 2001 From: Antoine Lecompte <38678863+nutgood@users.noreply.github.com> Date: Thu, 16 Jul 2026 09:56:40 -0400 Subject: [PATCH] feat(prod): continue prod (#267) --- .mise/config.toml | 6 - .mise/tasks/hetzner/talos-image | 76 ----- kubernetes/apps/prod/htz-fsn1/cilium-bgp.yaml | 3 +- .../htz-fsn1/{openebs => }/diskpools.yaml | 0 .../htz-fsn1/{ => infra}/cert-manager.yaml | 0 .../apps/prod/htz-fsn1/{ => infra}/envoy.yaml | 0 .../prod/htz-fsn1/infra/kustomization.yaml | 16 ++ .../{ => infra}/namespace-envoy-system.yaml | 0 .../namespace-openebs.yaml} | 0 .../htz-fsn1/{openebs => infra}/openebs.yaml | 0 .../apps/prod/htz-fsn1/kustomization.yaml | 19 +- .../apps/prod/htz-fsn1/lb-return-route.yaml | 2 +- .../apps/prod/htz-fsn1/netops/hyperglass.yaml | 5 +- .../prod/htz-fsn1/netops/kustomization.yaml | 8 +- .../apps/prod/htz-fsn1/netops/namespace.yaml | 19 -- .../prod/htz-fsn1/openebs/kustomization.yaml | 7 - kubernetes/clusters/prod/htz-fsn1/apps.yaml | 64 ++++- tf/.env.prod | 5 - tf/deployment/prod/htz-fsn1/fabric/README.md | 25 +- tf/deployment/prod/htz-fsn1/fabric/fabric.tf | 24 ++ tf/deployment/prod/htz-fsn1/fabric/netbox.tf | 6 +- tf/deployment/prod/htz-fsn1/netbird/dns.tf | 19 +- .../prod/htz-fsn1/netbird/netbird.auto.tfvars | 4 +- .../prod/htz-fsn1/netbird/netbird.tf | 31 +-- .../prod/htz-fsn1/talos/.terraform.lock.hcl | 65 ----- tf/deployment/prod/htz-fsn1/talos/README.md | 192 ++++++++----- .../prod/htz-fsn1/talos/addressing.tf | 17 +- .../htz-fsn1/talos/cilium-values.yaml.tftpl | 22 +- tf/deployment/prod/htz-fsn1/talos/cilium.tf | 9 +- .../prod/htz-fsn1/talos/clusters.auto.tfvars | 56 ++-- .../prod/htz-fsn1/talos/controlplane.tf | 184 ++++++------- .../prod/htz-fsn1/talos/discovery.tf | 7 +- tf/deployment/prod/htz-fsn1/talos/firewall.tf | 15 +- tf/deployment/prod/htz-fsn1/talos/flux.tf | 4 +- .../prod/htz-fsn1/talos/hcloud-firewall.tf | 29 -- tf/deployment/prod/htz-fsn1/talos/image.tf | 31 +-- .../prod/htz-fsn1/talos/migrations.tf | 33 --- .../prod/htz-fsn1/talos/netops-secrets.tf | 91 ++++++ tf/deployment/prod/htz-fsn1/talos/network.tf | 23 -- tf/deployment/prod/htz-fsn1/talos/outputs.tf | 23 +- .../prod/htz-fsn1/talos/providers.tf | 16 +- .../prod/htz-fsn1/talos/schematic-worker.yaml | 11 - .../prod/htz-fsn1/talos/schematic.yaml | 15 +- tf/deployment/prod/htz-fsn1/talos/talos.tf | 260 ++++++------------ .../prod/htz-fsn1/talos/terragrunt.hcl | 11 +- .../prod/htz-fsn1/talos/variables.tf | 88 +++--- tf/deployment/prod/htz-fsn1/talos/versions.tf | 17 -- tf/deployment/prod/htz-fsn1/talos/workers.tf | 33 ++- tf/shared/modules/core-fabric/chassis.tf | 10 +- tf/shared/modules/core-fabric/interfaces.tf | 29 ++ tf/shared/modules/core-fabric/variables.tf | 42 ++- tf/shared/modules/core-fabric/vlans.tf | 23 +- tf/shared/modules/fabric-addressing/main.tf | 17 +- .../modules/fabric-addressing/outputs.tf | 9 +- tf/shared/modules/fabric-netbox/ipam.tf | 10 +- 55 files changed, 806 insertions(+), 925 deletions(-) delete mode 100755 .mise/tasks/hetzner/talos-image rename kubernetes/apps/prod/htz-fsn1/{openebs => }/diskpools.yaml (100%) rename kubernetes/apps/prod/htz-fsn1/{ => infra}/cert-manager.yaml (100%) rename kubernetes/apps/prod/htz-fsn1/{ => infra}/envoy.yaml (100%) create mode 100644 kubernetes/apps/prod/htz-fsn1/infra/kustomization.yaml rename kubernetes/apps/prod/htz-fsn1/{ => infra}/namespace-envoy-system.yaml (100%) rename kubernetes/apps/prod/htz-fsn1/{openebs/namespace.yaml => infra/namespace-openebs.yaml} (100%) rename kubernetes/apps/prod/htz-fsn1/{openebs => infra}/openebs.yaml (100%) delete mode 100644 kubernetes/apps/prod/htz-fsn1/netops/namespace.yaml delete mode 100644 kubernetes/apps/prod/htz-fsn1/openebs/kustomization.yaml delete mode 100644 tf/deployment/prod/htz-fsn1/talos/hcloud-firewall.tf delete mode 100644 tf/deployment/prod/htz-fsn1/talos/migrations.tf create mode 100644 tf/deployment/prod/htz-fsn1/talos/netops-secrets.tf delete mode 100644 tf/deployment/prod/htz-fsn1/talos/network.tf delete mode 100644 tf/deployment/prod/htz-fsn1/talos/schematic-worker.yaml diff --git a/.mise/config.toml b/.mise/config.toml index b02063bd..e9def14e 100644 --- a/.mise/config.toml +++ b/.mise/config.toml @@ -26,12 +26,6 @@ opentofu = "1.11.5" terragrunt = "0.99.4" # ansible/mgmt convergence (mgmt:ansible task, run from CI on prod apply). "pipx:ansible-core" = "2.18.1" -# Hetzner Cloud — build/upload the Talos hcloud snapshot + manage images -# (hetzner:talos-image). hcloud-upload-image spins a temporary rescue server, -# dd's the factory raw image, and snapshots it (hcloud can't boot the Talos ISO). -hcloud = "1.66.0" -"github:apricote/hcloud-upload-image" = "1.5.0" - [tasks.dev] description = "Start all services in development mode" depends = ["install:deps", "common:build", "docker:start"] diff --git a/.mise/tasks/hetzner/talos-image b/.mise/tasks/hetzner/talos-image deleted file mode 100755 index 4171a06c..00000000 --- a/.mise/tasks/hetzner/talos-image +++ /dev/null @@ -1,76 +0,0 @@ -#!/usr/bin/env bash -#MISE description="Build + upload the Talos hcloud snapshot for the prod cluster. The schematic is TF-managed (talos_image_factory_schematic); this reads its image URL from tofu output. Idempotent unless FORCE=1." -# hcloud can't boot the Talos ISO, so the CP VMs need a Talos *snapshot*. -# hcloud-upload-image builds it the only way possible: spin a temporary rescue -# server, dd the factory hcloud-amd64 raw image, snapshot, tear down. The talos -# stack's data.hcloud_image then resolves it by label. -# -# mise run hetzner:talos-image # build for the prod father cluster -# FORCE=1 mise run hetzner:talos-image # rebuild even if a snapshot exists -set -euo pipefail -ROOT=$(git rev-parse --show-toplevel) -STACK="${TALOS_STACK:-tf/deployment/prod/htz-fsn1/talos}" -TFVARS="$ROOT/$STACK/clusters.auto.tfvars" -[ -f "$TFVARS" ] || { echo "talos-image: tfvars not found: $TFVARS" >&2; exit 1; } - -# OVH S3 backend cert verification on macOS (same shim as the infra:* tasks). -if [ -z "${SSL_CERT_FILE:-}" ] && [ "$(uname -s)" = "Darwin" ] && [ -f /etc/ssl/cert.pem ]; then - export SSL_CERT_FILE=/etc/ssl/cert.pem AWS_CA_BUNDLE=/etc/ssl/cert.pem -fi - -val() { grep -E "^[[:space:]]*$1[[:space:]]*=" "$TFVARS" | head -1 | sed -E 's/[^"]*"([^"]+)".*/\1/'; } -NAME=$(val name); VERSION=$(val talos_version); LOCATION=$(val cp_location) -SELECTOR="os=talos,cluster=${NAME},arch=amd64,version=${VERSION}" - -tg() { OP_ENV_FILE=tf/.env.prod "$ROOT/tf/op-run.sh" terragrunt --working-dir "$STACK" "$@"; } - -# The schematic is TF-managed (talos_image_factory_schematic). Pull its id straight -# from state (grep the exact `id =` line — robust against terragrunt log noise); -# if it isn't applied yet, create just it (targeted, dependency-free — touches no -# servers/data sources) and re-read. Then build the factory URLs from the id. -read_schematic_id() { - tg state show -no-color talos_image_factory_schematic.this 2>/dev/null \ - | grep -E '^[[:space:]]+id[[:space:]]+=' | head -1 | sed -E 's/.*"([^"]+)".*/\1/' -} -ID=$(read_schematic_id || true) -if [ -z "$ID" ]; then - echo "talos-image: registering the Image Factory schematic in TF (targeted apply)…" - tg apply -target=talos_image_factory_schematic.this -auto-approve - ID=$(read_schematic_id) -fi -[ -n "$ID" ] || { echo "talos-image: could not resolve the schematic id from $STACK" >&2; exit 1; } -URL="https://factory.talos.dev/image/${ID}/v${VERSION}/hcloud-amd64.raw.xz" -METAL_URL="https://factory.talos.dev/image/${ID}/v${VERSION}/metal-amd64.raw.xz" - -# Read-only hcloud token from 1Password (CI: OP_SERVICE_ACCOUNT_TOKEN; dev: -# interactive team-futo sign-in — same pattern as infra:plan). -if [ -z "${HCLOUD_TOKEN:-}" ]; then - if [ -n "${OP_SERVICE_ACCOUNT_TOKEN:-}" ]; then - HCLOUD_TOKEN=$(op read "op://yucca_tf_prod/HCLOUD_API_TOKEN/password") - else - HCLOUD_TOKEN=$(op read --account "${OP_ACCOUNT:-team-futo}" "op://yucca_tf_prod/HCLOUD_API_TOKEN/password") - fi - export HCLOUD_TOKEN -fi - -echo "cluster=$NAME talos=v$VERSION schematic=$ID" -echo " hcloud image : $URL" -echo " metal image : $METAL_URL (workers dd this in rescue)" - -# Idempotency: skip when a matching snapshot already exists. -EXISTING=$(hcloud image list --type snapshot --selector "$SELECTOR" \ - --output noheader --output columns=id 2>/dev/null || true) -if [ -n "$EXISTING" ] && [ -z "${FORCE:-}" ]; then - echo "talos-image: snapshot already exists (id ${EXISTING}). Set FORCE=1 to rebuild." - exit 0 -fi - -echo "talos-image: uploading the Talos v$VERSION hcloud snapshot (spins a temporary server)…" -hcloud-upload-image upload \ - --image-url "$URL" \ - --architecture x86 \ - --compression xz \ - --location "$LOCATION" \ - --labels "$SELECTOR" - -echo "talos-image: done — data.hcloud_image.talos (selector os=talos,cluster=${NAME},arch=amd64) now resolves." diff --git a/kubernetes/apps/prod/htz-fsn1/cilium-bgp.yaml b/kubernetes/apps/prod/htz-fsn1/cilium-bgp.yaml index 4eb141e9..edeeca0b 100644 --- a/kubernetes/apps/prod/htz-fsn1/cilium-bgp.yaml +++ b/kubernetes/apps/prod/htz-fsn1/cilium-bgp.yaml @@ -86,7 +86,8 @@ kind: CiliumBGPClusterConfig metadata: name: father spec: - # Workers only — the CPs live on the hcloud net, not the fabric, so they can't peer. + # Workers only — the CPs live on the kube-cp VLAN; the spine's iBGP group peers + # from the kube VLAN (10.40.10.0/24) only. nodeSelector: matchExpressions: - key: node-role.kubernetes.io/control-plane diff --git a/kubernetes/apps/prod/htz-fsn1/openebs/diskpools.yaml b/kubernetes/apps/prod/htz-fsn1/diskpools.yaml similarity index 100% rename from kubernetes/apps/prod/htz-fsn1/openebs/diskpools.yaml rename to kubernetes/apps/prod/htz-fsn1/diskpools.yaml diff --git a/kubernetes/apps/prod/htz-fsn1/cert-manager.yaml b/kubernetes/apps/prod/htz-fsn1/infra/cert-manager.yaml similarity index 100% rename from kubernetes/apps/prod/htz-fsn1/cert-manager.yaml rename to kubernetes/apps/prod/htz-fsn1/infra/cert-manager.yaml diff --git a/kubernetes/apps/prod/htz-fsn1/envoy.yaml b/kubernetes/apps/prod/htz-fsn1/infra/envoy.yaml similarity index 100% rename from kubernetes/apps/prod/htz-fsn1/envoy.yaml rename to kubernetes/apps/prod/htz-fsn1/infra/envoy.yaml diff --git a/kubernetes/apps/prod/htz-fsn1/infra/kustomization.yaml b/kubernetes/apps/prod/htz-fsn1/infra/kustomization.yaml new file mode 100644 index 00000000..25665ea7 --- /dev/null +++ b/kubernetes/apps/prod/htz-fsn1/infra/kustomization.yaml @@ -0,0 +1,16 @@ +--- +# prod@htz-fsn1 INFRA layer — the CRD/operator providers (cert-manager, +# envoy-gateway + Gateway API, OpenEBS/mayastor) + their namespaces. Applied by +# the `cluster-infra` Flux Kustomization (clusters/prod/htz-fsn1/apps.yaml), +# which `cluster-apps` dependsOn: everything here must be READY (healthChecks → +# operators running → CRDs registered) before the app layer's custom resources +# (Certificates, Gateways, EnvoyProxy, DiskPools) are even dry-run — the flat +# single-layer tree deadlocked on exactly that during the 2026-07 rebuild. +apiVersion: kustomize.config.k8s.io/v1beta1 +kind: Kustomization +resources: + - ./namespace-envoy-system.yaml + - ./namespace-openebs.yaml + - ./cert-manager.yaml + - ./envoy.yaml + - ./openebs.yaml diff --git a/kubernetes/apps/prod/htz-fsn1/namespace-envoy-system.yaml b/kubernetes/apps/prod/htz-fsn1/infra/namespace-envoy-system.yaml similarity index 100% rename from kubernetes/apps/prod/htz-fsn1/namespace-envoy-system.yaml rename to kubernetes/apps/prod/htz-fsn1/infra/namespace-envoy-system.yaml diff --git a/kubernetes/apps/prod/htz-fsn1/openebs/namespace.yaml b/kubernetes/apps/prod/htz-fsn1/infra/namespace-openebs.yaml similarity index 100% rename from kubernetes/apps/prod/htz-fsn1/openebs/namespace.yaml rename to kubernetes/apps/prod/htz-fsn1/infra/namespace-openebs.yaml diff --git a/kubernetes/apps/prod/htz-fsn1/openebs/openebs.yaml b/kubernetes/apps/prod/htz-fsn1/infra/openebs.yaml similarity index 100% rename from kubernetes/apps/prod/htz-fsn1/openebs/openebs.yaml rename to kubernetes/apps/prod/htz-fsn1/infra/openebs.yaml diff --git a/kubernetes/apps/prod/htz-fsn1/kustomization.yaml b/kubernetes/apps/prod/htz-fsn1/kustomization.yaml index f8fb9006..8f0b4d85 100644 --- a/kubernetes/apps/prod/htz-fsn1/kustomization.yaml +++ b/kubernetes/apps/prod/htz-fsn1/kustomization.yaml @@ -1,13 +1,15 @@ --- -# prod@htz-fsn1 cluster overlay — father. +# prod@htz-fsn1 cluster overlay — father: the APP layer (custom resources + +# workloads). The CRD/operator providers live in ./infra, applied by the +# `cluster-infra` Flux Kustomization that this layer dependsOn — see +# infra/kustomization.yaml for why (fresh-cluster dry-run deadlock). # # CURRENT SCOPE: Flux owns the cluster BASELINE (coredns, the Cilium BGP LB -# config, the netops stack, OpenEBS pools) — adopted from the hand-applied -# bring-up state. The platform/infra components and the yucca app set are -# DELIBERATELY not enabled yet: +# config, the netops stack, OpenEBS pools). The platform/infra components and +# the yucca app set are DELIBERATELY not enabled yet: # - components/infra needs real cluster-settings (RGW endpoint, o11y vmauth) -# and would collide with the in-cluster OpenEBS install (openebs/ here owns -# it via HelmRelease instead). +# and would collide with the in-cluster OpenEBS install (infra/openebs.yaml +# owns it via HelmRelease instead). # - components/roles/primary is the yucca WORKLOAD set — explicitly held back # until prod launch. # Re-enable by uncommenting `components:` below. @@ -18,10 +20,7 @@ resources: - ./coredns.yaml - ./cilium-bgp.yaml - ./lb-return-route.yaml - - ./cert-manager.yaml - - ./namespace-envoy-system.yaml - - ./envoy.yaml - - ./openebs + - ./diskpools.yaml - ./netops # components: # - ../../../components/infra diff --git a/kubernetes/apps/prod/htz-fsn1/lb-return-route.yaml b/kubernetes/apps/prod/htz-fsn1/lb-return-route.yaml index 2e9aacc2..cbde4221 100644 --- a/kubernetes/apps/prod/htz-fsn1/lb-return-route.yaml +++ b/kubernetes/apps/prod/htz-fsn1/lb-return-route.yaml @@ -26,7 +26,7 @@ spec: app: lb-return-route spec: hostNetwork: true - # Workers only — the CPs are on the hcloud net, not the fabric. + # Workers only — the CPs are on the kube-cp VLAN, not the kube VLAN. affinity: nodeAffinity: requiredDuringSchedulingIgnoredDuringExecution: diff --git a/kubernetes/apps/prod/htz-fsn1/netops/hyperglass.yaml b/kubernetes/apps/prod/htz-fsn1/netops/hyperglass.yaml index 5e47782f..8b0d4e94 100644 --- a/kubernetes/apps/prod/htz-fsn1/netops/hyperglass.yaml +++ b/kubernetes/apps/prod/htz-fsn1/netops/hyperglass.yaml @@ -3,9 +3,8 @@ # http://lg.father.fsn.htz.yucca.futo.network # # devices.yaml is NOT here: it embeds the netops password (netmiko can't key-auth -# through hyperglass config), so it's a Secret rendered at deploy time from -# op://yucca_tf_prod/NETOPS_FABRIC_PASSWORD (see the deploy script / runbook): -# kubectl -n netops create secret generic hyperglass-devices --from-file=devices.yaml +# through hyperglass config), so it's a Secret the talos stack renders from +# op://yucca_tf_prod/NETOPS_FABRIC_PASSWORD (netops-secrets.tf). apiVersion: v1 kind: ConfigMap metadata: diff --git a/kubernetes/apps/prod/htz-fsn1/netops/kustomization.yaml b/kubernetes/apps/prod/htz-fsn1/netops/kustomization.yaml index 8f525342..4ec1d4d5 100644 --- a/kubernetes/apps/prod/htz-fsn1/netops/kustomization.yaml +++ b/kubernetes/apps/prod/htz-fsn1/netops/kustomization.yaml @@ -1,11 +1,11 @@ --- -# The father netops stack. Secrets (netops-ssh, grafana-admin, hyperglass-devices) -# are NOT in git — they are created from 1Password (yucca_tf_prod: NETOPS_FABRIC_*, -# FATHER_GRAFANA_ADMIN) at bring-up; Flux only manages the workloads around them. +# The father netops stack. The namespace + its Secrets (netops-ssh, +# grafana-admin, hyperglass-devices) are NOT in git — the talos stack provisions +# them from 1Password (tf/deployment/prod/htz-fsn1/talos/netops-secrets.tf); +# Flux only manages the workloads around them. apiVersion: kustomize.config.k8s.io/v1beta1 kind: Kustomization resources: - - ./namespace.yaml - ./networkpolicies.yaml - ./junos-exporter.yaml - ./victoria-metrics.yaml diff --git a/kubernetes/apps/prod/htz-fsn1/netops/namespace.yaml b/kubernetes/apps/prod/htz-fsn1/netops/namespace.yaml deleted file mode 100644 index e96674f4..00000000 --- a/kubernetes/apps/prod/htz-fsn1/netops/namespace.yaml +++ /dev/null @@ -1,19 +0,0 @@ -# netops — in-cluster network operations stack for the htz-fsn1 fabric: -# junos_exporter + vmagent + VictoriaMetrics (30d high-granularity buffer) + -# Grafana, plus hyperglass/smokeping/oxidized. Everything is exposed ONLY on -# internal LoadBalancer VIPs (lb-internal pool, NetBird-reachable — never public). -# Applied by hand today; to be adopted by Flux when GitOps lands on father. -# -# privileged PodSecurity: VictoriaMetrics persists to a hostPath (no CSI on -# father yet — the NVMe data disks are unprovisioned); baseline forbids hostPath. -apiVersion: v1 -kind: Namespace -metadata: - name: netops - labels: - pod-security.kubernetes.io/enforce: privileged - annotations: - # Never prune: this namespace holds hand-created secrets (netops-ssh, - # grafana-admin, hyperglass-devices) and the VictoriaMetrics PVC — a prune - # would destroy state only a manual runbook can restore. - kustomize.toolkit.fluxcd.io/prune: disabled diff --git a/kubernetes/apps/prod/htz-fsn1/openebs/kustomization.yaml b/kubernetes/apps/prod/htz-fsn1/openebs/kustomization.yaml deleted file mode 100644 index ec6a1158..00000000 --- a/kubernetes/apps/prod/htz-fsn1/openebs/kustomization.yaml +++ /dev/null @@ -1,7 +0,0 @@ ---- -apiVersion: kustomize.config.k8s.io/v1beta1 -kind: Kustomization -resources: - - ./namespace.yaml - - ./openebs.yaml - - ./diskpools.yaml diff --git a/kubernetes/clusters/prod/htz-fsn1/apps.yaml b/kubernetes/clusters/prod/htz-fsn1/apps.yaml index 4e5a3ce9..72011ba3 100644 --- a/kubernetes/clusters/prod/htz-fsn1/apps.yaml +++ b/kubernetes/clusters/prod/htz-fsn1/apps.yaml @@ -1,10 +1,55 @@ # yaml-language-server: $schema=https://k8s-schemas.home-operations.com/kustomize.toolkit.fluxcd.io/kustomization_v1.json --- -# cluster-apps entry point for prod@htz-fsn1. Authored ahead of the cluster (the -# prod Talos/flux stack isn't built yet); activates once that stack provisions -# flux and it syncs kubernetes/clusters/prod/htz-fsn1. Precedence (last wins): -# cluster-settings-generated (TF) -> cluster-settings (human) -> image-versions -# (the committed, CI-promoted prod tag); keys are disjoint by design. +# prod@htz-fsn1 entry points — TWO layers, so a fresh cluster can't deadlock: +# the 2026-07 rebuild proved a flat tree wedges itself (kustomize-controller +# server-side dry-runs every object, so any CR whose CRD is missing blocks the +# WHOLE apply — including the HelmReleases that would install those CRDs). +# +# cluster-infra operators/CRD providers (apps/prod/htz-fsn1/infra) — wait: true, +# so Ready ⇒ operators running ⇒ CRDs registered. +# cluster-apps everything else (CRs + workloads) — dependsOn cluster-infra. +# +# Substitution precedence (last wins): cluster-settings-generated (TF) -> +# cluster-settings (human) -> image-versions (the committed, CI-promoted prod +# tag); keys are disjoint by design. Injected into the NESTED Kustomizations via +# the patches block (both layers' direct objects use no ${vars} themselves). +apiVersion: kustomize.toolkit.fluxcd.io/v1 +kind: Kustomization +metadata: + name: cluster-infra + namespace: flux-system +spec: + interval: 1h + retryInterval: 2m + path: ./kubernetes/apps/prod/htz-fsn1/infra + prune: true + sourceRef: + kind: GitRepository + name: flux-system + namespace: flux-system + # The health gate cluster-apps depends on: every nested Kustomization here + # carries healthChecks on its HelmRelease, so wait covers operator readiness. + wait: true + timeout: 10m + patches: + - target: + group: kustomize.toolkit.fluxcd.io + kind: Kustomization + patch: |- + apiVersion: kustomize.toolkit.fluxcd.io/v1 + kind: Kustomization + metadata: + name: _ + spec: + postBuild: + substituteFrom: + - kind: ConfigMap + name: cluster-settings-generated + - kind: ConfigMap + name: cluster-settings + - kind: ConfigMap + name: image-versions +--- apiVersion: kustomize.toolkit.fluxcd.io/v1 kind: Kustomization metadata: @@ -12,11 +57,12 @@ metadata: namespace: flux-system spec: interval: 1h - # Fast retry: this Kustomization applies CRs (Certificates, Gateways, - # DiskPools) whose CRDs its own children install — on fresh bootstrap the - # first apply races them, and without retryInterval a failure waits the - # full 1h interval. + # Fast retry: CRD registration can trail the infra layer's Ready by moments + # (mayastor's diskpool operator creates its CRD at startup) — without + # retryInterval a dry-run failure waits the full 1h interval. retryInterval: 2m + dependsOn: + - name: cluster-infra path: ./kubernetes/apps/prod/htz-fsn1 prune: true sourceRef: diff --git a/tf/.env.prod b/tf/.env.prod index b9d6a919..87f49439 100644 --- a/tf/.env.prod +++ b/tf/.env.prod @@ -32,11 +32,6 @@ export HETZNER_ROBOT_PASSWORD=op://yucca_tf_prod/HETZNER_WEBSERVICE_API_PASSWORD # one). The cloudflare provider reads CLOUDFLARE_API_TOKEN directly. export CLOUDFLARE_API_TOKEN=op://yucca_tf_prod/CLOUDFLARE_API_TOKEN/password -# ── Hetzner Cloud API (prod/htz-fsn1/talos — hcloud provider) ───────────────── -# Per-project read/write token for the control-plane VMs, network, snapshot + LB. -# TODO(prod): create the hcloud project + token, store at the path below. -export HCLOUD_TOKEN=op://yucca_tf_prod/HCLOUD_API_TOKEN/password - # ── NetBird setup keys (prod/htz-fsn1/talos — node-level overlay) ───────────── # Minted by the netbird stack. WORKER key (group: talos) — joins the bare-metal # workers to the prod htz-fsn1 NetBird network. diff --git a/tf/deployment/prod/htz-fsn1/fabric/README.md b/tf/deployment/prod/htz-fsn1/fabric/README.md index ce3b99d5..24e0da24 100644 --- a/tf/deployment/prod/htz-fsn1/fabric/README.md +++ b/tf/deployment/prod/htz-fsn1/fabric/README.md @@ -2,7 +2,8 @@ Manages the Falkenstein (site 40) switch fabric as code: -- **spine** (`corenetsw` VC) — shared site core: VC, 100G→4×25G breakout, VLAN stretch. +- **spine** (`corenetsw` VC) — shared site core: VC, per-port breakout (ports 1-3 + 100G→4×25G, port 0 100G→4×10G for the father CPs), VLAN stretch. - **cls1** (`cls1netsw` VC) — ceph cluster 1's leaf pair: public/private VLANs, IRB gateways, the `NO-CROSS-VLAN` filter, and the 48 server LAGs. @@ -15,22 +16,24 @@ Each ceph cluster = one leaf pair; the spine is shared across clusters. | site supernet | `10..0.0/16` | `10.40.0.0/16` | | management (vme) | `10..5.0/24` | `10.40.5.0/24` (spine `.115`, leaf `.125`) | | kube (VLAN) | `10...0/24` | `10.40.10.0/24` → vlan 10 | -| kube-cp (hcloud) | `10...0/24` | `10.40.11.0/24` (gw `.1`, CP VMs etcd + API LB) | +| kube-cp (VLAN) | `10...0/24` | `10.40.11.0/24` → vlan 11 (gw `.1` = spine IRB) | | cluster `n` /20 | `10...0/20` | `10.40.16.0/20` | | public (VLAN) | cluster /20, /23 idx 2 | `10.40.20.0/23` → vlan 20 | | private (VLAN) | cluster /20, /23 idx 3 | `10.40.22.0/23` → vlan 22 | | leaf vme | `.125 + (n-1)*10` | `.125` | -VLAN id == the network's third octet; gateway = `.1` (IRB on the leaf). +VLAN id == the network's third octet; gateway = `.1` (IRB on the leaf, except the +site-global kube/kube-cp VLANs, whose IRBs live on the spine). -> **`kube-cp` is not a fabric VLAN.** It's a small isolated **Hetzner Cloud private -> subnet** holding only the cloud control-plane VMs (etcd CP↔CP) + the API LB's -> private IP. CP↔worker control traffic and worker→API ride the **NetBird WireGuard -> mesh** (node IPs are NetBird addresses), and worker↔worker east-west rides the -> `kube` fabric net (`10.40.10.0/24`) at 50G via Cilium BGP. The API endpoint is a -> **Hetzner Cloud LB** (no L2 VIP — hcloud private nets are anti-spoofed/routed). -> `kube-cp` is carved from the site supernet only for collision-free IPAM and is -> **never** configured on the Junos switches. +> **`kube-cp` is the control-plane VLAN.** The father bare-metal CPs are its only +> members (etcd CP↔CP + apiserver + the Talos-elected API **VIP** `10.40.11.5`), +> hanging off the spine's port-0 **4×10G** breakout (ae4-6, one leg per VC member). +> The spine routes kube↔kube-cp between its two IRBs (`10.40.10.1` / `10.40.11.1`), +> which is how worker kubelets and the apiserver reach each other; worker↔worker +> east-west rides the `kube` VLAN at 50G. Operators reach the API over the NetBird +> kube-cp route (the CPs are the route peers). Historically kube-cp was an isolated +> Hetzner Cloud subnet for the retired cloud CP VMs + API LB — same CIDR, so the +> API DNS record and etcd addressing carried over unchanged. ## Layout diff --git a/tf/deployment/prod/htz-fsn1/fabric/fabric.tf b/tf/deployment/prod/htz-fsn1/fabric/fabric.tf index 95a847b2..1c9a8a2a 100644 --- a/tf/deployment/prod/htz-fsn1/fabric/fabric.tf +++ b/tf/deployment/prod/htz-fsn1/fabric/fabric.tf @@ -22,6 +22,11 @@ module "core" { vc_member_serials = var.spine_vc_serials + # Port 0 carries the father control-plane breakout at 10G (the CPs' Intel 82599 + # NICs are 10G-only; needs the QSFP+ 4x10G breakout cables — see cp_node_lags); + # ports 1-3 stay 25G (workers + mgmt + spares). + breakout_ports = { 0 = "10g", 1 = "25g", 2 = "25g", 3 = "25g" } + # father's bare-metal kube workers hang off the core (channelized 25G breakouts of # port 2, one leg per VC member). Each ae bundles the two ports cabled to one node # (pairs derived from LLDP — consecutive MACs on the node's dual-port Broadcom NIC): @@ -32,6 +37,23 @@ module "core" { ae3 = ["et-0/0/2:1", "et-1/0/2:1"] } + # father's bare-metal control planes: port-0 breakout legs at 10G (xe-), one leg + # per VC member, trunking the kube-cp VLAN. Pairing VERIFIED 2026-07-15 via MAC + # learning against the maintenance-mode nodes (NIC port 1 → FPC 0, port 2 → + # FPC 1, same leg index on both members): + # ae4 = harlan …0a:fe:c8/ca ae5 = imelda …09:68:68/6a ae6 = roscoe …65:07:40/42 + cp_node_lags = { + ae4 = ["xe-0/0/0:2", "xe-1/0/0:2"] + ae5 = ["xe-0/0/0:1", "xe-1/0/0:1"] + ae6 = ["xe-0/0/0:0", "xe-1/0/0:0"] + } + + # kube-cp VLAN + its spine IRB (10.40.11.1) — the spine routes kube↔kube-cp. + kube_cp = { + vlan_id = module.addr_site.kube_cp_vlan_id + cidr = module.addr_site.kube_cp_cidr + } + # Cilium node iBGP for LoadBalancer VIPs — the spine gets its first IRB (the kube net's # .1 gateway) and dynamic-peers the workers from the kube subnet, accepting the LB /32s # they advertise (covered by the transit aggregate, so reachable north-south). The @@ -54,6 +76,8 @@ module "core" { interfaces = [ "et-0/0/2:1", "et-0/0/2:2", "et-0/0/2:3", "et-1/0/2:1", "et-1/0/2:2", "et-1/0/2:3", + "xe-0/0/0:0", "xe-0/0/0:1", "xe-0/0/0:2", + "xe-1/0/0:0", "xe-1/0/0:1", "xe-1/0/0:2", "et-0/0/3:0", "et-1/0/3:0", "et-0/0/27", "et-0/0/30", "et-0/0/31", "et-1/0/30", "et-1/0/31", diff --git a/tf/deployment/prod/htz-fsn1/fabric/netbox.tf b/tf/deployment/prod/htz-fsn1/fabric/netbox.tf index 7e039148..ac7cdfa3 100644 --- a/tf/deployment/prod/htz-fsn1/fabric/netbox.tf +++ b/tf/deployment/prod/htz-fsn1/fabric/netbox.tf @@ -14,8 +14,9 @@ module "netbox" { # Site-global VLANs (present on every cluster). global_vlans = { - MGMT = { vid = module.addr_site.mgmt_vlan_id, prefix = module.addr_site.mgmt_cidr } - KUBE = { vid = module.addr_site.kube_vlan_id, prefix = module.addr_site.kube_cidr } + MGMT = { vid = module.addr_site.mgmt_vlan_id, prefix = module.addr_site.mgmt_cidr } + KUBE = { vid = module.addr_site.kube_vlan_id, prefix = module.addr_site.kube_cidr } + "KUBE-CP" = { vid = module.addr_site.kube_cp_vlan_id, prefix = module.addr_site.kube_cp_cidr } } clusters = { @@ -37,7 +38,6 @@ module "netbox" { # Pod/service CIDRs mirror the talos stack (talos.tf locals); the public carves # mirror the Cilium LB pools + node-egress + transit config in this stack. extra_prefixes = { - kube_cp = { prefix = module.addr_site.kube_cp_cidr, description = "Hetzner Cloud kube-cp: father CP VMs (etcd) + private API LB — not a fabric VLAN" } lb_internal = { prefix = module.addr_site.lb_internal_cidr, description = "father internal (NetBird-only) LoadBalancer VIPs — Cilium lb-internal pool, iBGP /32s to the spine" } pods = { prefix = "10.250.0.0/17", description = "father pod CIDR (Cilium, geneve over the kube VLAN)", status = "container" } services = { prefix = "10.250.128.0/17", description = "father service CIDR (ClusterIPs; kube-dns at .128.10)", status = "container" } diff --git a/tf/deployment/prod/htz-fsn1/netbird/dns.tf b/tf/deployment/prod/htz-fsn1/netbird/dns.tf index 527eb721..d2da60b6 100644 --- a/tf/deployment/prod/htz-fsn1/netbird/dns.tf +++ b/tf/deployment/prod/htz-fsn1/netbird/dns.tf @@ -96,16 +96,17 @@ resource "netbird_dns_record" "father_worker" { ttl = 300 } -# API endpoint — round-robin over the 3 CP IPs. NOT the LB: hcloud LBs refuse -# traffic from their own targets (the CPs), so the endpoint resolves straight to -# the CPs (reachable over the yucca-fsn-father-kube-cp route). +# API endpoint — the Talos-elected VIP on the kube-cp VLAN (etcd parks it on a +# healthy CP, so the record only answers where an apiserver runs). Reachable over +# the yucca-fsn-father-kube-cp route. (Historically round-robin over the CP IPs — +# the retired hcloud LB refused traffic from its own targets.) resource "netbird_dns_record" "father_kube_api" { - for_each = local.father_cps - zone_id = netbird_dns_zone.yucca_internal.id - name = local.father_kube_api_fqdn - type = "A" - content = each.value - ttl = 300 + count = var.talos_discovery_enabled ? 1 : 0 + zone_id = netbird_dns_zone.yucca_internal.id + name = local.father_kube_api_fqdn + type = "A" + content = local.talos_kube.api_vip + ttl = 300 } output "kube_api_fqdn" { diff --git a/tf/deployment/prod/htz-fsn1/netbird/netbird.auto.tfvars b/tf/deployment/prod/htz-fsn1/netbird/netbird.auto.tfvars index e8e4be66..54b7d896 100644 --- a/tf/deployment/prod/htz-fsn1/netbird/netbird.auto.tfvars +++ b/tf/deployment/prod/htz-fsn1/netbird/netbird.auto.tfvars @@ -20,8 +20,8 @@ groups = { talos = { resource = true } # Talos cluster nodes → yucca-prod-htz-fsn1-talos resources = { resource = true } # routed-subnet tag → yucca-prod-htz-fsn1-resources (Network resources tag in) # CP-only subset of `talos` — the ROUTER peer group for the kube-cp network. Only - # the cloud CPs sit on the kube-cp hcloud subnet, so only they can route it; if the - # router were the whole `talos` group the bare-metal WORKERS (also `talos`) would be + # the CPs sit on the kube-cp VLAN, so only they can route it; if the router were + # the whole `talos` group the bare-metal WORKERS (also `talos`) would be # treated as routers and never install the client route to kube-cp. resource = false: # it's a routing peer group, not a yucca-reachable tag (the CPs are already reachable # via `talos`). CPs join via the talos_cp setup key below (auto_groups tags them diff --git a/tf/deployment/prod/htz-fsn1/netbird/netbird.tf b/tf/deployment/prod/htz-fsn1/netbird/netbird.tf index 7ecccd73..f0ecfec4 100644 --- a/tf/deployment/prod/htz-fsn1/netbird/netbird.tf +++ b/tf/deployment/prod/htz-fsn1/netbird/netbird.tf @@ -16,13 +16,9 @@ locals { # group (flagged `resource = true` in netbird.auto.tfvars), so the module- # generated yucca→resources policy governs access — and resources never appear # as a policy source, so they can't reach each other. - # NB: the `kube-cp` network (the routed Hetzner Cloud Network for the cloud - # control-plane VMs) is deliberately NOT advertised here. The router peers are the - # mgmt nodes, which sit on the Juniper fabric and cannot reach the hcloud subnets. - # The CP plane is reached out-of-band via hcloud public IPs + the API LB's public - # frontend (both firewalled to the NetBird/operator ranges) — see the talos stack. - # TODO(prod): for a fully-private control plane, attach the mgmt nodes to the - # kube-cp vSwitch and add module.addr_site.kube_cp_cidr to this map. + # NB: the `kube-cp` VLAN is deliberately NOT in this map — it's routed by its + # own network below (via the CPs, the talos_cp group), keeping the API plane's + # mesh path independent of the mgmt routers. routed = { mgmt = { address = module.addr_site.mgmt_cidr, description = "OOB / vme management network" } # Internal LB VIPs (Grafana + netops UIs): NetBird peer -> mgmt router -> spine @@ -48,24 +44,25 @@ locals { } } - # father's cloud control-plane subnet (kube-cp), routed via the CPs ONLY (the - # talos_cp group — the CP-only subset of talos). They're the only peers on that - # hcloud subnet. Router must NOT be the whole `talos` group: the bare-metal workers - # are also `talos`, and a routing peer doesn't install a client route for its own - # network — so if the workers were routers they'd never get the kube-cp route (and - # their pods couldn't reach the apiserver). This is how NetBird peers (operators + - # workers) reach the private API LB (10.40.11.5) + the CPs. masquerade so return - # traffic is SNAT'd to the CP's kube-cp address. + # father's control-plane VLAN (kube-cp), routed via the CPs ONLY (the talos_cp + # group — the CP-only subset of talos). They're the only peers on that VLAN. + # Router must NOT be the whole `talos` group: the bare-metal workers are also + # `talos`, and a routing peer doesn't install a client route for its own + # network — so if the workers were routers they'd never get the kube-cp mesh + # route. (Worker→apiserver traffic itself rides the fabric — a static route via + # the spine IRB pinned in the machine config — not this mesh route.) This is + # how OPERATOR/CI peers reach the API VIP (10.40.11.5) + the CPs. masquerade so + # return traffic is SNAT'd to the CP's kube-cp address. # CP membership comes from the talos_cp setup key (netbird.auto.tfvars, auto_groups # [talos, talos_cp]); the talos stack joins CPs with it and workers with the plain # `talos` key, so re-provisioning keeps the split. "yucca-fsn-father-kube-cp" = { - description = "father control-plane subnet (kube-cp), routed via the CPs (talos_cp)." + description = "father control-plane VLAN (kube-cp), routed via the CPs (talos_cp)." router = { peer_groups = ["talos_cp"], masquerade = true } resources = { kube_cp = { address = module.addr_site.kube_cp_cidr - description = "kube-cp: CP VMs (etcd) + the private API LB (10.40.11.5)." + description = "kube-cp: bare-metal CPs (etcd) + the API VIP (10.40.11.5)." groups = ["resources"] } } diff --git a/tf/deployment/prod/htz-fsn1/talos/.terraform.lock.hcl b/tf/deployment/prod/htz-fsn1/talos/.terraform.lock.hcl index 341fa6e9..104057af 100644 --- a/tf/deployment/prod/htz-fsn1/talos/.terraform.lock.hcl +++ b/tf/deployment/prod/htz-fsn1/talos/.terraform.lock.hcl @@ -64,50 +64,6 @@ provider "registry.opentofu.org/hashicorp/kubernetes" { ] } -provider "registry.opentofu.org/hashicorp/random" { - version = "3.9.0" - constraints = "~> 3.6" - hashes = [ - "h1:U8KXqGCoNI9/guYbTvzgdtVk3fRthoG0UXwm1JoEpIs=", - "zh:03f1114cc20b8913523735ab76e0f0a2b16ce13c92923a53304bf85f07fc0dbc", - "zh:105b678ee72322a3067f105d7e05e940f6143238f377f6e87ff4ec909246ac2a", - "zh:55f3bbf13ea18cbace61a706566a80f25f33fe2b1780b6f3d7b582af2a05b6d2", - "zh:63adf996db48f082f7a6351eb485e219cd88795fc71e6ec60a837263ab0d2cb1", - "zh:7e99550738a4e3cc68b8a467714b0d69371025fe95e3326d5323d026d55653e9", - "zh:8342b54af3a18a37e075eeae61be57f4de2ba71b35d95c5075d402dd2c1f289d", - "zh:83ee18e32ac9dd5fc91298554b7c4cfa4c3a1db50f4c797945637cc93c0844ae", - "zh:993ecc0adbf6bd535a59fbc9b735d8c33950e6f6eb5e621d750da9b71d65d80a", - "zh:ad722bc59d4edbf1415e827fc007c0efe6e0e9462d5568bae20b34be1058a261", - "zh:ae9448e1f87b2f9a6c5197a0e9862162ec6b137cb3a3835e11522995d8939e7c", - "zh:bc9cdd3aac784f759125c6627f6f6416e8726a1c184eb9cf3e55b9edbc94c627", - "zh:c8e35b89572ba1c40a9b20022e033a3395fb8d42e7604d50c900f193ba10382e", - "zh:e2deaa8a9975ef81d9f62baed12c41286918b0a10908e0e031f13f69a3b730a1", - "zh:ee39707557210a0ab1098aa357d2cdfe502e5a312d0dbdffb09d08facc4d3fc5", - "zh:f81afe4eb63e8aa9e0ea71be6c990f0dc69cb360e7191c0742a991f4a5081b64", - ] -} - -provider "registry.opentofu.org/hetznercloud/hcloud" { - version = "1.66.0" - constraints = "~> 1.51" - hashes = [ - "h1:iVAGP8gRbZK0kJF7SiYJRt61wz0D5AF9q+WMsrAiBI0=", - "zh:1286cee6fb63dbcb18f53077bbb5e5d132a4e4d9f006af4e8d8edfc08d6bcdc8", - "zh:204460dacc044bda019a4a18b398e094289500c36913c7c9457f432adf31b8b2", - "zh:214175d50773481cbeaf9c9004e4121a3a1c9686c79424ebdc8ff189dd057d3e", - "zh:22b17bceff61cc13ad04a399ba87521356a3a134d4687273727473ae9eccf5f1", - "zh:368867dac5525c411de7e38f2e27de0a71854d1750867322ff2b9321128c88fb", - "zh:5289b75f8370bdbc4c6051d55cf33d0b1bd25dc6d71bfbd39b360249a37f1501", - "zh:81cb676aa50c5777df8fc80d4e69c9012330ae751f5e6f12bf6074bfd2e7c496", - "zh:ab08aead10643b21aa6b51af562b50492e12b9dd0ab7dca27a05aa63209b7d66", - "zh:af25c210d0570cf61ef767b2545bf9f3fb909178135f0e5e14bec0c1c9d07a63", - "zh:bcad66f4830c97118fa793723e53f8a4d27ddd34ea969ff259408842c2238331", - "zh:ce3ed323d75ae905d975925fa98c7054a7514c81276a485fc37da8232b53e39f", - "zh:d481bc0ef0c87ab1969c17777f526b2f59f823432d676145134c41a6d29bd98e", - "zh:ea7ef88df2c3ca154d86238920636d52a3c9066c7467543d3fa45f1e52ec2f7b", - ] -} - provider "registry.opentofu.org/siderolabs/talos" { version = "0.11.0" constraints = "~> 0.11" @@ -130,24 +86,3 @@ provider "registry.opentofu.org/siderolabs/talos" { ] } -provider "registry.terraform.io/futo-org/netbird" { - version = "1.0.2" - constraints = "1.0.2" - hashes = [ - "h1:CE91Uvc3FhpgpwgjVR96ueerThTZSCTcKF8Rp2+lzQw=", - "zh:1a1b727dd3971eb0f4c923c19fa95100e5b91b163ae0607f72a37d656f9d71a6", - "zh:5169993e54c6b184cdb359fb310477d85eeb99d1d91db14d367a29bf3119dc55", - "zh:573279b9532f916ee4bfb4e12faacf26e8e27df626ac37e1a5c42d5eab2350a4", - "zh:5cafd8cb073fc6774d959080c06b6faf821c3a7caaf3d876a40be8b8945a70fd", - "zh:66d40bbde6d756160a94e8c908582cb7806acccbebe0e9060b82cb4993f5adad", - "zh:6eb6493922345ec4306d8f0e95e799a6045ff3b25fd05973828e960757df5b97", - "zh:88c638c29d7dad339f1c4e90e54d0996be85a95574d5d9c5f125cb0bbb5b8902", - "zh:890df766e9b839623b1f0437355032a3c006226a6c200cd911e15ee1a9014e9f", - "zh:89cca24eeb5305cfcac51092bd0c330336af6613648f14a4c70f23a56fe4e93d", - "zh:91186b07d2ee1494bcd7e2c132ca1c61b62b2a1f2d325d2bbc4bfe6572ce327d", - "zh:aa6cbe8f5c8121d96861e293d3773cc210723fa410acf45512a337703d55d911", - "zh:d38dce0bcc61ca2482ee3b9f457ebf45929c7ebc8717487d9f1270aa3dcd3cfa", - "zh:dba9742f0a0cb5e190a10265ec00d20a4d266f8a6dd517cb4c1d8486335ed636", - "zh:e8eaabc072208c10ab554b4c89dbe32831a7d0125ab628af4a6530a02ea5a993", - ] -} diff --git a/tf/deployment/prod/htz-fsn1/talos/README.md b/tf/deployment/prod/htz-fsn1/talos/README.md index 5cd88e9d..9428be82 100644 --- a/tf/deployment/prod/htz-fsn1/talos/README.md +++ b/tf/deployment/prod/htz-fsn1/talos/README.md @@ -1,92 +1,156 @@ # prod/htz-fsn1/talos — the `father` cluster -Hybrid production Talos Kubernetes cluster: **3 Hetzner Cloud control-plane VMs + -3 Hetzner Robot bare-metal workers**, one cluster. Flux is **deferred** — this -stack brings up a healthy, Cilium-networked cluster and stops there. +All-bare-metal production Talos Kubernetes cluster: **3 Hetzner Robot control +planes + 3 Hetzner Robot workers**, every node plane on the Juniper fabric. The +previous hybrid topology (3 Hetzner Cloud CP VMs + hcloud API LB + NetBird as the +CP↔worker plane) is retired; NetBird remains on every node as the **operator / +backup plane** only. ## Topology ``` - ┌──────────── Hetzner Cloud (fsn1) ────────────┐ - │ cp-1 cp-2 cp-3 (CCX23, Talos snapshot) │ - │ • public IPv4 → NetBird + bootstrap + LB │ - │ • kube-cp 10.40.11.0/24 (eth1) → etcd │ - │ • API LB (lb11) → :6443, PRIVATE (lb_public=false) │ - └───────┬───────────────────────┬───────────────┘ - NetBird mesh│ (CP route to fabric │ private LB :6443 (api_dns_name, - via mgmt │ net, advertised) │ reached over the mesh) - ┌───────┴───────────────────────┴───────────────┐ - │ Juniper fabric — kube VLAN 10 / 10.40.10.0/24 │ - │ wk-1 .11 wk-2 .12 wk-3 .13 (bond0, 50G) │ - │ • pod↔pod rides geneve (routingMode: tunnel) over │ - │ the 50G fabric between workers │ - └─────────────────────────────────────────────────┘ + ┌──────────────── Juniper fabric (site 40) ─────────────────┐ + │ kube-cp VLAN 11 — 10.40.11.0/24 (gw .1 = spine IRB) │ + │ cp-harlan .11 cp-imelda .12 cp-roscoe .13 │ + │ API VIP 10.40.11.5 (Talos etcd-elected) │ + │ bond0 2×10G (ixgbe, spine port-0 4×10G breakout, ae4-6) │ + ├───────────────── spine routes irb.11 ↔ irb.10 ────────────┤ + │ kube VLAN 10 — 10.40.10.0/24 (gw .1 = spine IRB) │ + │ wk-jeanne .11 wk-sheron .12 wk-dianna .13 │ + │ bond0 2×25G (bnxt_en, spine port-2 4×25G breakout, ae1-3)│ + └───────────────────────────────────────────────────────────┘ + every node: onboard public NIC (DHCP) = default route/egress + + NetBird (operator plane; CPs route kube-cp) ``` | Plane | Path | Carries | |---|---|---| -| node / control | NetBird (CP) → mgmt routers → `kube` fabric | apiserver↔kubelet, CP→worker | -| API endpoint | private Hetzner Cloud LB (`api_dns_name`, NetBird-reachable) | kubelet→apiserver, operators | -| etcd | `kube-cp` hcloud private subnet (`10.40.11.0/24`) | CP↔CP | -| pod east-west | `kube` fabric VLAN 10 (50G), geneve tunnel | worker↔worker pods | +| etcd + API endpoint | `kube-cp` VLAN 11, VIP `10.40.11.5` (`api_dns_name` → VIP) | CP↔CP etcd, apiserver, VIP | +| CP↔worker | routed `kube`↔`kube-cp` via the spine IRBs (static routes in machine config) | apiserver↔kubelet, geneve | +| pod east-west | `kube` VLAN 10 (50G), geneve tunnel | worker↔worker pods | +| operators / CI | NetBird → kube-cp route (the CPs are the route peers) | talosctl/kubectl/TF | -No vSwitch, no BGP. **Cilium BGP is reserved for north-south later** (advertising -ingress/LoadBalancer VIPs to the fabric leaf, which already speaks BGP). +**Cilium BGP** (north-south LoadBalancer VIPs) stays workers-only against the +spine's VLAN-10 IRB. ## Bring-up flow (single `tf:apply`) -1. `image.tf` — look up the Talos amd64 hcloud snapshot (built once out-of-band). -2. `network.tf` — hcloud network + the kube-cp cloud subnet. -3. `controlplane.tf` — 3 CP VMs (config via `user_data`) + the API LB. -4. `talos.tf` — `talos_machine_bootstrap` against cp-1's public IP → kubeconfig. -5. `workers.tf` — `talos_machine_configuration_apply` to each worker over apid - (its fabric IP, reached via NetBird→mgmt→fabric). -6. `cilium.tf` — Cilium via Helm → post-CNI health gate → Ready cluster. +1. `image.tf` — register the schematic (one, metal, **no qemu-guest-agent**). +2. `controlplane.tf` — apid config apply to each CP (maintenance `maint_ip` on the + first pass) → install to disk (by serial) + reboot onto bond0.11. +3. `talos.tf` — `talos_machine_bootstrap` against cp-1 (`10.40.11.11`, reached + over the NetBird kube-cp route once cp-1's netbird is up) → kubeconfig. +4. `workers.tf` — apid apply to each worker (maintenance `maint_ip` first pass). +5. `cilium.tf` — Cilium via Helm → post-CNI health gate. +6. `flux.tf` — flux-operator + instance → GitOps takes over + (`kubernetes/clusters/prod/htz-fsn1`). ## Prerequisites (before `tf:apply`) -- **`HCLOUD_TOKEN`** — create the hcloud project + read/write token, store at - `op://yucca_tf_prod/HCLOUD_API_TOKEN` (see `tf/.env.prod`). -- **Talos schematic** — the extension set lives in `schematic.yaml` and is - registered with the factory by TF (`talos_image_factory_schematic`, `image.tf`); - the schematic id + image URLs derive from it. Edit `schematic.yaml` to change it. -- **hcloud snapshot** — build it once with `mise run hetzner:talos-image` (reads the - image URL from `tofu output`; idempotent, `FORCE=1` to rebuild). `image.tf` - resolves it by label. -- **NetBird setup key** — minted by the netbird stack; path in `tf/.env.prod`. -- **Workers in maintenance mode** — provisioned to Talos maintenance at their - `fabric_ip` (10.40.10.11/.12/.13). See the runbook below. -- **DNS** — after apply, point `api_dns_name` (output) at the LB public IPv4 - (output `api_dns_record`). -- **`trusted_cidrs`** — MUST include the source the TF runner dials the CP public - IPs from (CI egress / your NetBird range), or bootstrap (apid 50000) hangs. +- **Fabric** — the fabric stack applied with `breakout_ports` port 0 = 10g, the + kube-cp VLAN/IRB, and `cp_node_lags` ae4-6. Leg pairing in `../fabric/fabric.tf` + was verified 2026-07-15 by MAC-learning against the maintenance-mode nodes + (QSFP+ 4×10G breakout cables installed; all six legs link at 10G). Re-verify if + anything is re-cabled — LACP won't aggregate legs facing different nodes. +- **Nodes in maintenance mode** — every node with `provisioned = false` must be + in Talos maintenance at its `maint_ip`. Rescue → dd runbook below. +- **NetBird setup keys** — minted by the netbird stack; paths in `tf/.env.prod`. +- **Operator/CI NetBird networks selected** — the apply host reaches the CPs via + the `yucca-fsn-father-kube-cp` NetBird network (and the switches/workers via + `htz-fsn1-mgmt` / `htz-fsn1-kube`). With client ≥0.75 lazy network selection, + `netbird networks select ` or routes silently don't install. +- **`trusted_cidrs`** — MUST include the source the TF runner dials the CPs from + (the NetBird range), or bootstrap (apid 50000) hangs. > CI owns `tf:apply` (`.github/workflows/infra.yml`). Locally use `tf:plan` only. -## Phase-4: worker provisioning runbook (rescue → Talos maintenance) +## Node provisioning runbook (rescue → Talos maintenance) -The workers (Robot server numbers 3008210/11/12) must boot Talos in maintenance -mode at their fabric IP before this stack applies. Per worker: +Per node (CPs: Robot 3027819/3027863/3028524; workers: 3008210/11/12): -1. Robot → enable the **rescue system** (linux64) for the server, reboot into it. -2. `dd` the Talos **metal** image for the cluster schematic onto the boot disk: +1. Robot → enable the **rescue system** (linux64), reboot into it. +2. `dd` the Talos **metal** image for the cluster schematic onto the install disk + (`tofu output talos_metal_image_url`): ```sh - wget -O /tmp/talos.raw.xz \ - "https://factory.talos.dev/image//v1.13.4/metal-amd64.raw.xz" - xz -dc /tmp/talos.raw.xz | dd of=/dev/sda bs=4M && sync + wget -O /tmp/talos.raw.xz "$(tofu output -raw talos_metal_image_url)" + xz -dc /tmp/talos.raw.xz | dd of=/dev/sda bs=4M conv=fsync && sync ``` -3. Reboot off the rescue system → Talos comes up in maintenance mode. -4. Bring up `bond0` over the two 25G NICs with the tagged **kube VLAN 10** carrying - the node's `fabric_ip` (matching `clusters.auto.tfvars`), so the TF runner can - reach apid (50000) over the fabric. (This stack then pins the same config.) + Record the disk's SERIAL (`lsblk -d -o NAME,SERIAL`) — it pins + `install_serial` in `clusters.auto.tfvars`. +3. Reboot off the rescue system → Talos comes up in maintenance mode on the + public NIC (DHCP) = the node's `maint_ip`. (If it lands back in rescue, the + rescue flag didn't clear — just reboot again.) +4. Set the node's `provisioned = false` in tfvars → apply → flip to `true` once + it has joined. -This mirrors the mgmt-host reprovision pattern (`../mgmt-hosts.yaml` + the fabric -stack's `mgmt.tf`); a future iteration can drive it from TF/Ansible. +## Cutover runbook (hybrid → all-bare-metal REBUILD) — EXECUTED 2026-07-15 + +The rebuild keeps `talos_machine_secrets` (cluster PKI) but re-bootstraps etcd on +the new CPs and re-installs the workers. **In-cluster state (Mayastor/localpv) is +lost**; Flux redeploys everything. The steps below were executed 2026-07-15 (all +stacks now plan clean); kept as the reference for any future rebuild. Order +matters: + +1. **Fabric first** (CI orders fabric before talos): apply lands the kube-cp + VLAN/IRB + ae4-6. Port-0 10g channelization is already live on the spine + (set 2026-07-15, identical to the TF config) and the leg pairing is verified — + after the apply, `show lacp interfaces` should show ae4-6 collecting once the + CPs boot their bonds. +2. **New CPs in maintenance mode** (done 2026-07-15): rescue → dd → maintenance + at 178.63.124.20/.21/.22. +3. **State surgery** (forgets, no destroys — safe with prevent_destroy): + ```sh + tofu state rm talos_machine_bootstrap.this # re-bootstrap on the new cp-1 + tofu state rm talos_cluster_kubeconfig.this + tofu state rm helm_release.cilium helm_release.flux_operator helm_release.flux_instance + tofu state rm kubernetes_secret_v1.github_app kubernetes_namespace_v1.cert_manager kubernetes_secret_v1.cloudflare_api_token + ``` +4. **Reset the workers** to maintenance mode (wipes them — deliberate): + ```sh + talosctl -n 10.40.10.11 reset --graceful=false --reboot \ + --system-labels-to-wipe STATE --system-labels-to-wipe EPHEMERAL # × each worker + ``` + They come back in maintenance at their `maint_ip` (public DHCP). +5. **Merge/apply this stack**: destroys the hcloud CP VMs + LB + network + + firewall (their prevent_destroy left with the deleted config), applies CP + configs → bootstrap → workers → Cilium → Flux. +6. Flip every node's `provisioned = true` once joined; rotate operator + kubeconfigs (`op read`, secrets.tf rewrote them). + +Rebuild gotchas hit on 2026-07-15 (expect them again): +- **The first apply fails partway** — helm dials the apiserver seconds after + bootstrap (connection refused) and the kubernetes/1P resources throw + "inconsistent final plan" (provider config unknowable at plan). Just re-apply; + nothing is damaged. A helm wait-timeout can strand a `failed` release + ("cannot re-use a name that is still in use") — `helm uninstall` it first. +- **flux-operator waits on the first worker** (CPs are unschedulable) — worker + install+join takes longer than helm's 5m wait. Re-apply once workers are Ready. +- **CoreDNS chicken-and-egg**: `cluster.coreDNS.disabled=true` from t=0 means NO + cluster DNS until Flux deploys ours — but flux-operator needs DNS to fetch its + manifests from ghcr.io. Break the cycle once per rebuild: + `kubectl apply -f kubernetes/apps/prod/htz-fsn1/coredns.yaml` (the exact + objects Flux owns — it adopts them unchanged). +- ~~CRD deadlock (flat kustomization)~~ — fixed structurally after the rebuild: + the tree is layered (`cluster-infra` = operators/CRD providers with + `wait: true`; `cluster-apps` dependsOn it — see clusters/prod/htz-fsn1/ + apps.yaml). A fresh cluster converges without manual CRD pre-installs; the + only remaining hand-step is the CoreDNS one above (it predates Flux itself). +- **Stale NetBird peers**: re-provisioned nodes join as NEW peers; the old + same-named peers linger disconnected and break the netbird stack's + `data.netbird_peer` lookups ("cannot match multiple peers"). Delete the + disconnected duplicates (API/console) before applying the netbird stack. +- **`talosctl reset --wait=false`** — the default wait can never complete (the + node comes back at a different IP, in maintenance mode). +- ~~Hand-created netops secrets die with the cluster~~ — fixed: the netops + namespace + netops-ssh/grafana-admin/hyperglass-devices Secrets are TF-owned + now (netops-secrets.tf, sourced from 1P), restored by the normal apply. ## Notes -- **Bootstrap is one-shot.** Re-applying does not re-bootstrap; tainting - `talos_machine_bootstrap.this` re-rolls cluster identity — don't. -- CP VMs have `ignore_changes = [user_data, image]` so re-applies don't recycle - live nodes; change them deliberately (cordon/drain first). -- Flux activates later from `kubernetes/clusters/prod/htz-fsn1` (already scaffolded). +- **Bootstrap is one-shot.** Re-applying does not re-bootstrap; replacing + `talos_machine_bootstrap.this` re-rolls cluster identity — don't (the state-rm + in the cutover is the deliberate exception). +- The API VIP is etcd-elected: it exists only while a healthy CP holds it. The + bootstrap/operator path deliberately dials cp-1's IP, not the VIP. +- The old hcloud snapshot build task (`hetzner:talos-image`) is retired — all + nodes boot the factory **metal** image via the rescue-dd runbook. diff --git a/tf/deployment/prod/htz-fsn1/talos/addressing.tf b/tf/deployment/prod/htz-fsn1/talos/addressing.tf index d3590bfc..0110ca0b 100644 --- a/tf/deployment/prod/htz-fsn1/talos/addressing.tf +++ b/tf/deployment/prod/htz-fsn1/talos/addressing.tf @@ -1,16 +1,23 @@ # Site IP plan — the single source of truth (same module the fabric + netbird # stacks read). Gives us, derived from site_id (40): # addr_site.kube_cidr 10.40.10.0/24 — fabric VLAN 10, worker east-west (50G) -# addr_site.kube_cp_cidr 10.40.11.0/24 — isolated hcloud subnet, CP etcd + API LB +# addr_site.kube_cp_cidr 10.40.11.0/24 — fabric VLAN 11, CP etcd + API VIP # Nothing is hardcoded here; addresses below are cidrhost() offsets into these. module "addr_site" { source = "../../../../shared/modules/fabric-addressing" site_id = var.site_id } +# Cluster-1 view — only for the leaf vme address (netops-secrets.tf hyperglass). +module "addr_cls1" { + source = "../../../../shared/modules/fabric-addressing" + site_id = var.site_id + cluster_id = 1 +} + locals { - kube_cidr = module.addr_site.kube_cidr # 10.40.10.0/24 (fabric) - kube_gateway = cidrhost(module.addr_site.kube_cidr, 1) # .1 IRB on the cluster leaf - kube_cp_cidr = module.addr_site.kube_cp_cidr # 10.40.11.0/24 (hcloud) - kube_cp_gw = module.addr_site.kube_cp_gateway # .1 Hetzner Cloud Gateway + kube_cidr = module.addr_site.kube_cidr # 10.40.10.0/24 (fabric VLAN 10) + kube_gateway = cidrhost(module.addr_site.kube_cidr, 1) # .1 IRB on the spine + kube_cp_cidr = module.addr_site.kube_cp_cidr # 10.40.11.0/24 (fabric VLAN 11) + kube_cp_gw = module.addr_site.kube_cp_gateway # .1 IRB on the spine } diff --git a/tf/deployment/prod/htz-fsn1/talos/cilium-values.yaml.tftpl b/tf/deployment/prod/htz-fsn1/talos/cilium-values.yaml.tftpl index 8a252166..d535ffd9 100644 --- a/tf/deployment/prod/htz-fsn1/talos/cilium-values.yaml.tftpl +++ b/tf/deployment/prod/htz-fsn1/talos/cilium-values.yaml.tftpl @@ -1,9 +1,9 @@ -# Cilium for the hybrid prod cluster. Talos sets cni:none + proxy:disabled, so -# nodes stay NotReady until this lands the datapath. +# Cilium for the prod cluster. Talos sets cni:none + proxy:disabled, so nodes +# stay NotReady until this lands the datapath. # # TUNNEL (geneve) routing — see the routingMode block below for the full why: -# the cluster spans two L2 domains (CPs on kube-cp hcloud, workers on the fabric -# VLAN) bridged only by NetBird, which native routing/autoDirectNodeRoutes can't +# the cluster spans two L2 domains (CPs on the kube-cp VLAN, workers on the kube +# VLAN) routed by the spine IRBs, which autoDirectNodeRoutes (same-L2-only) can't # span. Worker↔worker east-west still rides the 50G fabric, just encapsulated. # # securityContext + cgroup blocks are MANDATORY on Talos. Ref: Talos "Deploying Cilium". @@ -12,17 +12,17 @@ ipam: mode: kubernetes # Tunnel (geneve) routing. The cluster spans two L2 domains — CPs on the kube-cp -# hcloud net, workers on the fabric VLAN — bridged only by NetBird. Native routing -# can't span that; a geneve overlay carries pod↔pod over whatever host-to-host path -# exists (the 50G fabric for worker↔worker, NetBird for CP↔worker). East-west still -# rides the fabric, just encapsulated (~50B overhead, negligible at 50G). +# VLAN (11), workers on the kube VLAN (10) — routed by the spine's IRBs. +# autoDirectNodeRoutes only works within one L2; a geneve overlay carries pod↔pod +# over the routed node-to-node path instead. East-west still rides the fabric, +# just encapsulated (~50B overhead, negligible at 50G). routingMode: tunnel tunnelProtocol: geneve # Masquerade pod traffic to non-pod destinations (internet, node IPs); pod↔pod is -# tunnelled so it isn't masqueraded. ip-masq-agent excludes only the pod CIDR, so the -# geneve underlay + node-IP traffic egressing wt0 is SNAT'd to the node's NetBird -# address (which WireGuard accepts). Requires BPF masquerade on. +# tunnelled so it isn't masqueraded. ip-masq-agent excludes only the pod CIDR, so +# node-IP traffic egressing wt0 (NetBird, the operator plane) is still SNAT'd to +# the node's mesh address (which WireGuard accepts). Requires BPF masquerade on. enableIPv4Masquerade: true bpf: masquerade: true diff --git a/tf/deployment/prod/htz-fsn1/talos/cilium.tf b/tf/deployment/prod/htz-fsn1/talos/cilium.tf index 7e38ab19..aa73d795 100644 --- a/tf/deployment/prod/htz-fsn1/talos/cilium.tf +++ b/tf/deployment/prod/htz-fsn1/talos/cilium.tf @@ -1,7 +1,8 @@ # Cilium CNI — installed post-bootstrap in the same apply (helm provider bound to # the bootstrap CP, providers.tf). Talos set cni:none + proxy:disabled, so nodes -# go Ready only once this lands the datapath. Native routing + autoDirectNodeRoutes -# keeps worker east-west on the 50G fabric (cilium-values.yaml.tftpl). +# go Ready only once this lands the datapath. Geneve tunnel routing spans the two +# routed fabric VLANs (kube/kube-cp); worker east-west still rides the 50G fabric +# (cilium-values.yaml.tftpl). resource "helm_release" "cilium" { name = "cilium" namespace = "kube-system" @@ -27,9 +28,9 @@ data "talos_cluster_health" "post_cni" { count = var.bootstrap_health_gate ? 1 : 0 client_configuration = talos_machine_secrets.this.client_configuration - control_plane_nodes = local.cp_private_ips + control_plane_nodes = local.cp_ips worker_nodes = [for w in var.cluster.workers : w.fabric_ip] - endpoints = local.cp_private_ips + endpoints = local.cp_ips timeouts = { read = "10m" } diff --git a/tf/deployment/prod/htz-fsn1/talos/clusters.auto.tfvars b/tf/deployment/prod/htz-fsn1/talos/clusters.auto.tfvars index 4a84a6d8..11325f5d 100644 --- a/tf/deployment/prod/htz-fsn1/talos/clusters.auto.tfvars +++ b/tf/deployment/prod/htz-fsn1/talos/clusters.auto.tfvars @@ -1,11 +1,12 @@ -# ── prod hybrid Talos cluster: `father` ────────────────────────────────────── -# Topology (see README.md): 3 Hetzner Cloud CP VMs + 3 Hetzner Robot bare-metal -# workers. CP↔worker rides the NetBird mesh; worker↔worker east-west rides the -# 50G fabric (kube VLAN 10) via Cilium BGP; etcd CP↔CP on a private hcloud subnet; -# API via a public Hetzner Cloud LB. +# ── prod bare-metal Talos cluster: `father` ────────────────────────────────── +# Topology (see README.md): 3 Hetzner Robot bare-metal CPs on the kube-cp fabric +# VLAN 11 (etcd + API VIP 10.40.11.5) + 3 Hetzner Robot bare-metal workers on the +# kube fabric VLAN 10. The spine routes kube↔kube-cp between its IRBs; NetBird +# stays on every node as the operator/backup plane (kube-cp is routed to the mesh +# via the CPs). No Hetzner Cloud anywhere — the cloud CP VMs + API LB are retired. # -# Adding/replacing a node = edit here + `tf:plan` (CI applies). Workers must -# already be in Talos maintenance mode at their fabric_ip (see Phase-4 runbook). +# Adding/replacing a node = edit here + `tf:plan` (CI applies). Nodes must +# already be in Talos maintenance mode at their maint_ip (see the runbook). cluster = { name = "father" # prod K8s cluster (Star Wars; staging = luke) @@ -14,7 +15,6 @@ cluster = { # The node extension set lives in schematic.yaml (managed via # talos_image_factory_schematic in image.tf) — no schematic id to paste here. - install_disk = "/dev/sda" # CP install target (hcloud VMs: virtio /dev/sda) — the ONLY consumer is cp_install_patch; workers install by NVMe serial (workers.tf) cilium_version = "1.19.5" hubble = true @@ -22,23 +22,23 @@ cluster = { # NetBird peer address range for this deployment (firewall trust for the mesh). netbird_node_cidr = "10.254.0.0/15" - # ── Cloud control plane (Hetzner Cloud, fsn1) ────────────────────────────── - cp_count = 3 - # PINNED to the live nodes (verified against discovery/cp_nodes 2026-07-07). - # Order follows cp_ip_offset: kaycee=.11, bettie=.12, ofelia=.13. A wrong name - # here RENAMES a live control plane — check twice. - cp_names = ["kaycee", "bettie", "ofelia"] - cp_server_type = "ccx23" # 4 vCPU / 16 GB, dedicated x86 - cp_location = "fsn1" - cp_ip_offset = 11 # CP private IPs → 10.40.11.11 / .12 / .13 (etcd) - lb_type = "lb11" - lb_ip_offset = 5 # API LB private IP → 10.40.11.5 - lb_public = false # private-only LB; the API endpoint stays on the kube-cp net + # ── Bare-metal control planes (Hetzner Robot; kube-cp VLAN 11, gw .1 = spine) ── + # `name` keys the apply resources (stable across list edits). VIP 10.40.11.5 + # (= the retired hcloud LB IP, so api_dns_name carried over unchanged). + # provisioned=false → the one-time install apply dials maint_ip (maintenance + # mode); flip true per node as it comes up. + cps = [ + { name = "harlan", cp_ip = "10.40.11.11", maint_ip = "178.63.124.20", robot_id = 3027819, install_serial = "17451A00D9F8" }, + { name = "imelda", cp_ip = "10.40.11.12", maint_ip = "178.63.124.21", robot_id = 3027863, install_serial = "1708162471F6" }, + { name = "roscoe", cp_ip = "10.40.11.13", maint_ip = "178.63.124.22", robot_id = 3028524, install_serial = "18201C72C94D" }, + ] + # 2×10G Intel 82599ES SFP+ (ixgbe) enslaved into bond0 (tagged kube-cp VLAN 11, + # spine port-0 breakout ae4-6). The onboard 1G (e1000e) stays the DHCP + # public/egress NIC (default route + NetBird endpoint). + cp_bond_driver = "ixgbe" + vip_offset = 5 # API VIP → 10.40.11.5 - # ── Bare-metal workers (Hetzner Robot dedicated; sequential after mgmt-1/2) ── - # `name` keys the apply resources (stable across list edits — removing or - # reordering an entry no longer touches the others). Names verified against - # the live nodes 2026-07-07. + # ── Bare-metal workers (Hetzner Robot; kube VLAN 10) ───────────────────────── workers = [ { name = "jeanne", fabric_ip = "10.40.10.11", maint_ip = "178.63.124.38", robot_id = 3008210, install_serial = "S64GNNFX503099" }, { name = "sheron", fabric_ip = "10.40.10.12", maint_ip = "178.63.124.37", robot_id = 3008211, install_serial = "S64GNJ0WC25870" }, @@ -52,8 +52,8 @@ cluster = { } # Operator/CI sources allowed on the Talos host firewall (apid 50000 + apiserver -# 6443), on top of the node planes. NetBird peer range ONLY — no public IPs (the -# hcloud firewall also blocks public apiserver/apid; see hcloud-firewall.tf). A -# re-bootstrap dials apid on a CP public IP, so temporarily re-add the operator's -# /32 here (and open 50000 on the hcloud firewall) for that one step. +# 6443), on top of the node planes. NetBird peer range ONLY — no public IPs. +# Operators reach the CPs over the NetBird kube-cp route (the CPs are the route +# peers); a re-bootstrap that must dial apid before the mesh is up goes through +# a maint_ip (maintenance mode is unauthenticated — no firewall yet). trusted_cidrs = ["10.254.0.0/15"] diff --git a/tf/deployment/prod/htz-fsn1/talos/controlplane.tf b/tf/deployment/prod/htz-fsn1/talos/controlplane.tf index 749cbb02..dad88310 100644 --- a/tf/deployment/prod/htz-fsn1/talos/controlplane.tf +++ b/tf/deployment/prod/htz-fsn1/talos/controlplane.tf @@ -1,117 +1,99 @@ -# Spread placement group — forces the 3 CPs onto DISTINCT physical hosts, so no -# single host failure can take out >1 etcd member / break quorum. (hcloud caps a -# spread group at 10 servers; 3 is fine.) -resource "hcloud_placement_group" "control_plane" { - name = "yucca-${var.region_code}-${var.cluster.name}-cp" - type = "spread" - labels = { cluster = var.cluster.name, role = "control-plane" } +# ── Bare-metal control planes ──────────────────────────────────────────────── +# Applied over apid to nodes already in Talos maintenance mode at their maint_ip +# (Hetzner public DHCP on the onboard 1G NIC). After the install+reboot they hold +# their kube-cp VLAN address and join the NetBird mesh. +# +# bond0 (2×10G LACP, ixgbe) → vlan 11 (kube-cp) = cp_ip — etcd + apiserver + VIP +# route to the kube VLAN via the kube-cp IRB (10.40.11.1) — apiserver→kubelet +# default route via the onboard 1G public NIC (DHCP) — egress + NetBird endpoint +# +# The API VIP (10.40.11.5) is Talos-managed on the VLAN: etcd elects one holder, +# so it's only up while the cluster is healthy — exactly what the api_dns_name +# record points at. +# +# CPs are PROVISIONED to maintenance mode out of band — see the runbook +# (./README.md): Hetzner rescue → dd the Talos metal image → reboot. This stack +# assumes they're already there. + +locals { + # Keyed by hostname (cp_node_map) — same stable key as the apply resource. + cp_node_patches = { for hostname, n in local.cp_node_map : hostname => [ + # Install disk by SERIAL (never by name — enumeration swaps across boots). + yamlencode({ + machine = { install = { + diskSelector = { serial = n.install_serial } + image = local.install_image + } } + }), + yamlencode({ + machine = { + network = { + interfaces = [{ + interface = "bond0" + dhcp = false + # Bond members selected by NIC driver (cp_bond_driver, ixgbe) — exactly + # the two 10G SFP+ ports; the onboard 1G public NIC is e1000e. + bond = { + mode = "802.3ad" + lacpRate = "fast" + xmitHashPolicy = "layer3+4" + miimon = 100 + deviceSelectors = local.c.cp_bond_driver != null ? [{ + driver = local.c.cp_bond_driver + }] : null + interfaces = local.c.cp_bond_driver != null ? null : local.c.cp_bond_interfaces + } + vlans = [{ + vlanId = module.addr_site.kube_cp_vlan_id # 11 + addresses = ["${n.cp_ip}/${local.kube_cp_prefix}"] + # The workers live one IRB away — pin the return route so + # apiserver→kubelet + geneve ride the fabric, not the mesh. + routes = [{ network = local.kube_cidr, gateway = local.kube_cp_gw }] + vip = { ip = local.api_vip } + }] + }] + } + } + }), + yamlencode({ apiVersion = "v1alpha1", kind = "HostnameConfig", auto = "off", hostname = hostname }), + ] } } -# ── Cloud control-plane VMs ────────────────────────────────────────────────── -# 3× CCX23 booted from the Talos snapshot, configured via user_data (the per-CP -# machine config from talos.tf). Public IPv4 = NetBird NAT traversal + the TF -# runner's bootstrap path; private IP (eth1, kube-cp subnet) = etcd. Talos ignores -# SSH, so no ssh_keys. ignore_changes keeps re-applies from recycling live nodes. -resource "hcloud_server" "control_plane" { - count = var.cluster.cp_count - name = local.cp_hostnames[count.index] - image = data.hcloud_image.talos.id - server_type = var.cluster.cp_server_type - location = var.cluster.cp_location - placement_group_id = hcloud_placement_group.control_plane.id - - user_data = data.talos_machine_configuration.cp[count.index].machine_configuration - - public_net { - ipv4_enabled = true - ipv6_enabled = false - } - - network { - network_id = hcloud_network.kube_cp.id - ip = local.cp_private_ips[count.index] - } - - labels = { cluster = var.cluster.name, role = "control-plane" } - - depends_on = [hcloud_network_subnet.kube_cp] - - lifecycle { - ignore_changes = [user_data, image] - # etcd members — replacing one rolls quorum; destroying all rolls the - # cluster. Any legitimate replace (scale-down, location change) must - # temporarily lift this flag, deliberately. - prevent_destroy = true - } -} - -# Live CP config sync — user_data only configures a CP at CREATION (and is -# ignore_changes above), so config edits never reached running CPs; today they were -# hand-patched via talosctl. This applies the current rendered config to each live CP -# on every apply (mode auto: no reboot for the config we manage). Notably it keeps the -# worker /etc/hosts entries fresh: a re-provisioned worker gets a new NetBird IP, and -# without this the apiserver keeps dialing the dead one. +# Keyed by HOSTNAME, not list position: removing or reordering a CP in tfvars +# must never shift another node's resource address (a shift = replace = an etcd +# member reset). Matches the workers.tf pattern. resource "talos_machine_configuration_apply" "cp" { - count = var.cluster.cp_count + for_each = local.cp_node_map client_configuration = talos_machine_secrets.this.client_configuration - machine_configuration_input = data.talos_machine_configuration.cp[count.index].machine_configuration - node = local.cp_private_ips[count.index] - endpoint = local.cp_private_ips[count.index] + machine_configuration_input = data.talos_machine_configuration.cp.machine_configuration + # The FIRST apply targets the maintenance-mode node at its Hetzner public IP + # (provisioned=false); the config brings up bond0.11 at cp_ip + joins NetBird and + # the node reboots into the cluster. Every later apply targets the LIVE node at + # its kube-cp IP (over the NetBird kube-cp route). Flip provisioned in tfvars per + # CP as it comes up. + node = each.value.provisioned ? each.value.cp_ip : each.value.maint_ip + endpoint = each.value.provisioned ? each.value.cp_ip : each.value.maint_ip + config_patches = local.cp_node_patches[each.key] + apply_mode = "auto" - depends_on = [hcloud_server.control_plane] + # reset=false: decommissioning an etcd member must be a deliberate + # `talosctl reset` (after `etcd leave`), never a terraform destroy side effect. + # NB: on_destroy is read from STATE, so this protects only after it has been + # applied once. + on_destroy = { + reboot = true + reset = false + graceful = false + } lifecycle { # cp_netbird_patch is silently OMITTED when the setup key is "" (the # credential-less validate default) — an env-less apply would strip NetBird - # from the live CP configs. Fail loudly instead. + # from the live CP configs, cutting the operators' kube-cp route. Fail loudly. precondition { condition = length(var.netbird_talos_cp_setup_key) > 0 error_message = "netbird_talos_cp_setup_key is empty — run applies through tf/op-run.sh (op run env missing or op:// ref resolved empty)." } } } - -# ── API load balancer ──────────────────────────────────────────────────────── -# Fronts the 3 CPs on 6443. Private IP (kube-cp) is the in-cluster target; the -# public frontend (lb_public) is what operators + workers dial via api_dns_name. -# certSANs (talos.tf) already include api_dns_name + the private LB IP. -resource "hcloud_load_balancer" "kube_api" { - name = "yucca-${var.region_code}-${var.cluster.name}-kube-api" - load_balancer_type = var.cluster.lb_type - location = var.cluster.cp_location - labels = { cluster = var.cluster.name } -} - -resource "hcloud_load_balancer_network" "kube_api" { - load_balancer_id = hcloud_load_balancer.kube_api.id - network_id = hcloud_network.kube_cp.id - ip = local.lb_private_ip - enable_public_interface = var.cluster.lb_public - depends_on = [hcloud_network_subnet.kube_cp] -} - -resource "hcloud_load_balancer_service" "kube_api" { - load_balancer_id = hcloud_load_balancer.kube_api.id - protocol = "tcp" - listen_port = 6443 - destination_port = 6443 - - health_check { - protocol = "tcp" - port = 6443 - interval = 10 - timeout = 5 - retries = 3 - } -} - -# Target the CPs over their PRIVATE IPs (LB is attached to the same network). -resource "hcloud_load_balancer_target" "kube_api" { - count = var.cluster.cp_count - type = "server" - load_balancer_id = hcloud_load_balancer.kube_api.id - server_id = hcloud_server.control_plane[count.index].id - use_private_ip = true - depends_on = [hcloud_load_balancer_network.kube_api] -} diff --git a/tf/deployment/prod/htz-fsn1/talos/discovery.tf b/tf/deployment/prod/htz-fsn1/talos/discovery.tf index 58031bc6..9763c7f2 100644 --- a/tf/deployment/prod/htz-fsn1/talos/discovery.tf +++ b/tf/deployment/prod/htz-fsn1/talos/discovery.tf @@ -35,13 +35,14 @@ output "discovery" { } kubernetes = { cluster_name = var.cluster.name - api_endpoint = local.cluster_endpoint # https://:6443 (LB) + api_endpoint = local.cluster_endpoint # https://:6443 (VIP) + api_vip = local.api_vip # Talos-elected VIP on kube-cp (the api_dns_name A record) operator_endpoint = local.operator_endpoint # direct bootstrap-CP apiserver - cp_node_ips = local.cp_private_ips # kube-cp IPs; operators/yuctl reach via NetBird + cp_node_ips = local.cp_ips # kube-cp IPs; operators/yuctl reach via NetBird worker_node_ips = [for w in var.cluster.workers : w.fabric_ip] # Node NAME → IP maps (short wordlist names). Consumed by the netbird stack # (yucca.futo.network records) instead of hardcoding names in two stacks. - cp_nodes = { for i, n in var.cluster.cp_names : n => local.cp_private_ips[i] } + cp_nodes = { for n in var.cluster.cps : n.name => n.cp_ip } worker_nodes = { for w in var.cluster.workers : w.name => w.fabric_ip } kubeconfig_ref = "op://${local._disc_vault}/${local._kubeconfig_title}/password" talosconfig_ref = "op://${local._disc_vault}/${local._talosconfig_title}/password" diff --git a/tf/deployment/prod/htz-fsn1/talos/firewall.tf b/tf/deployment/prod/htz-fsn1/talos/firewall.tf index 72f24f97..368c41ba 100644 --- a/tf/deployment/prod/htz-fsn1/talos/firewall.tf +++ b/tf/deployment/prod/htz-fsn1/talos/firewall.tf @@ -1,15 +1,16 @@ # Talos host ingress firewall (default-deny + per-service allow-lists). Governs # HOST-network ports only; pod/ClusterIP traffic rides Cilium. # -# Trust planes for this hybrid cluster: +# Trust planes: # kube_cidr 10.40.10.0/24 workers' fabric IPs (east-west, BGP) -# kube_cp_cidr 10.40.11.0/24 CP private IPs + the API LB (etcd, LB health-checks) -# netbird_node_cidr 10.254.0.0/15 the NetBird mesh (apiserver↔kubelet, node control) +# kube_cp_cidr 10.40.11.0/24 CP IPs + the API VIP (etcd, apiserver) +# netbird_node_cidr 10.254.0.0/15 the NetBird mesh (operators, backup plane) # trusted_cidrs operator/CI source ranges # -# ⚠️ The TF runner dials the CP PUBLIC IPs for bootstrap (apid 50000) and the -# helm/kubernetes providers (apiserver 6443). Its source IP MUST be in -# trusted_cidrs (e.g. the CI runner's egress / NetBird range) or those steps hang. +# ⚠️ The TF runner dials the CP kube-cp IPs (over the NetBird kube-cp route) for +# bootstrap (apid 50000) and the helm/kubernetes providers (apiserver 6443). Its +# source IP MUST be in trusted_cidrs (e.g. the CI runner's NetBird range) or +# those steps hang. locals { firewall_allow = concat([local.kube_cidr, local.kube_cp_cidr, local.c.netbird_node_cidr], var.trusted_cidrs) operator_allow = local.firewall_allow @@ -41,7 +42,7 @@ locals { }), # Cilium geneve overlay (tunnel routing): pod↔pod is encapsulated node-to-node # (UDP 6081). Required across BOTH L2 domains — worker↔worker over the fabric and - # CP↔worker over the mesh — or pod-to-pod traffic is silently dropped. + # CP↔worker routed via the spine IRBs — or pod-to-pod traffic is silently dropped. yamlencode({ apiVersion = "v1alpha1" kind = "NetworkRuleConfig" diff --git a/tf/deployment/prod/htz-fsn1/talos/flux.tf b/tf/deployment/prod/htz-fsn1/talos/flux.tf index 6996b55b..0d774857 100644 --- a/tf/deployment/prod/htz-fsn1/talos/flux.tf +++ b/tf/deployment/prod/htz-fsn1/talos/flux.tf @@ -2,9 +2,7 @@ # Helm (OCI charts), then Flux reconciles this repo's kubernetes/clusters/prod # path on its own. Lands after the cluster + CNI (providers.tf helm/kubernetes). # -# ⚠ TEMPORARY: the sync ref is feat/prod (var.flux_git_ref) so father can be -# GitOps-managed before the branch merges. Flip flux_git_ref to "main" (the -# default once this merges) — nothing else changes. +# Sync ref = main (var.flux_git_ref default; CI guards it stays that way). resource "helm_release" "flux_operator" { name = "flux-operator" diff --git a/tf/deployment/prod/htz-fsn1/talos/hcloud-firewall.tf b/tf/deployment/prod/htz-fsn1/talos/hcloud-firewall.tf deleted file mode 100644 index 5c913ed7..00000000 --- a/tf/deployment/prod/htz-fsn1/talos/hcloud-firewall.tf +++ /dev/null @@ -1,29 +0,0 @@ -# hcloud firewall for the control-plane VMs — enforces "no public access to the -# cluster" at the cloud edge. Only NetBird's WireGuard is allowed inbound from the -# internet; apiserver (6443) + apid (50000) + everything else is dropped. The -# cluster is reached ONLY over NetBird (WireGuard tunnel, arrives on 51820/udp) or -# the private kube-cp network — neither of which this filters (hcloud firewalls -# apply to the public interface; private-net + in-tunnel traffic is untouched). -# -# Egress is unrestricted (hcloud default) — image pulls, NetBird signal/relay, etc. -# -# NB: a future re-bootstrap dials apid (50000) on a CP public IP — temporarily add -# an operator-source rule for 50000, or bootstrap from a NetBird-reachable path. -resource "hcloud_firewall" "control_plane" { - name = "yucca-${var.region_code}-${var.cluster.name}-cp" - labels = { cluster = var.cluster.name, role = "control-plane" } - - rule { - direction = "in" - protocol = "udp" - port = "51820" - source_ips = ["0.0.0.0/0", "::/0"] - description = "NetBird WireGuard (P2P)" - } -} - -# Attach to the CPs without recreating them. -resource "hcloud_firewall_attachment" "control_plane" { - firewall_id = hcloud_firewall.control_plane.id - server_ids = hcloud_server.control_plane[*].id -} diff --git a/tf/deployment/prod/htz-fsn1/talos/image.tf b/tf/deployment/prod/htz-fsn1/talos/image.tf index 6d3ee37f..1e88327f 100644 --- a/tf/deployment/prod/htz-fsn1/talos/image.tf +++ b/tf/deployment/prod/htz-fsn1/talos/image.tf @@ -1,33 +1,12 @@ # Talos Image Factory schematic — the extension set, managed in TF. The resource -# registers schematic.yaml with the factory and returns its deterministic id; we -# derive the CP (hcloud-amd64) + worker (metal-amd64) image URLs from it. No -# hand-pasted schematic id, no out-of-band curl. +# registers schematic.yaml with the factory and returns its deterministic id; the +# metal installer/image URLs derive from it. No hand-pasted schematic id, no +# out-of-band curl. One schematic for every node (all bare-metal). resource "talos_image_factory_schematic" "this" { schematic = file("${path.module}/schematic.yaml") } -# Worker (bare-metal) schematic — same set minus qemu-guest-agent (see the file). -resource "talos_image_factory_schematic" "worker" { - schematic = file("${path.module}/schematic-worker.yaml") -} - locals { - talos_schematic_id = talos_image_factory_schematic.this.id - talos_worker_schematic_id = talos_image_factory_schematic.worker.id - talos_hcloud_image_url = "https://factory.talos.dev/image/${local.talos_schematic_id}/v${var.cluster.talos_version}/hcloud-amd64.raw.xz" - talos_metal_image_url = "https://factory.talos.dev/image/${local.talos_worker_schematic_id}/v${var.cluster.talos_version}/metal-amd64.raw.xz" -} - -# Talos amd64 image as an hcloud snapshot. hcloud can't boot the Talos ISO, so the -# snapshot is built ONCE, out of band, by: -# -# mise run hetzner:talos-image -# -# (it reads talos_hcloud_image_url from this stack's output, spins a temporary -# rescue server, dd's the image, snapshots, tears down). This data source then -# resolves it by label — rebuild only on a Talos version bump. -data "hcloud_image" "talos" { - with_selector = "os=talos,cluster=${var.cluster.name},arch=amd64" - with_architecture = "x86" - most_recent = true + talos_schematic_id = talos_image_factory_schematic.this.id + talos_metal_image_url = "https://factory.talos.dev/image/${local.talos_schematic_id}/v${var.cluster.talos_version}/metal-amd64.raw.xz" } diff --git a/tf/deployment/prod/htz-fsn1/talos/migrations.tf b/tf/deployment/prod/htz-fsn1/talos/migrations.tf deleted file mode 100644 index 2cbdaaae..00000000 --- a/tf/deployment/prod/htz-fsn1/talos/migrations.tf +++ /dev/null @@ -1,33 +0,0 @@ -# ── One-shot state migration: count → hostname-keyed workers ───────────────── -# Moves the existing worker apply instances to their stable hostname keys and -# forgets the retired node-names shuffle module WITHOUT destroying anything. -# The gate before merging: a local `mise tf:plan` must show exactly these three -# moves + one "removed from state" + the on_destroy in-place updates — zero -# destroys, zero replaces, zero diffs on the CP resources. -# DELETE this file once CI has applied it (the blocks are inert afterwards, but -# the `removed` block conflicts if module "names" is ever reintroduced). - -moved { - from = talos_machine_configuration_apply.worker[0] - to = talos_machine_configuration_apply.worker["yucca-htz-fsn-father-k8s-jeanne"] -} - -moved { - from = talos_machine_configuration_apply.worker[1] - to = talos_machine_configuration_apply.worker["yucca-htz-fsn-father-k8s-sheron"] -} - -moved { - from = talos_machine_configuration_apply.worker[2] - to = talos_machine_configuration_apply.worker["yucca-htz-fsn-father-k8s-dianna"] -} - -# Forget (don't destroy) the retired names module — its only resource is a -# random_shuffle; destroying would be inert, but "forget" keeps the plan clean. -removed { - from = module.names - - lifecycle { - destroy = false - } -} diff --git a/tf/deployment/prod/htz-fsn1/talos/netops-secrets.tf b/tf/deployment/prod/htz-fsn1/talos/netops-secrets.tf new file mode 100644 index 00000000..ee25b4f6 --- /dev/null +++ b/tf/deployment/prod/htz-fsn1/talos/netops-secrets.tf @@ -0,0 +1,91 @@ +# ─── netops namespace + fabric-credential Secrets ──────────────────────────── +# The netops stack (kubernetes/apps/prod/htz-fsn1/netops/) mounts fabric +# credentials that must NEVER be in git: the read-only `netops` Junos login's +# SSH key + password (fabric stack, fabric.tf netops_users) and the Grafana +# admin password. Historically these were hand-created (`kubectl create secret`) +# and DIED WITH THE CLUSTER on the 2026-07 rebuild — now they're provisioned +# here from the same 1Password items, so a rebuild restores them with the stack. +# Flux owns the workloads around them; TF owns the namespace + these Secrets +# (the namespace also carries the VictoriaMetrics hostPath PVC, so it must +# survive flux prunes — TF ownership replaces the old prune-disabled manifest). + +data "onepassword_item" "netops_ssh_key" { + vault = data.onepassword_vault.prod.uuid + title = "NETOPS_FABRIC_SSH_PRIVATE_KEY" # DOCUMENT item; file id_ed25519 +} + +data "onepassword_item" "netops_fabric_password" { + vault = data.onepassword_vault.prod.uuid + title = "NETOPS_FABRIC_PASSWORD" +} + +data "onepassword_item" "grafana_admin" { + vault = data.onepassword_vault.prod.uuid + title = "FATHER_GRAFANA_ADMIN" +} + +resource "kubernetes_namespace_v1" "netops" { + metadata { + name = "netops" + labels = { + # VictoriaMetrics persists to a hostPath (/var/mnt); baseline forbids it. + "pod-security.kubernetes.io/enforce" = "privileged" + } + annotations = { + # Belt-and-braces from the flux-owned era (the namespace manifest is gone + # from the tree, but flux's GC honors this if it ever re-tracks the object). + "kustomize.toolkit.fluxcd.io/prune" = "disabled" + } + } +} + +# SSH key for junos-exporter (NETCONF scrape) + oxidized (config backup) — both +# mount key `id_ed25519` and log in as the `netops` Junos user. +resource "kubernetes_secret_v1" "netops_ssh" { + metadata { + name = "netops-ssh" + namespace = kubernetes_namespace_v1.netops.metadata[0].name + } + data = { + id_ed25519 = one([for f in data.onepassword_item.netops_ssh_key.file : f.content if f.name == "id_ed25519"]) + } +} + +resource "kubernetes_secret_v1" "grafana_admin" { + metadata { + name = "grafana-admin" + namespace = kubernetes_namespace_v1.netops.metadata[0].name + } + data = { + password = data.onepassword_item.grafana_admin.password + } +} + +# hyperglass device inventory — embeds the netops PASSWORD (netmiko can't +# key-auth through hyperglass config), hence a Secret and not the configmap. +resource "kubernetes_secret_v1" "hyperglass_devices" { + metadata { + name = "hyperglass-devices" + namespace = kubernetes_namespace_v1.netops.metadata[0].name + } + # Spine only: hyperglass's juniper directives require source4 AND source6 per + # device, and only the spine has both (lo0 + the transit v6) — the leaf has no + # public/v6 presence, so LG queries from it would be meaningless anyway. + data = { + "devices.yaml" = yamlencode({ + devices = [ + { + name = "corenetsw" + description = "spine VC (QFX5200-32C x2)" + address = module.addr_site.spine_mgmt_ip + platform = "juniper" + attrs = { + source4 = "69.48.224.254" # lo0 (fabric stack, transits.loopback) + source6 = "2a01:4a0:1338:226::2" # transit /64 local (fabric stack, transits.local_v6) + } + credential = { username = "netops", password = data.onepassword_item.netops_fabric_password.password } + }, + ] + }) + } +} diff --git a/tf/deployment/prod/htz-fsn1/talos/network.tf b/tf/deployment/prod/htz-fsn1/talos/network.tf deleted file mode 100644 index 7933f04d..00000000 --- a/tf/deployment/prod/htz-fsn1/talos/network.tf +++ /dev/null @@ -1,23 +0,0 @@ -# Isolated Hetzner Cloud network for the control plane. Holds ONLY the 3 CP VMs -# (etcd CP↔CP on private IPs) + the API LB's private IP. The workers do NOT join -# it — CP↔worker rides the NetBird mesh. Range = kube-cp (10.40.11.0/24), carved -# from the site supernet for collision-free IPAM (see fabric-addressing). -resource "hcloud_network" "kube_cp" { - name = "yucca-${var.region_code}-${var.cluster.name}-kube-cp" - ip_range = local.kube_cp_cidr - labels = { cluster = var.cluster.name, plane = "control" } -} - -# Single cloud subnet for the CP VMs + LB (no vSwitch subnet — workers aren't here). -resource "hcloud_network_subnet" "kube_cp" { - network_id = hcloud_network.kube_cp.id - type = "cloud" - network_zone = "eu-central" # fsn1/nbg1/hel1 - ip_range = local.kube_cp_cidr -} - -locals { - # Deterministic private IPs in the kube-cp subnet. - cp_private_ips = [for i in range(var.cluster.cp_count) : cidrhost(local.kube_cp_cidr, var.cluster.cp_ip_offset + i)] - lb_private_ip = cidrhost(local.kube_cp_cidr, var.cluster.lb_ip_offset) -} diff --git a/tf/deployment/prod/htz-fsn1/talos/outputs.tf b/tf/deployment/prod/htz-fsn1/talos/outputs.tf index 436e52e9..470882d1 100644 --- a/tf/deployment/prod/htz-fsn1/talos/outputs.tf +++ b/tf/deployment/prod/htz-fsn1/talos/outputs.tf @@ -4,36 +4,29 @@ output "cluster_summary" { cluster_name = var.cluster.name api_endpoint = local.cluster_endpoint api_dns_name = local.api_dns_name + api_vip = local.api_vip operator_endpoint = local.operator_endpoint - lb_private_ip = local.lb_private_ip - cp_public_ips = local.cp_public_ips - cp_private_ips = local.cp_private_ips + cp_ips = local.cp_ips worker_fabric_ips = [for w in var.cluster.workers : w.fabric_ip] } } -# DNS hint: api_dns_name resolves to the PRIVATE LB IP (lb_public = false — the -# API is reachable only over the NetBird mesh; the netbird stack's +# DNS hint: api_dns_name resolves to the Talos-elected VIP on the kube-cp VLAN +# (the API is reachable only over the NetBird kube-cp route; the netbird stack's # yucca.futo.network zone serves the record). No public A record exists. output "api_dns_record" { - description = "The internal record: → (NetBird DNS)." - value = "${local.api_dns_name} A ${local.lb_private_ip}" + description = "The internal record: → (NetBird DNS)." + value = "${local.api_dns_name} A ${local.api_vip}" } -# Image Factory outputs — `mise run hetzner:talos-image` reads the hcloud URL to -# build the snapshot; the metal URL feeds the worker rescue-install runbook. +# Image Factory outputs — the metal URL feeds the rescue-install runbook (README). output "talos_schematic_id" { description = "Image Factory schematic id (from schematic.yaml)." value = local.talos_schematic_id } -output "talos_hcloud_image_url" { - description = "Factory hcloud-amd64 raw.xz URL — input to hcloud-upload-image (CP snapshot)." - value = local.talos_hcloud_image_url -} - output "talos_metal_image_url" { - description = "Factory metal-amd64 raw.xz URL — workers dd this in rescue." + description = "Factory metal-amd64 raw.xz URL — nodes dd this in rescue." value = local.talos_metal_image_url } diff --git a/tf/deployment/prod/htz-fsn1/talos/providers.tf b/tf/deployment/prod/htz-fsn1/talos/providers.tf index e59083e4..1e2be9a0 100644 --- a/tf/deployment/prod/htz-fsn1/talos/providers.tf +++ b/tf/deployment/prod/htz-fsn1/talos/providers.tf @@ -1,11 +1,8 @@ -# hcloud — the control-plane VMs, their private subnet, the API LB, and the Talos -# snapshot lookup. Token comes from HCLOUD_TOKEN (op run --env-file=tf/.env.prod). -provider "hcloud" {} - # helm + kubernetes bind to ONE cluster: the bootstrap CP's directly-reachable -# (public) apiserver — up immediately after bootstrap and in the cert SANs, unlike -# the LB which only goes healthy once an apiserver answers. Creds come from the -# Talos-minted admin kubeconfig (known after talos_cluster_kubeconfig applies). +# apiserver (over the NetBird kube-cp route) — up immediately after bootstrap and +# in the cert SANs, unlike the VIP which only settles once etcd elects a holder. +# Creds come from the Talos-minted admin kubeconfig (known after +# talos_cluster_kubeconfig applies). provider "helm" { kubernetes = { host = local.operator_endpoint @@ -25,8 +22,3 @@ provider "kubernetes" { # 1Password — persists the kube/talosconfig into yucca_tf_prod (secrets.tf). Auth # via OP_SERVICE_ACCOUNT_TOKEN (op run). provider "onepassword" {} - -# netbird — read-only worker peer lookups: their mesh IPs feed the CPs' -# extraHostEntries so the apiserver dials worker kubelets peer-to-peer over the -# mesh (no mgmt route in the path). PAT via NB_PAT (op run). -provider "netbird" {} diff --git a/tf/deployment/prod/htz-fsn1/talos/schematic-worker.yaml b/tf/deployment/prod/htz-fsn1/talos/schematic-worker.yaml deleted file mode 100644 index a912fe6f..00000000 --- a/tf/deployment/prod/htz-fsn1/talos/schematic-worker.yaml +++ /dev/null @@ -1,11 +0,0 @@ -# Talos Image Factory schematic for the father WORKERS (bare-metal Hetzner Robot). -# Split from schematic.yaml (the CP/VM set): qemu-guest-agent must NOT be here — on -# metal there is no virtio port, the extension service waits forever for -# /dev/virtio-ports/org.qemu.guest_agent.0, and the boot sequence never reaches -# `running`, failing every talos health check (and with it every TF plan/apply). -customization: - systemExtensions: - officialExtensions: - - siderolabs/netbird # node-level overlay (worker↔apiserver) - - siderolabs/intel-ucode # worker Xeon microcode - - siderolabs/util-linux-tools # fstrim et al. diff --git a/tf/deployment/prod/htz-fsn1/talos/schematic.yaml b/tf/deployment/prod/htz-fsn1/talos/schematic.yaml index e5c99092..500a890e 100644 --- a/tf/deployment/prod/htz-fsn1/talos/schematic.yaml +++ b/tf/deployment/prod/htz-fsn1/talos/schematic.yaml @@ -1,12 +1,13 @@ # Talos Image Factory schematic for the `father` cluster — the single source of -# truth for the node extension set. Registered with the factory by TF -# (talos_image_factory_schematic, image.tf); its id derives the CP (hcloud) and -# worker (metal) image URLs. One schematic covers both platforms — the extras are -# harmless no-ops on the other (qemu-guest-agent on metal, intel-ucode on a VM). +# truth for the node extension set (CPs + workers, all bare-metal Hetzner Robot). +# Registered with the factory by TF (talos_image_factory_schematic, image.tf); its +# id derives the metal installer/image URLs. NB: qemu-guest-agent must NOT be here — +# on metal there is no virtio port, the extension service waits forever for +# /dev/virtio-ports/org.qemu.guest_agent.0, and the boot sequence never reaches +# `running`, failing every talos health check (and with it every TF plan/apply). customization: systemExtensions: officialExtensions: - - siderolabs/netbird # node-level overlay (CP↔fabric route) - - siderolabs/qemu-guest-agent # hcloud control-plane VMs - - siderolabs/intel-ucode # worker Xeon microcode + - siderolabs/netbird # node-level overlay (operator plane / kube-cp route) + - siderolabs/intel-ucode # Xeon microcode - siderolabs/util-linux-tools # fstrim et al. diff --git a/tf/deployment/prod/htz-fsn1/talos/talos.tf b/tf/deployment/prod/htz-fsn1/talos/talos.tf index f5ce7b8a..6bc6220a 100644 --- a/tf/deployment/prod/htz-fsn1/talos/talos.tf +++ b/tf/deployment/prod/htz-fsn1/talos/talos.tf @@ -1,22 +1,18 @@ -# ── Talos bring-up (hybrid) ────────────────────────────────────────────────── -# Cloud CPs are configured via hcloud user_data (controlplane.tf); bare-metal -# workers via apid apply (workers.tf). Both join the SAME cluster (one set of -# machine_secrets) and the SAME NetBird mesh (node IPs are NetBird addresses). +# ── Talos bring-up (all bare-metal) ────────────────────────────────────────── +# CPs and workers are both driven over apid (controlplane.tf / workers.tf) into +# ONE cluster (one set of machine_secrets). All node planes ride the fabric: # -# node plane (CP↔worker, etcd-client, apiserver↔kubelet) → NetBird (100.64/10) -# etcd (CP↔CP) → kube-cp hcloud subnet -# worker↔worker pod east-west → kube fabric (50G), Cilium BGP -# API endpoint → public Hetzner Cloud LB +# etcd (CP↔CP) + apiserver + API VIP → kube-cp fabric VLAN 11 (10.40.11.0/24) +# worker↔worker pod east-west → kube fabric VLAN 10 (50G), Cilium BGP +# CP↔worker (apiserver↔kubelet, geneve) → routed kube↔kube-cp via the spine IRBs +# operators/CI → NetBird mesh (kube-cp routed via the CPs) # -# Bootstrap/kubeconfig/health dial the CP PUBLIC IPs (firewalled) — the only thing -# the TF runner can reach before NetBird/the LB settle. +# NetBird stays on every node as the operator/backup plane — node-to-node traffic +# no longer depends on it (static fabric routes are pinned in the machine configs). -# Node names are EXPLICIT in tfvars (cluster.cp_names + workers[*].name) — the -# node-names shuffle module was retired here: auto-picked names re-roll when the -# pool input changes (adding an explicit name shrinks the shuffle pool → every -# auto name changes → every node renames), and positional slotting meant a -# cp_count change renamed all workers. migrations.tf forgets the old module -# state without destroying anything. +# Node names are EXPLICIT in tfvars (cluster.cps[*].name + workers[*].name) — +# auto-picked names re-roll when the pool input changes, silently renaming (= +# replacing) live nodes. locals { c = var.cluster @@ -26,49 +22,40 @@ locals { pod_cidr = "10.250.0.0/17" # 10.250.0.0 – 10.250.127.255 service_cidr = "10.250.128.0/17" # 10.250.128.0 – 10.250.255.255 - # Factory installers (keep each schematic's extensions). Workers consult theirs on - # install/upgrade; CPs boot the hcloud snapshot and only consult this on a reinstall. - # SPLIT per role: the worker schematic drops qemu-guest-agent (blocks metal boot). - cp_install_image = "factory.talos.dev/metal-installer/${local.talos_schematic_id}:v${local.c.talos_version}" - worker_install_image = "factory.talos.dev/metal-installer/${local.talos_worker_schematic_id}:v${local.c.talos_version}" + # Factory installer (keeps the schematic's extensions) — one schematic for every + # node (all metal now); consulted on install/upgrade. + install_image = "factory.talos.dev/metal-installer/${local.talos_schematic_id}:v${local.c.talos_version}" - # Private API endpoint: a NetBird DNS-zone name resolving to the PRIVATE LB IP - # (10.40.11.5). Name = kube....yucca.futo.network. It's - # in the cert SANs + on each CP as a host-entry; NetBird peers resolve it via the - # yucca.futo.network zone (netbird stack) and reach the LB over the kube-cp route - # (CPs are the route peers). Node-side traffic never resolves it: kubelets dial - # KubePrism (127.0.0.1:7445), which load-balances to the CP IPs directly. - # legacy_api_dns_name (the old yucca.internal name) stays in the SANs + host - # entries so pre-migration kubeconfigs keep verifying — drop it once rotated. - api_dns_name = "kube.${local.c.name}.${var.region_code}.${var.provider_code}.yucca.futo.network" - legacy_api_dns_name = "kube.${local.c.name}.${var.region_code}.${var.provider_code}.yucca.internal" - cluster_endpoint = "https://${local.api_dns_name}:6443" + # API endpoint: a NetBird DNS-zone name resolving to the Talos-elected VIP + # (10.40.11.5, kube-cp VLAN — same IP the retired hcloud LB held, so the record + # carried over). Name = kube....yucca.futo.network. + # It's in the cert SANs + on each node as a host-entry; NetBird peers resolve it + # via the yucca.futo.network zone (netbird stack) and reach the VIP over the + # kube-cp route (CPs are the route peers). Node-side traffic never resolves it: + # kubelets dial KubePrism (127.0.0.1:7445), which load-balances to the CP IPs. + api_dns_name = "kube.${local.c.name}.${var.region_code}.${var.provider_code}.yucca.futo.network" + cluster_endpoint = "https://${local.api_dns_name}:6443" - kube_cp_prefix = split("/", local.kube_cp_cidr)[1] # 24 + api_vip = cidrhost(local.kube_cp_cidr, local.c.vip_offset) # 10.40.11.5 + kube_cp_prefix = split("/", local.kube_cp_cidr)[1] # 24 - # Hostnames: yucca-htz-fsn-father-k8s-. Workers additionally get a - # hostname-keyed map — the STABLE key for the apply resources (workers.tf) and - # the netbird peer lookups, so list edits can't shift another node's identity. - node_prefix = "yucca-${var.provider_code}-${var.region_code}-${local.c.name}-k8s" - cp_hostnames = [for n in local.c.cp_names : "${local.node_prefix}-${n}"] - workers_named = [for w in local.c.workers : merge(w, { hostname = "${local.node_prefix}-${w.name}" })] - worker_hostnames = [for w in local.workers_named : w.hostname] - worker_node_map = { for w in local.workers_named : w.hostname => w } + # Hostnames: yucca-htz-fsn-father-k8s-. Both roles get hostname-keyed + # maps — the STABLE key for the apply resources, so list edits can't shift + # another node's identity. + node_prefix = "yucca-${var.provider_code}-${var.region_code}-${local.c.name}-k8s" + cps_named = [for n in local.c.cps : merge(n, { hostname = "${local.node_prefix}-${n.name}" })] + cp_node_map = { for n in local.cps_named : n.hostname => n } + cp_ips = local.cps_named[*].cp_ip + workers_named = [for w in local.c.workers : merge(w, { hostname = "${local.node_prefix}-${w.name}" })] + worker_node_map = { for w in local.workers_named : w.hostname => w } - # apiserver cert SANs — the names/IPs clients dial. NOT the CP public IPs (those - # don't exist until the servers are created from this very config). + # apiserver cert SANs — the names/IPs clients dial. apiserver_cert_sans = concat( - [local.api_dns_name, local.legacy_api_dns_name, local.lb_private_ip], - local.cp_private_ips, + [local.api_dns_name, local.api_vip], + local.cp_ips, ["127.0.0.1", "localhost"], ) - # ── Shared patches (every node) ────────────────────────────────────────── - cp_install_patch = yamlencode({ - machine = { install = { disk = local.c.install_disk, image = local.cp_install_image } } - }) - - # Talos's default forwards coredns's upstream queries to the host DNS on a link-local # address (169.254.116.108) — unreachable from pods under Cilium's eBPF datapath # (bpf.masquerade), so every EXTERNAL lookup from a pod times out while cluster.local @@ -77,20 +64,13 @@ locals { machine = { features = { hostDNS = { forwardKubeDNSToHost = false } } } }) - # NetBird node-level overlay. CPs and workers join with DIFFERENT setup keys so they - # land in different groups: CPs → [talos, talos_cp], workers → [talos]. talos_cp is - # the CP-only router group for the kube-cp network — the workers must NOT be in it, or - # NetBird treats them as kube-cp routers and they never install the client route (their - # pods can't reach the apiserver). See the netbird stack for the group/router wiring. - # Both CP + worker run netbird in its normal (modern) mode. The pod→routed-subnet - # problem — Cilium's eBPF host-routing does its FIB lookup against the MAIN table - # only, so any route netbird parks in a policy table (it has been observed using - # both main and table 7120 across versions/restarts) is invisible to POD egress, - # and pod→apiserver via kube-cp gets "no route to host" — is fixed DETERMINISTICALLY - # on the workers by worker_netbird_route_patch (a Talos-managed main-table route), - # NOT by pinning netbird to its deprecated NB_USE_LEGACY_ROUTING mode. CPs don't - # run the eBPF pod-datapath to routed subnets (their control-plane pods are - # hostNetwork → host stack, which honors policy routing), so they need no route. + # NetBird node-level overlay — the OPERATOR plane (kube-cp routed to the mesh via + # the CPs) and a backup path; node-to-node traffic rides the fabric via the static + # routes pinned below. CPs and workers join with DIFFERENT setup keys so they land + # in different groups: CPs → [talos, talos_cp], workers → [talos]. talos_cp is the + # CP-only router group for the kube-cp network — the workers must NOT be in it, or + # NetBird treats them as kube-cp routers and they never install the client route. + # See the netbird stack for the group/router wiring. netbird_env = ["NB_MANAGEMENT_URL=https://api.netbird.io"] cp_netbird_patch = var.netbird_talos_cp_setup_key != "" ? yamlencode({ apiVersion = "v1alpha1" @@ -105,30 +85,10 @@ locals { environment = concat(["NB_SETUP_KEY=${var.netbird_talos_setup_key}"], local.netbird_env) }) : "" - # DETERMINISTIC pod→apiserver fix: a Talos-managed route for the kube-cp subnet - # (apiserver + private API LB, reachable only over the mesh) into the MAIN table - # via wt0 — exactly where Cilium's eBPF FIB lookup reads. netbird still installs - # its own route (its table is version-dependent); ours guarantees main is - # populated regardless, so pod egress to kube-cp always resolves. Declaring a - # route on wt0 does NOT disturb the netbird extension (it keeps owning wt0's - # address; Talos only adds the route). Workers only — CPs are ON kube-cp. - worker_netbird_route_patch = yamlencode({ - machine = { - network = { - interfaces = [{ - interface = "wt0" - routes = [{ network = local.kube_cp_cidr }] - }] - } - } - }) - # nodeIP selection: - # CPs → kube-cp hcloud private subnet (apiserver↔CP-kubelet stays private; - # decoupled from NetBird readiness at boot) - # workers → kube fabric IP. All workers share VLAN-10 L2, so Cilium - # autoDirectNodeRoutes routes pod east-west directly over the 50G - # fabric — no BGP, no overlay. + # CPs → kube-cp fabric VLAN 11 (etcd + apiserver↔CP-kubelet) + # workers → kube fabric VLAN 10 (all workers share the L2, so Cilium routes pod + # east-west directly over the 50G fabric) # clusterDNS must be set explicitly: Talos defaults it to 10.96.0.10 (the upstream # default service CIDR's DNS) and does NOT derive it from our serviceSubnets — the # kube-dns Service actually lands at cidrhost(service_cidr, 10). Without this every @@ -196,13 +156,10 @@ locals { } }) - cp_base_patches = compact([local.cp_install_patch, local.hostdns_patch, local.cp_netbird_patch, local.cp_nodeip_patch]) - # Workers ARE NetBird peers: the apiserver lives on the kube-cp hcloud net, only - # reachable over the mesh, and workers resolve the API endpoint via the yucca.internal - # NetBird DNS zone. nodeIP stays on the fabric (worker_nodeip_patch) so pod east-west - # rides VLAN 10; only the API control path uses NetBird. - # (install patch is PER-WORKER — by disk serial — appended in workers.tf) - worker_base_patches = compact([local.hostdns_patch, local.worker_mayastor_patch, local.worker_volumes_patch, local.worker_netbird_patch, local.worker_netbird_route_patch, local.worker_nodeip_patch]) + cp_base_patches = compact([local.hostdns_patch, local.cp_netbird_patch, local.cp_nodeip_patch]) + # (install patch is PER-NODE — by disk serial — appended in controlplane.tf / + # workers.tf, along with the bond/VLAN network patches.) + worker_base_patches = compact([local.hostdns_patch, local.worker_mayastor_patch, local.worker_volumes_patch, local.worker_netbird_patch, local.worker_nodeip_patch]) # ── Control-plane cluster config (same on every CP) ────────────────────── cp_cluster_patch = yamlencode({ @@ -222,21 +179,13 @@ locals { coreDNS = { disabled = true } apiServer = { certSANs = local.apiserver_cert_sans - # Dial kubelets by HOSTNAME first (Talos's default is InternalIP-first). The - # worker InternalIPs are fabric addresses only reachable via the mgmt NetBird - # routers — a single flappy bridge that intermittently broke logs/exec. Worker - # hostnames resolve (via the CPs' extraHostEntries below) to the workers' OWN - # NetBird IPs, so apiserver→kubelet is peer-to-peer over the mesh — the same - # always-on tunnels the kubelets already use to reach the apiserver. The CP - # hostnames resolve via hcloud DNS to their kube-cp IPs, unchanged. - extraArgs = { "kubelet-preferred-address-types" = "Hostname,InternalIP,ExternalIP" } # hostNetwork pods get /etc/hosts COPIED at sandbox creation — a host-level # extraHostEntries refresh never reaches the RUNNING apiserver. Stamping the # entry-set hash into the pod spec forces kubelet to recreate the pod (fresh - # /etc/hosts) whenever a worker's mesh IP changes (e.g. re-provision). - env = { MESH_HOSTS_REVISION = substr(sha256(jsonencode(local.cp_host_entries)), 0, 12) } + # /etc/hosts) whenever the entry set changes (e.g. a node add). + env = { HOSTS_REVISION = substr(sha256(jsonencode(local.cp_host_entries)), 0, 12) } } - # Pin etcd to the kube-cp hcloud subnet so CP↔CP etcd stays off the mesh. + # Pin etcd to the kube-cp VLAN so CP↔CP etcd stays off the mesh + public NICs. etcd = { advertisedSubnets = [local.kube_cp_cidr] } } }) @@ -244,25 +193,20 @@ locals { # CP node extras: # • ip_forward — the CPs are the NetBird route peers for the kube-cp subnet # (yucca-fsn-father-kube-cp), so they must forward overlay↔subnet traffic. - # • extraHostEntries — on the CPs, resolve api_dns_name to the 3 CP private IPs - # (round-robin, all in the cert SANs). NOT the LB VIP (CPs are LB targets → - # hcloud hairpin), and NOT 127.0.0.1 (a joining CP must reach a WORKING - # apiserver — a peer's — to register its etcd membership; its own apiserver - # isn't up until etcd joins). Off-node peers resolve api_dns_name via the - # NetBird yucca.internal zone. - # • kubelet dialing (worker_mesh_kubelet): the apiserver prefers the Hostname node - # address (cp_cluster_patch), so every node hostname must resolve on the CPs: - # CP hostnames → their kube-cp IPs (stable), worker hostnames → their NetBird - # IPs (data.netbird_peer — the peer-to-peer mesh path, no mgmt route). Talos - # host-dns can't resolve NetBird DNS zones, hence /etc/hosts, which the - # hostNetwork apiserver inherits. + # • extraHostEntries — resolve api_dns_name to the 3 CP IPs (round-robin, all in + # the cert SANs). NOT the VIP (a joining CP must reach a WORKING apiserver — a + # peer's — to register its etcd membership; the VIP may be parked on itself), + # and NOT 127.0.0.1. Every node hostname also resolves to its fabric IP so + # apiserver→kubelet dials ride the fabric (Talos host-dns can't resolve NetBird + # DNS zones, hence /etc/hosts, which the hostNetwork apiserver inherits). + # Off-node peers resolve api_dns_name via the NetBird yucca.futo.network zone. cp_host_entries = concat( - [for ip in local.cp_private_ips : { ip = ip, aliases = [local.api_dns_name, local.legacy_api_dns_name] }], - [for i, ip in local.cp_private_ips : { ip = ip, aliases = [local.cp_hostnames[i]] }], - # Iterate the tfvars LIST (not the hostname-keyed data map, whose lexical - # order differs) — entry order is part of the rendered CP config, and - # reordering it would churn every CP's machine config for nothing. - [for w in local.workers_named : { ip = data.netbird_peer.worker[w.hostname].ip, aliases = [w.hostname] } if local.c.worker_mesh_kubelet], + [for ip in local.cp_ips : { ip = ip, aliases = [local.api_dns_name] }], + # Iterate the tfvars LISTS (not the hostname-keyed maps, whose lexical order + # differs) — entry order is part of the rendered CP config, and reordering it + # would churn every CP's machine config for nothing. + [for n in local.cps_named : { ip = n.cp_ip, aliases = [n.hostname] }], + [for w in local.workers_named : { ip = w.fabric_ip, aliases = [w.hostname] }], ) cp_extras_patch = yamlencode({ @@ -271,28 +215,6 @@ locals { network = { extraHostEntries = local.cp_host_entries } } }) - - # ── Per-CP patches (hostname + hcloud private NIC for etcd) ─────────────── - # eth0 = hcloud public (DHCP, default route); eth1 = hcloud private (etcd). - # eth1 MUST be DHCP: hcloud private networks are SDN, not L2 — servers reach each - # other via the network gateway, and hcloud's DHCP is what installs the private - # IP (the one pinned in the hcloud_server network block) + the gateway route. A - # static /24 here makes the node try direct same-subnet ARP, which the SDN doesn't - # answer → the CPs can't reach each other → etcd never forms. VERIFY eth1 is the - # private NIC on the snapshot (else use a deviceSelector). - cp_node_patches = [for i in range(local.c.cp_count) : [ - yamlencode({ - machine = { - network = { - interfaces = [{ - interface = "eth1" - dhcp = true - }] - } - } - }), - yamlencode({ apiVersion = "v1alpha1", kind = "HostnameConfig", auto = "off", hostname = local.cp_hostnames[i] }), - ]] } # Cluster PKI (sensitive). @@ -307,21 +229,9 @@ resource "talos_machine_secrets" "this" { } } -# Worker NetBird peers — their mesh IPs feed the CPs' /etc/hosts (cp_host_entries) -# so the apiserver dials worker kubelets peer-to-peer. Lookup is by peer name -# (= the worker hostname; the netbird stack keeps one live peer per node). Gated: -# on a greenfield bootstrap the workers aren't peers yet — set -# cluster.worker_mesh_kubelet = false, then flip it after they join. -data "netbird_peer" "worker" { - for_each = { for hostname, w in local.worker_node_map : hostname => w if local.c.worker_mesh_kubelet } - name = each.key -} - -# Per-CP machine config — rendered into hcloud user_data (controlplane.tf). Each -# CP gets the shared + CP-cluster + its own per-node patches. +# CP base config — per-CP install/network/hostname patches are added at apply +# time (controlplane.tf). data "talos_machine_configuration" "cp" { - count = local.c.cp_count - cluster_name = local.c.name machine_type = "controlplane" cluster_endpoint = local.cluster_endpoint @@ -331,7 +241,6 @@ data "talos_machine_configuration" "cp" { config_patches = concat( local.cp_base_patches, [local.cp_cluster_patch, local.cp_extras_patch], - local.cp_node_patches[count.index], local.common_firewall_patches, local.cp_firewall_patches, ) @@ -349,16 +258,15 @@ data "talos_machine_configuration" "worker" { config_patches = concat(local.worker_base_patches, local.common_firewall_patches) } -# ── Bootstrap / kubeconfig / health (dial CP public IPs) ────────────────────── +# ── Bootstrap / kubeconfig / health ─────────────────────────────────────────── locals { - cp_public_ips = hcloud_server.control_plane[*].ipv4_address - # Everything the talos provider dials — bootstrap, kubeconfig, talosconfig, health, - # and the helm/kubernetes providers — uses the PRIVATE kube-cp IPs, reachable from - # the apply host over the NetBird kube-cp route (and in the cert SANs). No public - # access is required to bring the cluster up (the CPs keep public IPs only for - # NetBird NAT traversal + egress; apid/apiserver are firewalled off the internet). - bootstrap_endpoint = local.cp_private_ips[0] - operator_endpoint = "https://${local.cp_private_ips[0]}:6443" + # Everything the talos provider dials — bootstrap, kubeconfig, talosconfig, + # health, and the helm/kubernetes providers — uses the kube-cp IPs, reachable + # from the apply host over the NetBird kube-cp route (and in the cert SANs). + # During a greenfield bring-up the route appears as soon as the first CP boots + # into the cluster and joins the mesh (the CPs are the route peers). + bootstrap_endpoint = local.cp_ips[0] + operator_endpoint = "https://${local.cp_ips[0]}:6443" } # One-shot bootstrap against the first CP. Re-running rolls cluster identity. @@ -368,7 +276,7 @@ resource "talos_machine_bootstrap" "this" { endpoint = local.bootstrap_endpoint timeouts = { create = "10m" } - depends_on = [hcloud_server.control_plane] + depends_on = [talos_machine_configuration_apply.cp] lifecycle { # A replace re-bootstraps a LIVE cluster (identity roll). The re-bootstrap @@ -385,12 +293,12 @@ resource "talos_cluster_kubeconfig" "this" { depends_on = [talos_machine_bootstrap.this] } -# talosconfig endpoints = CP private kube-cp IPs (reached over NetBird; no public). +# talosconfig endpoints = CP kube-cp IPs (reached over NetBird; no public). data "talos_client_configuration" "this" { cluster_name = local.c.name client_configuration = talos_machine_secrets.this.client_configuration - endpoints = local.cp_private_ips - nodes = concat(local.cp_private_ips, [for w in local.c.workers : w.fabric_ip]) + endpoints = local.cp_ips + nodes = concat(local.cp_ips, [for w in local.c.workers : w.fabric_ip]) } locals { @@ -403,9 +311,9 @@ data "talos_cluster_health" "this" { count = var.bootstrap_health_gate ? 1 : 0 client_configuration = talos_machine_secrets.this.client_configuration - control_plane_nodes = local.cp_private_ips + control_plane_nodes = local.cp_ips worker_nodes = [for w in local.c.workers : w.fabric_ip] - endpoints = local.cp_private_ips + endpoints = local.cp_ips skip_kubernetes_checks = true timeouts = { read = "10m" } diff --git a/tf/deployment/prod/htz-fsn1/talos/terragrunt.hcl b/tf/deployment/prod/htz-fsn1/talos/terragrunt.hcl index 910133f6..44b6909e 100644 --- a/tf/deployment/prod/htz-fsn1/talos/terragrunt.hcl +++ b/tf/deployment/prod/htz-fsn1/talos/terragrunt.hcl @@ -4,11 +4,10 @@ include "root" { # clusters.auto.tfvars is loaded automatically by OpenTofu in this directory. # State backend + partition/region/stack (prod/htz-fsn1/talos) are derived by the -# root config. Unlike the austin talos stack (which talks straight to bare-metal -# nodes already in maintenance mode), this stack is HYBRID: it provisions the 3 -# Hetzner Cloud control-plane VMs (+ a small private subnet for etcd + the API -# load balancer) AND drives Talos on the 3 bare-metal workers. CP↔worker traffic -# rides the NetBird mesh; worker east-west rides the 50G fabric. See ./README.md. +# root config. Like the austin talos stack, this talks straight to bare-metal +# nodes already in Talos maintenance mode: 3 CPs on the kube-cp fabric VLAN +# (etcd + API VIP) + 3 workers on the kube VLAN, routed by the spine IRBs. +# See ./README.md. # -# Secrets (HCLOUD_TOKEN, NetBird setup key, S3 state) are injected by +# Secrets (NetBird setup keys, S3 state) are injected by # op run --env-file=tf/.env.prod diff --git a/tf/deployment/prod/htz-fsn1/talos/variables.tf b/tf/deployment/prod/htz-fsn1/talos/variables.tf index 98fdcc38..0498387a 100644 --- a/tf/deployment/prod/htz-fsn1/talos/variables.tf +++ b/tf/deployment/prod/htz-fsn1/talos/variables.tf @@ -1,9 +1,10 @@ -# Hybrid prod cluster topology — the single source of truth (clusters.auto.tfvars). -# One object, not a map: this stack's bring-up is bespoke (cloud CP via hcloud -# user_data + bare-metal workers via apid apply), so a for_each map buys nothing. +# All-bare-metal prod cluster topology — the single source of truth +# (clusters.auto.tfvars). One object, not a map: this stack's bring-up is bespoke +# (CPs + workers both driven over apid, but with different planes/volumes), so a +# for_each map buys nothing. variable "cluster" { - description = "The prod hybrid Talos cluster (Star Wars name; prod = 'father')." + description = "The prod bare-metal Talos cluster (Star Wars name; prod = 'father')." type = object({ name = string talos_version = string @@ -12,47 +13,51 @@ variable "cluster" { # The Image Factory schematic (extension set) is managed in TF — see # schematic.yaml + talos_image_factory_schematic in image.tf. The schematic id # and image URLs derive from it, so they're NOT inputs here. - install_disk = string cilium_version = string hubble = bool - # NetBird mesh range node IPs come from (kubelet nodeIP.validSubnets). THIS - # account assigns 10.254.0.0/15 (see clusters.auto.tfvars) — not the NetBird - # Cloud default of 100.64.0.0/10. The node plane (CP↔worker) rides this mesh. + # NetBird mesh range (host firewall trust + operator plane). THIS account + # assigns 10.254.0.0/15 (see clusters.auto.tfvars) — not the NetBird Cloud + # default of 100.64.0.0/10. netbird_node_cidr = string - # ── Cloud control plane (Hetzner Cloud) ────────────────────────────────── - cp_count = number # 3 - # EXPLICIT node names (wordlist-style), one per CP, in cp_ip_offset order. - # Names are PINNED — never auto-shuffled — so node identity can't silently - # re-roll on a list edit (renaming a live node's hostname = renaming its - # Kubernetes node = effectively replacing it). - cp_names = list(string) - cp_server_type = string # ccx23 (dedicated vCPU x86) - cp_location = string # fsn1 - cp_ip_offset = number # CP[i] private (kube-cp) IP = cidrhost(kube_cp, offset+i) - lb_type = string # lb11 - lb_ip_offset = number # API LB private IP = cidrhost(kube_cp, offset) - lb_public = bool # also expose a public frontend (operators/workers reach it) - - # ── Bare-metal workers (Hetzner Robot) ──────────────────────────────────── - # maint_ip = the Hetzner public IP the node comes up on in Talos maintenance - # mode (DHCP) — the endpoint for the one-time config apply. fabric_ip = the - # post-install kube (VLAN 10) address (nodeIP + worker east-west); the apiserver - # reaches the kubelet there via NetBird→mgmt, and the node reaches the apiserver - # over its own NetBird peer. - workers = list(object({ - name = string # EXPLICIT node name (see cp_names) — keys the apply resources; renaming = node replacement - # Install-disk NVMe serial — NOT a device name: nvme0/nvme1 enumeration is - # not stable across boots (observed swapping), and a name-based install - # target could point an upgrade at the DATA disk. + # ── Bare-metal control planes (Hetzner Robot; kube-cp fabric VLAN 11) ───── + # cp_ip = the post-install kube-cp (VLAN 11) address — etcd + apiserver + + # nodeIP; the spine routes kube↔kube-cp. maint_ip = the Hetzner public IP the + # node comes up on in Talos maintenance mode (DHCP on the onboard 1G NIC) — + # the endpoint for the one-time install apply. + cps = list(object({ + name = string # EXPLICIT node name (wordlist-style) — keys the apply resources; renaming = node replacement + # Install-disk serial — NOT a device name: sda/sdb enumeration is not + # stable across boots, and a name-based install target could point an + # upgrade at the wrong disk. install_serial = string - fabric_ip = string # 10.40.10.x on the kube fabric VLAN + cp_ip = string # 10.40.11.x on the kube-cp fabric VLAN maint_ip = string # Hetzner public IP (maintenance-mode apid endpoint) robot_id = number # Hetzner Robot server number (provisioning/doc) provisioned = optional(bool, true) # false ONLY while first-provisioning: config - # applies then target maint_ip (maintenance mode); true = target fabric_ip (live). + # applies then target maint_ip (maintenance mode); true = target cp_ip (live). + })) + # CP fabric bond members (2×10G Intel 82599 SFP+). Selected by NIC driver — + # ixgbe matches exactly the two 10G ports (the onboard 1G public NIC is e1000e). + cp_bond_driver = optional(string) + cp_bond_interfaces = optional(list(string), []) + # API VIP = cidrhost(kube_cp, vip_offset) — Talos etcd-elected, floats between + # the CPs on VLAN 11. 5 keeps the retired hcloud LB's IP, so the api_dns_name + # record (NetBird DNS zone) carried over unchanged. + vip_offset = number + + # ── Bare-metal workers (Hetzner Robot; kube fabric VLAN 10) ─────────────── + # maint_ip/fabric_ip semantics as for cps; nodeIP = fabric_ip (worker east-west + # rides VLAN 10 at 50G, apiserver↔kubelet routes via the spine IRBs). + workers = list(object({ + name = string + install_serial = string + fabric_ip = string # 10.40.10.x on the kube fabric VLAN + maint_ip = string + robot_id = number + provisioned = optional(bool, true) })) # Fabric bond members. Prefer worker_bond_driver (a Talos deviceSelector by NIC # driver, e.g. "bnxt_en") — robust across per-node PCI naming. worker_bond_interfaces @@ -62,21 +67,10 @@ variable "cluster" { # Worker default route (egress for image pulls + NetBird): via the kube fabric # IRB gateway (fabric transit) when true, else the Hetzner public NIC (DHCP). worker_default_route_via_fabric = optional(bool, true) - - # apiserver→kubelet rides the mesh peer-to-peer: the CPs get /etc/hosts entries - # mapping each worker hostname to its NetBird IP (data.netbird_peer lookups) and - # the apiserver prefers the Hostname node address. Requires the workers to BE - # NetBird peers — set false for a greenfield bootstrap (no peers to look up yet), - # flip true once the workers have joined. See cp_extras_patch in talos.tf. - worker_mesh_kubelet = optional(bool, true) }) validation { - condition = length(var.cluster.cp_names) == var.cluster.cp_count - error_message = "cluster.cp_names must have exactly cp_count entries (one name per CP, in cp_ip_offset order)." - } - validation { - condition = length(distinct(concat(var.cluster.cp_names, var.cluster.workers[*].name))) == var.cluster.cp_count + length(var.cluster.workers) + condition = length(distinct(concat(var.cluster.cps[*].name, var.cluster.workers[*].name))) == length(var.cluster.cps) + length(var.cluster.workers) error_message = "Node names must be unique across CPs and workers." } } diff --git a/tf/deployment/prod/htz-fsn1/talos/versions.tf b/tf/deployment/prod/htz-fsn1/talos/versions.tf index d709a802..ecd61681 100644 --- a/tf/deployment/prod/htz-fsn1/talos/versions.tf +++ b/tf/deployment/prod/htz-fsn1/talos/versions.tf @@ -6,17 +6,6 @@ terraform { source = "siderolabs/talos" version = "~> 0.11" } - # Hetzner Cloud — the 3 control-plane VMs, their private subnet (etcd), and the - # public API load balancer. Token via HCLOUD_TOKEN (op run --env-file). - hcloud = { - source = "hetznercloud/hcloud" - version = "~> 1.51" - } - # Hostname picks for the talos nodes (node-names module → random_shuffle). - random = { - source = "hashicorp/random" - version = "~> 3.6" - } # Cilium install (CNI) post-bootstrap, in the same apply. helm = { source = "hashicorp/helm" @@ -33,11 +22,5 @@ terraform { source = "1Password/onepassword" version = "~> 2.1" } - # Worker NetBird peer lookups — the mesh addresses the CP apiserver dials for - # worker kubelets (see cp_extras_patch). Auth via NB_PAT (op run). - netbird = { - source = "registry.terraform.io/futo-org/netbird" - version = "1.0.2" - } } } diff --git a/tf/deployment/prod/htz-fsn1/talos/workers.tf b/tf/deployment/prod/htz-fsn1/talos/workers.tf index a2bb90ea..44f9d1af 100644 --- a/tf/deployment/prod/htz-fsn1/talos/workers.tf +++ b/tf/deployment/prod/htz-fsn1/talos/workers.tf @@ -1,18 +1,16 @@ # ── Bare-metal workers ─────────────────────────────────────────────────────── # Applied over apid to nodes already in Talos maintenance mode at their fabric_ip -# (reachable from the TF runner via NetBird → mgmt → fabric `kube` net). After the -# install+reboot they keep that fabric IP and join the NetBird mesh. +# or maint_ip (see below). After the install+reboot they keep that fabric IP and +# join the NetBird mesh (operator/backup plane only). # -# bond0 (2×25G LACP) → vlan 10 (kube) = fabric_ip — nodeIP + worker east-west (50G) -# default route via the kube IRB gateway (fabric transit) for egress +# bond0 (2×25G LACP) → vlan 10 (kube) = fabric_ip — nodeIP + worker east-west (50G) +# route to kube-cp (apiserver + VIP) via the kube IRB (10.40.10.1) — the fabric +# path to the control plane; the old wt0 (NetBird) route is retired +# default route via the Hetzner public NIC (DHCP) for egress # -# Workers are NOT NetBird peers: nodeIP = fabric_ip, so Cilium autoDirectNodeRoutes -# routes pod east-west directly over the shared VLAN-10 L2 (no BGP). The apiserver -# reaches worker kubelets via the CPs' NetBird route to the kube net (mgmt routers). -# -# Workers are PROVISIONED to maintenance mode out of band — see the Phase-4 runbook -# (./README.md): Hetzner rescue → dd the Talos metal image → bring up bond0.10 at -# fabric_ip. This stack assumes they're already there. +# Workers are PROVISIONED to maintenance mode out of band — see the runbook +# (./README.md): Hetzner rescue → dd the Talos metal image → reboot. This stack +# assumes they're already there. locals { kube_prefix = split("/", local.kube_cidr)[1] # 24 @@ -23,7 +21,7 @@ locals { yamlencode({ machine = { install = { diskSelector = { serial = w.install_serial } - image = local.worker_install_image + image = local.install_image } } }), yamlencode({ @@ -47,9 +45,14 @@ locals { vlans = [{ vlanId = module.addr_site.kube_vlan_id # 10 addresses = ["${w.fabric_ip}/${local.kube_prefix}"] - routes = var.cluster.worker_default_route_via_fabric ? [ - { network = "0.0.0.0/0", gateway = local.kube_gateway }, - ] : [] + # kube-cp (apiserver + API VIP) lives one IRB away — pin the route so + # kubelet→apiserver + geneve to the CPs ride the fabric, not the mesh. + routes = concat( + [{ network = local.kube_cp_cidr, gateway = local.kube_gateway }], + var.cluster.worker_default_route_via_fabric ? [ + { network = "0.0.0.0/0", gateway = local.kube_gateway }, + ] : [], + ) }] }] } diff --git a/tf/shared/modules/core-fabric/chassis.tf b/tf/shared/modules/core-fabric/chassis.tf index 1a739aab..acdcc54f 100644 --- a/tf/shared/modules/core-fabric/chassis.tf +++ b/tf/shared/modules/core-fabric/chassis.tf @@ -1,6 +1,6 @@ -# Preprovisioned spine VC + the 100G->4x25G breakout. Breakout (channel-speed) -# has no typed jeremmfr resource, so it's pushed as raw set-config. The -# `aggregated-devices ethernet device-count` line is auto-managed by +# Preprovisioned spine VC + the per-port breakout channelization. Breakout +# (channel-speed) has no typed jeremmfr resource, so it's pushed as raw set-config. +# The `aggregated-devices ethernet device-count` line is auto-managed by # junos_interface_physical (computed from the ae interfaces) — not set here. resource "junos_virtual_chassis" "spine" { preprovisioned = true @@ -19,8 +19,8 @@ resource "junos_null_load_config" "breakout" { action = "set" config = join("\n", flatten([ for fpc in [0, 1] : [ - for p in var.breakout_ports : - "set chassis fpc ${fpc} pic 0 port ${p} channel-speed ${var.breakout_speed}" + for p, speed in var.breakout_ports : + "set chassis fpc ${fpc} pic 0 port ${p} channel-speed ${speed}" ] ])) } diff --git a/tf/shared/modules/core-fabric/interfaces.tf b/tf/shared/modules/core-fabric/interfaces.tf index 623cdf19..696c8bc0 100644 --- a/tf/shared/modules/core-fabric/interfaces.tf +++ b/tf/shared/modules/core-fabric/interfaces.tf @@ -74,6 +74,35 @@ resource "junos_interface_physical" "node_lag" { vlan_members = ["vlan${var.kube_vlan_id}"] } +# Control-plane node bonds — same pattern as node_lags, but the trunk carries the +# kube-cp VLAN (the CPs' only fabric presence; kube↔kube-cp routes via the IRBs). +locals { + cp_node_lag_members = merge([for ae, ports in var.cp_node_lags : { for p in ports : p => ae }]...) +} + +resource "junos_interface_physical" "cp_node_lag_member" { + for_each = local.cp_node_lag_members + name = each.key + ether_opts { + ae_8023ad = each.value + } +} + +resource "junos_interface_physical" "cp_node_lag" { + for_each = var.cp_node_lags + name = each.key + mtu = 9216 + parent_ether_opts { + lacp { + mode = "active" + } + } + trunk = true + vlan_members = ["vlan${var.kube_cp.vlan_id}"] + + depends_on = [junos_vlan.this] +} + # Management-node ports (mgmt-1, mgmt-2) — one channelized port-3 leg per VC member, # each a single-port trunk of the stretched VLANs. Identical config per node. resource "junos_interface_physical" "mgmt_node" { diff --git a/tf/shared/modules/core-fabric/variables.tf b/tf/shared/modules/core-fabric/variables.tf index ff8e1aef..cd3d0cbd 100644 --- a/tf/shared/modules/core-fabric/variables.tf +++ b/tf/shared/modules/core-fabric/variables.tf @@ -27,15 +27,14 @@ variable "vc_member_serials" { } variable "breakout_ports" { - type = list(number) - default = [0, 1, 2, 3] - description = "QSFP28 ports channelized 100G->4x25G on each VC member." -} - -variable "breakout_speed" { - type = string - default = "25g" - description = "Per-channel speed for the breakout ports." + type = map(string) + default = { 0 = "25g", 1 = "25g", 2 = "25g", 3 = "25g" } + description = <<-EOT + QSFP28 ports channelized on each VC member: port number -> per-channel speed. + 25g -> et-/0/:0..3 legs; 10g -> xe-/0/:0..3. NB: 10g + channelization needs a QSFP+ (40G-class) breakout cable — the QFX5200 silently + falls back to unchannelized 100G on a QSFP28 cable. + EOT } variable "kube_vlan_id" { @@ -72,6 +71,31 @@ variable "node_lags" { EOT } +variable "cp_node_lags" { + type = map(list(string)) + default = {} + description = <<-EOT + Control-plane node LACP bonds terminated on the core — same shape and rules as + node_lags (key = ae name; value = the two member sub-ports, one per VC member, + cabled to the SAME node), but the trunk carries the kube-cp VLAN instead of + kube. Requires var.kube_cp. Members are 10G breakout legs (xe-…). + EOT +} + +variable "kube_cp" { + type = object({ + vlan_id = number + cidr = string + }) + default = null + description = <<-EOT + Kubernetes control-plane network on the fabric: creates the kube-cp VLAN + its + IRB (.1) on the spine — the second spine IRB, making the spine the router + between kube (workers) and kube-cp (bare-metal CPs: etcd + the API VIP). + null = no kube-cp VLAN. + EOT +} + variable "node_bgp" { type = object({ peer_range = string # the kube CIDR — nodes dynamic-peer from it; the IRB is its .1 diff --git a/tf/shared/modules/core-fabric/vlans.tf b/tf/shared/modules/core-fabric/vlans.tf index 5bfb1390..dba8016f 100644 --- a/tf/shared/modules/core-fabric/vlans.tf +++ b/tf/shared/modules/core-fabric/vlans.tf @@ -1,14 +1,17 @@ -# Stretched VLANs — L2 only on the spine EXCEPT the kube VLAN, which gets an IRB when -# node_bgp is set (the spine's only L3 interface, = the Cilium iBGP peer + VLAN-10 -# gateway; see bgp-nodes.tf). Other gateways live on the leaves. +# Stretched VLANs — L2 only on the spine EXCEPT the kube VLAN (IRB when node_bgp +# is set: the Cilium iBGP peer + VLAN-10 gateway, see bgp-nodes.tf) and the +# kube-cp VLAN (IRB when kube_cp is set: the CPs' gateway — the spine routes +# kube↔kube-cp). Other gateways live on the leaves. locals { - spine_vlans = { + spine_vlans = merge({ "vlan${var.public_vlan_id}" = { id = var.public_vlan_id, l3 = null } "vlan${var.private_vlan_id}" = { id = var.private_vlan_id, l3 = null } "vlan${var.kube_vlan_id}" = { id = var.kube_vlan_id, l3 = var.node_bgp == null ? null : "irb.${var.kube_vlan_id}" } "vlan${var.mgmt_vlan_id}" = { id = var.mgmt_vlan_id, l3 = null } "vlan${var.host_mgmt_vlan_id}" = { id = var.host_mgmt_vlan_id, l3 = null } - } + }, var.kube_cp == null ? {} : { + "vlan${var.kube_cp.vlan_id}" = { id = var.kube_cp.vlan_id, l3 = "irb.${var.kube_cp.vlan_id}" } + }) } resource "junos_vlan" "this" { @@ -17,3 +20,13 @@ resource "junos_vlan" "this" { vlan_id = tostring(each.value.id) l3_interface = each.value.l3 } + +# kube-cp IRB — the spine is the kube-cp gateway (.1). Bare-metal CPs sit on this +# VLAN only; worker↔CP (kubelet↔apiserver, geneve) routes irb.↔irb.. +resource "junos_interface_logical" "kube_cp_irb" { + count = var.kube_cp == null ? 0 : 1 + name = "irb.${var.kube_cp.vlan_id}" + family_inet { + address { cidr_ip = "${cidrhost(var.kube_cp.cidr, 1)}/${split("/", var.kube_cp.cidr)[1]}" } + } +} diff --git a/tf/shared/modules/fabric-addressing/main.tf b/tf/shared/modules/fabric-addressing/main.tf index 2e16867d..9b615e9b 100644 --- a/tf/shared/modules/fabric-addressing/main.tf +++ b/tf/shared/modules/fabric-addressing/main.tf @@ -11,16 +11,15 @@ locals { kube_cidr = "10.${var.site_id}.${var.kube_octet}.0/24" kube_vlan_id = var.kube_octet - # Site-global Kubernetes control-plane subnet ("kube-cp"). NOT a Juniper fabric - # VLAN — it's a small, isolated Hetzner Cloud private subnet holding ONLY the - # cloud control-plane VMs (for etcd CP↔CP) + the Kubernetes API LB's private IP. - # The workers do NOT join it: CP↔worker control + worker→API ride the NetBird - # WireGuard mesh (node IPs are NetBird addresses), and worker↔worker east-west - # rides the `kube` fabric net at 50G. Carved from the site supernet only for - # collision-free IPAM; it is NEVER configured on the Junos switches. - # kube-cp 10...0/24 (Hetzner Cloud subnet, gw .1) + # Site-global Kubernetes control-plane VLAN ("kube-cp") — a fabric VLAN like + # `kube`: holds the bare-metal control-plane nodes (etcd CP↔CP + apiserver) and + # the cluster's API VIP. Workers do NOT join it — the spine routes kube↔kube-cp + # via its two IRBs. (Historically this was an isolated Hetzner Cloud private + # subnet for the cloud CP VMs + API LB; same CIDR, now on the switches.) + # kube-cp 10...0/24 -> vlan (gw .1 = spine IRB) kube_cp_cidr = "10.${var.site_id}.${var.kube_cp_octet}.0/24" - kube_cp_gateway = cidrhost(local.kube_cp_cidr, 1) # .1 — Hetzner Cloud Gateway + kube_cp_vlan_id = var.kube_cp_octet + kube_cp_gateway = cidrhost(local.kube_cp_cidr, 1) # .1 — spine IRB # Internal (NetBird-only) Kubernetes LoadBalancer VIP range. Like kube-cp it is # NEVER a switch VLAN: Cilium assigns VIPs from it and the workers advertise the diff --git a/tf/shared/modules/fabric-addressing/outputs.tf b/tf/shared/modules/fabric-addressing/outputs.tf index 2a5c71f1..bd3f4e20 100644 --- a/tf/shared/modules/fabric-addressing/outputs.tf +++ b/tf/shared/modules/fabric-addressing/outputs.tf @@ -90,12 +90,17 @@ output "kube_vlan_id" { output "kube_cp_cidr" { value = local.kube_cp_cidr - description = "Site-global Kubernetes control-plane subnet 'kube-cp' (10...0/24) — an isolated Hetzner Cloud private subnet for the CP VMs (etcd) + the API LB. NOT a fabric VLAN." + description = "Site-global Kubernetes control-plane network 'kube-cp' (10...0/24), a fabric VLAN — bare-metal CPs (etcd) + the API VIP." +} + +output "kube_cp_vlan_id" { + value = local.kube_cp_vlan_id + description = "Site-global kube-cp VLAN id (== kube_cp_octet, e.g. 11)." } output "kube_cp_gateway" { value = local.kube_cp_gateway - description = "Hetzner Cloud Gateway (.1) for the kube-cp subnet." + description = "Spine IRB gateway (.1) for the kube-cp network." } output "lb_internal_cidr" { diff --git a/tf/shared/modules/fabric-netbox/ipam.tf b/tf/shared/modules/fabric-netbox/ipam.tf index c522b2de..0fed7c65 100644 --- a/tf/shared/modules/fabric-netbox/ipam.tf +++ b/tf/shared/modules/fabric-netbox/ipam.tf @@ -62,10 +62,12 @@ resource "netbox_prefix" "network" { } resource "netbox_ip_address" "gateway" { - for_each = local.networks - ip_address = "${each.value.gateway}/${split("/", each.value.prefix)[1]}" - status = "active" - dns_name = "gw-${var.site.code}-C${split("-", each.key)[0]}-${lower(each.value.role)}" + for_each = local.networks + ip_address = "${each.value.gateway}/${split("/", each.value.prefix)[1]}" + status = "active" + # lower(): NetBox normalizes dns_name to lowercase on write — mixed case here + # is a perpetual plan diff. + dns_name = lower("gw-${var.site.code}-c${split("-", each.key)[0]}-${each.value.role}") description = "IRB gateway for ${var.site.code}-C${split("-", each.key)[0]}-${each.value.role}" }