feat(prod): continue prod (#267)

This commit is contained in:
Antoine Lecompte
2026-07-16 09:56:40 -04:00
committed by GitHub
parent 0357aac743
commit 1f9b84bb1d
55 changed files with 806 additions and 925 deletions
-6
View File
@@ -26,12 +26,6 @@ opentofu = "1.11.5"
terragrunt = "0.99.4"
# ansible/mgmt convergence (mgmt:ansible task, run from CI on prod apply).
"pipx:ansible-core" = "2.18.1"
# Hetzner Cloud — build/upload the Talos hcloud snapshot + manage images
# (hetzner:talos-image). hcloud-upload-image spins a temporary rescue server,
# dd's the factory raw image, and snapshots it (hcloud can't boot the Talos ISO).
hcloud = "1.66.0"
"github:apricote/hcloud-upload-image" = "1.5.0"
[tasks.dev]
description = "Start all services in development mode"
depends = ["install:deps", "common:build", "docker:start"]
-76
View File
@@ -1,76 +0,0 @@
#!/usr/bin/env bash
#MISE description="Build + upload the Talos hcloud snapshot for the prod cluster. The schematic is TF-managed (talos_image_factory_schematic); this reads its image URL from tofu output. Idempotent unless FORCE=1."
# hcloud can't boot the Talos ISO, so the CP VMs need a Talos *snapshot*.
# hcloud-upload-image builds it the only way possible: spin a temporary rescue
# server, dd the factory hcloud-amd64 raw image, snapshot, tear down. The talos
# stack's data.hcloud_image then resolves it by label.
#
# mise run hetzner:talos-image # build for the prod father cluster
# FORCE=1 mise run hetzner:talos-image # rebuild even if a snapshot exists
set -euo pipefail
ROOT=$(git rev-parse --show-toplevel)
STACK="${TALOS_STACK:-tf/deployment/prod/htz-fsn1/talos}"
TFVARS="$ROOT/$STACK/clusters.auto.tfvars"
[ -f "$TFVARS" ] || { echo "talos-image: tfvars not found: $TFVARS" >&2; exit 1; }
# OVH S3 backend cert verification on macOS (same shim as the infra:* tasks).
if [ -z "${SSL_CERT_FILE:-}" ] && [ "$(uname -s)" = "Darwin" ] && [ -f /etc/ssl/cert.pem ]; then
export SSL_CERT_FILE=/etc/ssl/cert.pem AWS_CA_BUNDLE=/etc/ssl/cert.pem
fi
val() { grep -E "^[[:space:]]*$1[[:space:]]*=" "$TFVARS" | head -1 | sed -E 's/[^"]*"([^"]+)".*/\1/'; }
NAME=$(val name); VERSION=$(val talos_version); LOCATION=$(val cp_location)
SELECTOR="os=talos,cluster=${NAME},arch=amd64,version=${VERSION}"
tg() { OP_ENV_FILE=tf/.env.prod "$ROOT/tf/op-run.sh" terragrunt --working-dir "$STACK" "$@"; }
# The schematic is TF-managed (talos_image_factory_schematic). Pull its id straight
# from state (grep the exact `id =` line — robust against terragrunt log noise);
# if it isn't applied yet, create just it (targeted, dependency-free — touches no
# servers/data sources) and re-read. Then build the factory URLs from the id.
read_schematic_id() {
tg state show -no-color talos_image_factory_schematic.this 2>/dev/null \
| grep -E '^[[:space:]]+id[[:space:]]+=' | head -1 | sed -E 's/.*"([^"]+)".*/\1/'
}
ID=$(read_schematic_id || true)
if [ -z "$ID" ]; then
echo "talos-image: registering the Image Factory schematic in TF (targeted apply)…"
tg apply -target=talos_image_factory_schematic.this -auto-approve
ID=$(read_schematic_id)
fi
[ -n "$ID" ] || { echo "talos-image: could not resolve the schematic id from $STACK" >&2; exit 1; }
URL="https://factory.talos.dev/image/${ID}/v${VERSION}/hcloud-amd64.raw.xz"
METAL_URL="https://factory.talos.dev/image/${ID}/v${VERSION}/metal-amd64.raw.xz"
# Read-only hcloud token from 1Password (CI: OP_SERVICE_ACCOUNT_TOKEN; dev:
# interactive team-futo sign-in — same pattern as infra:plan).
if [ -z "${HCLOUD_TOKEN:-}" ]; then
if [ -n "${OP_SERVICE_ACCOUNT_TOKEN:-}" ]; then
HCLOUD_TOKEN=$(op read "op://yucca_tf_prod/HCLOUD_API_TOKEN/password")
else
HCLOUD_TOKEN=$(op read --account "${OP_ACCOUNT:-team-futo}" "op://yucca_tf_prod/HCLOUD_API_TOKEN/password")
fi
export HCLOUD_TOKEN
fi
echo "cluster=$NAME talos=v$VERSION schematic=$ID"
echo " hcloud image : $URL"
echo " metal image : $METAL_URL (workers dd this in rescue)"
# Idempotency: skip when a matching snapshot already exists.
EXISTING=$(hcloud image list --type snapshot --selector "$SELECTOR" \
--output noheader --output columns=id 2>/dev/null || true)
if [ -n "$EXISTING" ] && [ -z "${FORCE:-}" ]; then
echo "talos-image: snapshot already exists (id ${EXISTING}). Set FORCE=1 to rebuild."
exit 0
fi
echo "talos-image: uploading the Talos v$VERSION hcloud snapshot (spins a temporary server)…"
hcloud-upload-image upload \
--image-url "$URL" \
--architecture x86 \
--compression xz \
--location "$LOCATION" \
--labels "$SELECTOR"
echo "talos-image: done — data.hcloud_image.talos (selector os=talos,cluster=${NAME},arch=amd64) now resolves."
@@ -86,7 +86,8 @@ kind: CiliumBGPClusterConfig
metadata:
name: father
spec:
# Workers only — the CPs live on the hcloud net, not the fabric, so they can't peer.
# Workers only — the CPs live on the kube-cp VLAN; the spine's iBGP group peers
# from the kube VLAN (10.40.10.0/24) only.
nodeSelector:
matchExpressions:
- key: node-role.kubernetes.io/control-plane
@@ -0,0 +1,16 @@
---
# prod@htz-fsn1 INFRA layer — the CRD/operator providers (cert-manager,
# envoy-gateway + Gateway API, OpenEBS/mayastor) + their namespaces. Applied by
# the `cluster-infra` Flux Kustomization (clusters/prod/htz-fsn1/apps.yaml),
# which `cluster-apps` dependsOn: everything here must be READY (healthChecks →
# operators running → CRDs registered) before the app layer's custom resources
# (Certificates, Gateways, EnvoyProxy, DiskPools) are even dry-run — the flat
# single-layer tree deadlocked on exactly that during the 2026-07 rebuild.
apiVersion: kustomize.config.k8s.io/v1beta1
kind: Kustomization
resources:
- ./namespace-envoy-system.yaml
- ./namespace-openebs.yaml
- ./cert-manager.yaml
- ./envoy.yaml
- ./openebs.yaml
@@ -1,13 +1,15 @@
---
# prod@htz-fsn1 cluster overlay — father.
# prod@htz-fsn1 cluster overlay — father: the APP layer (custom resources +
# workloads). The CRD/operator providers live in ./infra, applied by the
# `cluster-infra` Flux Kustomization that this layer dependsOn — see
# infra/kustomization.yaml for why (fresh-cluster dry-run deadlock).
#
# CURRENT SCOPE: Flux owns the cluster BASELINE (coredns, the Cilium BGP LB
# config, the netops stack, OpenEBS pools) — adopted from the hand-applied
# bring-up state. The platform/infra components and the yucca app set are
# DELIBERATELY not enabled yet:
# config, the netops stack, OpenEBS pools). The platform/infra components and
# the yucca app set are DELIBERATELY not enabled yet:
# - components/infra needs real cluster-settings (RGW endpoint, o11y vmauth)
# and would collide with the in-cluster OpenEBS install (openebs/ here owns
# it via HelmRelease instead).
# and would collide with the in-cluster OpenEBS install (infra/openebs.yaml
# owns it via HelmRelease instead).
# - components/roles/primary is the yucca WORKLOAD set — explicitly held back
# until prod launch.
# Re-enable by uncommenting `components:` below.
@@ -18,10 +20,7 @@ resources:
- ./coredns.yaml
- ./cilium-bgp.yaml
- ./lb-return-route.yaml
- ./cert-manager.yaml
- ./namespace-envoy-system.yaml
- ./envoy.yaml
- ./openebs
- ./diskpools.yaml
- ./netops
# components:
# - ../../../components/infra
@@ -26,7 +26,7 @@ spec:
app: lb-return-route
spec:
hostNetwork: true
# Workers only — the CPs are on the hcloud net, not the fabric.
# Workers only — the CPs are on the kube-cp VLAN, not the kube VLAN.
affinity:
nodeAffinity:
requiredDuringSchedulingIgnoredDuringExecution:
@@ -3,9 +3,8 @@
# http://lg.father.fsn.htz.yucca.futo.network
#
# devices.yaml is NOT here: it embeds the netops password (netmiko can't key-auth
# through hyperglass config), so it's a Secret rendered at deploy time from
# op://yucca_tf_prod/NETOPS_FABRIC_PASSWORD (see the deploy script / runbook):
# kubectl -n netops create secret generic hyperglass-devices --from-file=devices.yaml
# through hyperglass config), so it's a Secret the talos stack renders from
# op://yucca_tf_prod/NETOPS_FABRIC_PASSWORD (netops-secrets.tf).
apiVersion: v1
kind: ConfigMap
metadata:
@@ -1,11 +1,11 @@
---
# The father netops stack. Secrets (netops-ssh, grafana-admin, hyperglass-devices)
# are NOT in git — they are created from 1Password (yucca_tf_prod: NETOPS_FABRIC_*,
# FATHER_GRAFANA_ADMIN) at bring-up; Flux only manages the workloads around them.
# The father netops stack. The namespace + its Secrets (netops-ssh,
# grafana-admin, hyperglass-devices) are NOT in git — the talos stack provisions
# them from 1Password (tf/deployment/prod/htz-fsn1/talos/netops-secrets.tf);
# Flux only manages the workloads around them.
apiVersion: kustomize.config.k8s.io/v1beta1
kind: Kustomization
resources:
- ./namespace.yaml
- ./networkpolicies.yaml
- ./junos-exporter.yaml
- ./victoria-metrics.yaml
@@ -1,19 +0,0 @@
# netops — in-cluster network operations stack for the htz-fsn1 fabric:
# junos_exporter + vmagent + VictoriaMetrics (30d high-granularity buffer) +
# Grafana, plus hyperglass/smokeping/oxidized. Everything is exposed ONLY on
# internal LoadBalancer VIPs (lb-internal pool, NetBird-reachable — never public).
# Applied by hand today; to be adopted by Flux when GitOps lands on father.
#
# privileged PodSecurity: VictoriaMetrics persists to a hostPath (no CSI on
# father yet — the NVMe data disks are unprovisioned); baseline forbids hostPath.
apiVersion: v1
kind: Namespace
metadata:
name: netops
labels:
pod-security.kubernetes.io/enforce: privileged
annotations:
# Never prune: this namespace holds hand-created secrets (netops-ssh,
# grafana-admin, hyperglass-devices) and the VictoriaMetrics PVC — a prune
# would destroy state only a manual runbook can restore.
kustomize.toolkit.fluxcd.io/prune: disabled
@@ -1,7 +0,0 @@
---
apiVersion: kustomize.config.k8s.io/v1beta1
kind: Kustomization
resources:
- ./namespace.yaml
- ./openebs.yaml
- ./diskpools.yaml
+55 -9
View File
@@ -1,10 +1,55 @@
# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/kustomize.toolkit.fluxcd.io/kustomization_v1.json
---
# cluster-apps entry point for prod@htz-fsn1. Authored ahead of the cluster (the
# prod Talos/flux stack isn't built yet); activates once that stack provisions
# flux and it syncs kubernetes/clusters/prod/htz-fsn1. Precedence (last wins):
# cluster-settings-generated (TF) -> cluster-settings (human) -> image-versions
# (the committed, CI-promoted prod tag); keys are disjoint by design.
# prod@htz-fsn1 entry points — TWO layers, so a fresh cluster can't deadlock:
# the 2026-07 rebuild proved a flat tree wedges itself (kustomize-controller
# server-side dry-runs every object, so any CR whose CRD is missing blocks the
# WHOLE apply — including the HelmReleases that would install those CRDs).
#
# cluster-infra operators/CRD providers (apps/prod/htz-fsn1/infra) — wait: true,
# so Ready ⇒ operators running ⇒ CRDs registered.
# cluster-apps everything else (CRs + workloads) — dependsOn cluster-infra.
#
# Substitution precedence (last wins): cluster-settings-generated (TF) ->
# cluster-settings (human) -> image-versions (the committed, CI-promoted prod
# tag); keys are disjoint by design. Injected into the NESTED Kustomizations via
# the patches block (both layers' direct objects use no ${vars} themselves).
apiVersion: kustomize.toolkit.fluxcd.io/v1
kind: Kustomization
metadata:
name: cluster-infra
namespace: flux-system
spec:
interval: 1h
retryInterval: 2m
path: ./kubernetes/apps/prod/htz-fsn1/infra
prune: true
sourceRef:
kind: GitRepository
name: flux-system
namespace: flux-system
# The health gate cluster-apps depends on: every nested Kustomization here
# carries healthChecks on its HelmRelease, so wait covers operator readiness.
wait: true
timeout: 10m
patches:
- target:
group: kustomize.toolkit.fluxcd.io
kind: Kustomization
patch: |-
apiVersion: kustomize.toolkit.fluxcd.io/v1
kind: Kustomization
metadata:
name: _
spec:
postBuild:
substituteFrom:
- kind: ConfigMap
name: cluster-settings-generated
- kind: ConfigMap
name: cluster-settings
- kind: ConfigMap
name: image-versions
---
apiVersion: kustomize.toolkit.fluxcd.io/v1
kind: Kustomization
metadata:
@@ -12,11 +57,12 @@ metadata:
namespace: flux-system
spec:
interval: 1h
# Fast retry: this Kustomization applies CRs (Certificates, Gateways,
# DiskPools) whose CRDs its own children install — on fresh bootstrap the
# first apply races them, and without retryInterval a failure waits the
# full 1h interval.
# Fast retry: CRD registration can trail the infra layer's Ready by moments
# (mayastor's diskpool operator creates its CRD at startup) — without
# retryInterval a dry-run failure waits the full 1h interval.
retryInterval: 2m
dependsOn:
- name: cluster-infra
path: ./kubernetes/apps/prod/htz-fsn1
prune: true
sourceRef:
-5
View File
@@ -32,11 +32,6 @@ export HETZNER_ROBOT_PASSWORD=op://yucca_tf_prod/HETZNER_WEBSERVICE_API_PASSWORD
# one). The cloudflare provider reads CLOUDFLARE_API_TOKEN directly.
export CLOUDFLARE_API_TOKEN=op://yucca_tf_prod/CLOUDFLARE_API_TOKEN/password
# ── Hetzner Cloud API (prod/htz-fsn1/talos — hcloud provider) ─────────────────
# Per-project read/write token for the control-plane VMs, network, snapshot + LB.
# TODO(prod): create the hcloud project + token, store at the path below.
export HCLOUD_TOKEN=op://yucca_tf_prod/HCLOUD_API_TOKEN/password
# ── NetBird setup keys (prod/htz-fsn1/talos — node-level overlay) ─────────────
# Minted by the netbird stack. WORKER key (group: talos) — joins the bare-metal
# workers to the prod htz-fsn1 NetBird network.
+14 -11
View File
@@ -2,7 +2,8 @@
Manages the Falkenstein (site 40) switch fabric as code:
- **spine** (`corenetsw` VC) — shared site core: VC, 100G→4×25G breakout, VLAN stretch.
- **spine** (`corenetsw` VC) — shared site core: VC, per-port breakout (ports 1-3
100G→4×25G, port 0 100G→4×10G for the father CPs), VLAN stretch.
- **cls1** (`cls1netsw` VC) — ceph cluster 1's leaf pair: public/private VLANs, IRB
gateways, the `NO-CROSS-VLAN` filter, and the 48 server LAGs.
@@ -15,22 +16,24 @@ Each ceph cluster = one leaf pair; the spine is shared across clusters.
| site supernet | `10.<site>.0.0/16` | `10.40.0.0/16` |
| management (vme) | `10.<site>.5.0/24` | `10.40.5.0/24` (spine `.115`, leaf `.125`) |
| kube (VLAN) | `10.<site>.<kube_octet>.0/24` | `10.40.10.0/24` → vlan 10 |
| kube-cp (hcloud) | `10.<site>.<kube_cp_octet>.0/24` | `10.40.11.0/24` (gw `.1`, CP VMs etcd + API LB) |
| kube-cp (VLAN) | `10.<site>.<kube_cp_octet>.0/24` | `10.40.11.0/24` → vlan 11 (gw `.1` = spine IRB) |
| cluster `n` /20 | `10.<site>.<n*16>.0/20` | `10.40.16.0/20` |
| public (VLAN) | cluster /20, /23 idx 2 | `10.40.20.0/23` → vlan 20 |
| private (VLAN) | cluster /20, /23 idx 3 | `10.40.22.0/23` → vlan 22 |
| leaf vme | `.125 + (n-1)*10` | `.125` |
VLAN id == the network's third octet; gateway = `.1` (IRB on the leaf).
VLAN id == the network's third octet; gateway = `.1` (IRB on the leaf, except the
site-global kube/kube-cp VLANs, whose IRBs live on the spine).
> **`kube-cp` is not a fabric VLAN.** It's a small isolated **Hetzner Cloud private
> subnet** holding only the cloud control-plane VMs (etcd CP↔CP) + the API LB's
> private IP. CP↔worker control traffic and worker→API ride the **NetBird WireGuard
> mesh** (node IPs are NetBird addresses), and worker↔worker east-west rides the
> `kube` fabric net (`10.40.10.0/24`) at 50G via Cilium BGP. The API endpoint is a
> **Hetzner Cloud LB** (no L2 VIP — hcloud private nets are anti-spoofed/routed).
> `kube-cp` is carved from the site supernet only for collision-free IPAM and is
> **never** configured on the Junos switches.
> **`kube-cp` is the control-plane VLAN.** The father bare-metal CPs are its only
> members (etcd CP↔CP + apiserver + the Talos-elected API **VIP** `10.40.11.5`),
> hanging off the spine's port-0 **4×10G** breakout (ae4-6, one leg per VC member).
> The spine routes kube↔kube-cp between its two IRBs (`10.40.10.1` / `10.40.11.1`),
> which is how worker kubelets and the apiserver reach each other; worker↔worker
> east-west rides the `kube` VLAN at 50G. Operators reach the API over the NetBird
> kube-cp route (the CPs are the route peers). Historically kube-cp was an isolated
> Hetzner Cloud subnet for the retired cloud CP VMs + API LB — same CIDR, so the
> API DNS record and etcd addressing carried over unchanged.
## Layout
@@ -22,6 +22,11 @@ module "core" {
vc_member_serials = var.spine_vc_serials
# Port 0 carries the father control-plane breakout at 10G (the CPs' Intel 82599
# NICs are 10G-only; needs the QSFP+ 4x10G breakout cables — see cp_node_lags);
# ports 1-3 stay 25G (workers + mgmt + spares).
breakout_ports = { 0 = "10g", 1 = "25g", 2 = "25g", 3 = "25g" }
# father's bare-metal kube workers hang off the core (channelized 25G breakouts of
# port 2, one leg per VC member). Each ae bundles the two ports cabled to one node
# (pairs derived from LLDP — consecutive MACs on the node's dual-port Broadcom NIC):
@@ -32,6 +37,23 @@ module "core" {
ae3 = ["et-0/0/2:1", "et-1/0/2:1"]
}
# father's bare-metal control planes: port-0 breakout legs at 10G (xe-), one leg
# per VC member, trunking the kube-cp VLAN. Pairing VERIFIED 2026-07-15 via MAC
# learning against the maintenance-mode nodes (NIC port 1 → FPC 0, port 2 →
# FPC 1, same leg index on both members):
# ae4 = harlan …0a:fe:c8/ca ae5 = imelda …09:68:68/6a ae6 = roscoe …65:07:40/42
cp_node_lags = {
ae4 = ["xe-0/0/0:2", "xe-1/0/0:2"]
ae5 = ["xe-0/0/0:1", "xe-1/0/0:1"]
ae6 = ["xe-0/0/0:0", "xe-1/0/0:0"]
}
# kube-cp VLAN + its spine IRB (10.40.11.1) — the spine routes kube↔kube-cp.
kube_cp = {
vlan_id = module.addr_site.kube_cp_vlan_id
cidr = module.addr_site.kube_cp_cidr
}
# Cilium node iBGP for LoadBalancer VIPs — the spine gets its first IRB (the kube net's
# .1 gateway) and dynamic-peers the workers from the kube subnet, accepting the LB /32s
# they advertise (covered by the transit aggregate, so reachable north-south). The
@@ -54,6 +76,8 @@ module "core" {
interfaces = [
"et-0/0/2:1", "et-0/0/2:2", "et-0/0/2:3",
"et-1/0/2:1", "et-1/0/2:2", "et-1/0/2:3",
"xe-0/0/0:0", "xe-0/0/0:1", "xe-0/0/0:2",
"xe-1/0/0:0", "xe-1/0/0:1", "xe-1/0/0:2",
"et-0/0/3:0", "et-1/0/3:0",
"et-0/0/27",
"et-0/0/30", "et-0/0/31", "et-1/0/30", "et-1/0/31",
+3 -3
View File
@@ -14,8 +14,9 @@ module "netbox" {
# Site-global VLANs (present on every cluster).
global_vlans = {
MGMT = { vid = module.addr_site.mgmt_vlan_id, prefix = module.addr_site.mgmt_cidr }
KUBE = { vid = module.addr_site.kube_vlan_id, prefix = module.addr_site.kube_cidr }
MGMT = { vid = module.addr_site.mgmt_vlan_id, prefix = module.addr_site.mgmt_cidr }
KUBE = { vid = module.addr_site.kube_vlan_id, prefix = module.addr_site.kube_cidr }
"KUBE-CP" = { vid = module.addr_site.kube_cp_vlan_id, prefix = module.addr_site.kube_cp_cidr }
}
clusters = {
@@ -37,7 +38,6 @@ module "netbox" {
# Pod/service CIDRs mirror the talos stack (talos.tf locals); the public carves
# mirror the Cilium LB pools + node-egress + transit config in this stack.
extra_prefixes = {
kube_cp = { prefix = module.addr_site.kube_cp_cidr, description = "Hetzner Cloud kube-cp: father CP VMs (etcd) + private API LB — not a fabric VLAN" }
lb_internal = { prefix = module.addr_site.lb_internal_cidr, description = "father internal (NetBird-only) LoadBalancer VIPs — Cilium lb-internal pool, iBGP /32s to the spine" }
pods = { prefix = "10.250.0.0/17", description = "father pod CIDR (Cilium, geneve over the kube VLAN)", status = "container" }
services = { prefix = "10.250.128.0/17", description = "father service CIDR (ClusterIPs; kube-dns at .128.10)", status = "container" }
+10 -9
View File
@@ -96,16 +96,17 @@ resource "netbird_dns_record" "father_worker" {
ttl = 300
}
# API endpoint — round-robin over the 3 CP IPs. NOT the LB: hcloud LBs refuse
# traffic from their own targets (the CPs), so the endpoint resolves straight to
# the CPs (reachable over the yucca-fsn-father-kube-cp route).
# API endpoint — the Talos-elected VIP on the kube-cp VLAN (etcd parks it on a
# healthy CP, so the record only answers where an apiserver runs). Reachable over
# the yucca-fsn-father-kube-cp route. (Historically round-robin over the CP IPs —
# the retired hcloud LB refused traffic from its own targets.)
resource "netbird_dns_record" "father_kube_api" {
for_each = local.father_cps
zone_id = netbird_dns_zone.yucca_internal.id
name = local.father_kube_api_fqdn
type = "A"
content = each.value
ttl = 300
count = var.talos_discovery_enabled ? 1 : 0
zone_id = netbird_dns_zone.yucca_internal.id
name = local.father_kube_api_fqdn
type = "A"
content = local.talos_kube.api_vip
ttl = 300
}
output "kube_api_fqdn" {
@@ -20,8 +20,8 @@ groups = {
talos = { resource = true } # Talos cluster nodes → yucca-prod-htz-fsn1-talos
resources = { resource = true } # routed-subnet tag → yucca-prod-htz-fsn1-resources (Network resources tag in)
# CP-only subset of `talos` — the ROUTER peer group for the kube-cp network. Only
# the cloud CPs sit on the kube-cp hcloud subnet, so only they can route it; if the
# router were the whole `talos` group the bare-metal WORKERS (also `talos`) would be
# the CPs sit on the kube-cp VLAN, so only they can route it; if the router were
# the whole `talos` group the bare-metal WORKERS (also `talos`) would be
# treated as routers and never install the client route to kube-cp. resource = false:
# it's a routing peer group, not a yucca-reachable tag (the CPs are already reachable
# via `talos`). CPs join via the talos_cp setup key below (auto_groups tags them
+14 -17
View File
@@ -16,13 +16,9 @@ locals {
# group (flagged `resource = true` in netbird.auto.tfvars), so the module-
# generated yucca→resources policy governs access — and resources never appear
# as a policy source, so they can't reach each other.
# NB: the `kube-cp` network (the routed Hetzner Cloud Network for the cloud
# control-plane VMs) is deliberately NOT advertised here. The router peers are the
# mgmt nodes, which sit on the Juniper fabric and cannot reach the hcloud subnets.
# The CP plane is reached out-of-band via hcloud public IPs + the API LB's public
# frontend (both firewalled to the NetBird/operator ranges) — see the talos stack.
# TODO(prod): for a fully-private control plane, attach the mgmt nodes to the
# kube-cp vSwitch and add module.addr_site.kube_cp_cidr to this map.
# NB: the `kube-cp` VLAN is deliberately NOT in this map — it's routed by its
# own network below (via the CPs, the talos_cp group), keeping the API plane's
# mesh path independent of the mgmt routers.
routed = {
mgmt = { address = module.addr_site.mgmt_cidr, description = "OOB / vme management network" }
# Internal LB VIPs (Grafana + netops UIs): NetBird peer -> mgmt router -> spine
@@ -48,24 +44,25 @@ locals {
}
}
# father's cloud control-plane subnet (kube-cp), routed via the CPs ONLY (the
# talos_cp group — the CP-only subset of talos). They're the only peers on that
# hcloud subnet. Router must NOT be the whole `talos` group: the bare-metal workers
# are also `talos`, and a routing peer doesn't install a client route for its own
# network — so if the workers were routers they'd never get the kube-cp route (and
# their pods couldn't reach the apiserver). This is how NetBird peers (operators +
# workers) reach the private API LB (10.40.11.5) + the CPs. masquerade so return
# traffic is SNAT'd to the CP's kube-cp address.
# father's control-plane VLAN (kube-cp), routed via the CPs ONLY (the talos_cp
# group — the CP-only subset of talos). They're the only peers on that VLAN.
# Router must NOT be the whole `talos` group: the bare-metal workers are also
# `talos`, and a routing peer doesn't install a client route for its own
# network — so if the workers were routers they'd never get the kube-cp mesh
# route. (Worker→apiserver traffic itself rides the fabric — a static route via
# the spine IRB pinned in the machine config — not this mesh route.) This is
# how OPERATOR/CI peers reach the API VIP (10.40.11.5) + the CPs. masquerade so
# return traffic is SNAT'd to the CP's kube-cp address.
# CP membership comes from the talos_cp setup key (netbird.auto.tfvars, auto_groups
# [talos, talos_cp]); the talos stack joins CPs with it and workers with the plain
# `talos` key, so re-provisioning keeps the split.
"yucca-fsn-father-kube-cp" = {
description = "father control-plane subnet (kube-cp), routed via the CPs (talos_cp)."
description = "father control-plane VLAN (kube-cp), routed via the CPs (talos_cp)."
router = { peer_groups = ["talos_cp"], masquerade = true }
resources = {
kube_cp = {
address = module.addr_site.kube_cp_cidr
description = "kube-cp: CP VMs (etcd) + the private API LB (10.40.11.5)."
description = "kube-cp: bare-metal CPs (etcd) + the API VIP (10.40.11.5)."
groups = ["resources"]
}
}
-65
View File
@@ -64,50 +64,6 @@ provider "registry.opentofu.org/hashicorp/kubernetes" {
]
}
provider "registry.opentofu.org/hashicorp/random" {
version = "3.9.0"
constraints = "~> 3.6"
hashes = [
"h1:U8KXqGCoNI9/guYbTvzgdtVk3fRthoG0UXwm1JoEpIs=",
"zh:03f1114cc20b8913523735ab76e0f0a2b16ce13c92923a53304bf85f07fc0dbc",
"zh:105b678ee72322a3067f105d7e05e940f6143238f377f6e87ff4ec909246ac2a",
"zh:55f3bbf13ea18cbace61a706566a80f25f33fe2b1780b6f3d7b582af2a05b6d2",
"zh:63adf996db48f082f7a6351eb485e219cd88795fc71e6ec60a837263ab0d2cb1",
"zh:7e99550738a4e3cc68b8a467714b0d69371025fe95e3326d5323d026d55653e9",
"zh:8342b54af3a18a37e075eeae61be57f4de2ba71b35d95c5075d402dd2c1f289d",
"zh:83ee18e32ac9dd5fc91298554b7c4cfa4c3a1db50f4c797945637cc93c0844ae",
"zh:993ecc0adbf6bd535a59fbc9b735d8c33950e6f6eb5e621d750da9b71d65d80a",
"zh:ad722bc59d4edbf1415e827fc007c0efe6e0e9462d5568bae20b34be1058a261",
"zh:ae9448e1f87b2f9a6c5197a0e9862162ec6b137cb3a3835e11522995d8939e7c",
"zh:bc9cdd3aac784f759125c6627f6f6416e8726a1c184eb9cf3e55b9edbc94c627",
"zh:c8e35b89572ba1c40a9b20022e033a3395fb8d42e7604d50c900f193ba10382e",
"zh:e2deaa8a9975ef81d9f62baed12c41286918b0a10908e0e031f13f69a3b730a1",
"zh:ee39707557210a0ab1098aa357d2cdfe502e5a312d0dbdffb09d08facc4d3fc5",
"zh:f81afe4eb63e8aa9e0ea71be6c990f0dc69cb360e7191c0742a991f4a5081b64",
]
}
provider "registry.opentofu.org/hetznercloud/hcloud" {
version = "1.66.0"
constraints = "~> 1.51"
hashes = [
"h1:iVAGP8gRbZK0kJF7SiYJRt61wz0D5AF9q+WMsrAiBI0=",
"zh:1286cee6fb63dbcb18f53077bbb5e5d132a4e4d9f006af4e8d8edfc08d6bcdc8",
"zh:204460dacc044bda019a4a18b398e094289500c36913c7c9457f432adf31b8b2",
"zh:214175d50773481cbeaf9c9004e4121a3a1c9686c79424ebdc8ff189dd057d3e",
"zh:22b17bceff61cc13ad04a399ba87521356a3a134d4687273727473ae9eccf5f1",
"zh:368867dac5525c411de7e38f2e27de0a71854d1750867322ff2b9321128c88fb",
"zh:5289b75f8370bdbc4c6051d55cf33d0b1bd25dc6d71bfbd39b360249a37f1501",
"zh:81cb676aa50c5777df8fc80d4e69c9012330ae751f5e6f12bf6074bfd2e7c496",
"zh:ab08aead10643b21aa6b51af562b50492e12b9dd0ab7dca27a05aa63209b7d66",
"zh:af25c210d0570cf61ef767b2545bf9f3fb909178135f0e5e14bec0c1c9d07a63",
"zh:bcad66f4830c97118fa793723e53f8a4d27ddd34ea969ff259408842c2238331",
"zh:ce3ed323d75ae905d975925fa98c7054a7514c81276a485fc37da8232b53e39f",
"zh:d481bc0ef0c87ab1969c17777f526b2f59f823432d676145134c41a6d29bd98e",
"zh:ea7ef88df2c3ca154d86238920636d52a3c9066c7467543d3fa45f1e52ec2f7b",
]
}
provider "registry.opentofu.org/siderolabs/talos" {
version = "0.11.0"
constraints = "~> 0.11"
@@ -130,24 +86,3 @@ provider "registry.opentofu.org/siderolabs/talos" {
]
}
provider "registry.terraform.io/futo-org/netbird" {
version = "1.0.2"
constraints = "1.0.2"
hashes = [
"h1:CE91Uvc3FhpgpwgjVR96ueerThTZSCTcKF8Rp2+lzQw=",
"zh:1a1b727dd3971eb0f4c923c19fa95100e5b91b163ae0607f72a37d656f9d71a6",
"zh:5169993e54c6b184cdb359fb310477d85eeb99d1d91db14d367a29bf3119dc55",
"zh:573279b9532f916ee4bfb4e12faacf26e8e27df626ac37e1a5c42d5eab2350a4",
"zh:5cafd8cb073fc6774d959080c06b6faf821c3a7caaf3d876a40be8b8945a70fd",
"zh:66d40bbde6d756160a94e8c908582cb7806acccbebe0e9060b82cb4993f5adad",
"zh:6eb6493922345ec4306d8f0e95e799a6045ff3b25fd05973828e960757df5b97",
"zh:88c638c29d7dad339f1c4e90e54d0996be85a95574d5d9c5f125cb0bbb5b8902",
"zh:890df766e9b839623b1f0437355032a3c006226a6c200cd911e15ee1a9014e9f",
"zh:89cca24eeb5305cfcac51092bd0c330336af6613648f14a4c70f23a56fe4e93d",
"zh:91186b07d2ee1494bcd7e2c132ca1c61b62b2a1f2d325d2bbc4bfe6572ce327d",
"zh:aa6cbe8f5c8121d96861e293d3773cc210723fa410acf45512a337703d55d911",
"zh:d38dce0bcc61ca2482ee3b9f457ebf45929c7ebc8717487d9f1270aa3dcd3cfa",
"zh:dba9742f0a0cb5e190a10265ec00d20a4d266f8a6dd517cb4c1d8486335ed636",
"zh:e8eaabc072208c10ab554b4c89dbe32831a7d0125ab628af4a6530a02ea5a993",
]
}
+128 -64
View File
@@ -1,92 +1,156 @@
# prod/htz-fsn1/talos — the `father` cluster
Hybrid production Talos Kubernetes cluster: **3 Hetzner Cloud control-plane VMs +
3 Hetzner Robot bare-metal workers**, one cluster. Flux is **deferred** — this
stack brings up a healthy, Cilium-networked cluster and stops there.
All-bare-metal production Talos Kubernetes cluster: **3 Hetzner Robot control
planes + 3 Hetzner Robot workers**, every node plane on the Juniper fabric. The
previous hybrid topology (3 Hetzner Cloud CP VMs + hcloud API LB + NetBird as the
CP↔worker plane) is retired; NetBird remains on every node as the **operator /
backup plane** only.
## Topology
```
┌──────────── Hetzner Cloud (fsn1) ────────────┐
│ cp-1 cp-2 cp-3 (CCX23, Talos snapshot) │
│ • public IPv4 → NetBird + bootstrap + LB │
│ • kube-cp 10.40.11.0/24 (eth1) → etcd │
│ • API LB (lb11) → :6443, PRIVATE (lb_public=false) │
└───────┬───────────────────────┬───────────────┘
NetBird mesh│ (CP route to fabric │ private LB :6443 (api_dns_name,
via mgmt │ net, advertised) │ reached over the mesh)
┌───────┴───────────────────────┴───────────────┐
│ Juniper fabric — kube VLAN 10 / 10.40.10.0/24 │
│ wk-1 .11 wk-2 .12 wk-3 .13 (bond0, 50G) │
│ • pod↔pod rides geneve (routingMode: tunnel) over │
│ the 50G fabric between workers │
└─────────────────────────────────────────────────┘
┌──────────────── Juniper fabric (site 40) ─────────────────┐
│ kube-cp VLAN 11 — 10.40.11.0/24 (gw .1 = spine IRB) │
│ cp-harlan .11 cp-imelda .12 cp-roscoe .13 │
│ API VIP 10.40.11.5 (Talos etcd-elected) │
│ bond0 2×10G (ixgbe, spine port-0 4×10G breakout, ae4-6) │
├───────────────── spine routes irb.11 ↔ irb.10 ────────────┤
│ kube VLAN 10 — 10.40.10.0/24 (gw .1 = spine IRB) │
│ wk-jeanne .11 wk-sheron .12 wk-dianna .13 │
│ bond0 2×25G (bnxt_en, spine port-2 4×25G breakout, ae1-3)│
└───────────────────────────────────────────────────────────┘
every node: onboard public NIC (DHCP) = default route/egress
+ NetBird (operator plane; CPs route kube-cp)
```
| Plane | Path | Carries |
|---|---|---|
| node / control | NetBird (CP) → mgmt routers → `kube` fabric | apiserver↔kubelet, CP→worker |
| API endpoint | private Hetzner Cloud LB (`api_dns_name`, NetBird-reachable) | kubelet→apiserver, operators |
| etcd | `kube-cp` hcloud private subnet (`10.40.11.0/24`) | CP↔CP |
| pod east-west | `kube` fabric VLAN 10 (50G), geneve tunnel | worker↔worker pods |
| etcd + API endpoint | `kube-cp` VLAN 11, VIP `10.40.11.5` (`api_dns_name` → VIP) | CP↔CP etcd, apiserver, VIP |
| CP↔worker | routed `kube`↔`kube-cp` via the spine IRBs (static routes in machine config) | apiserver↔kubelet, geneve |
| pod east-west | `kube` VLAN 10 (50G), geneve tunnel | worker↔worker pods |
| operators / CI | NetBird → kube-cp route (the CPs are the route peers) | talosctl/kubectl/TF |
No vSwitch, no BGP. **Cilium BGP is reserved for north-south later** (advertising
ingress/LoadBalancer VIPs to the fabric leaf, which already speaks BGP).
**Cilium BGP** (north-south LoadBalancer VIPs) stays workers-only against the
spine's VLAN-10 IRB.
## Bring-up flow (single `tf:apply`)
1. `image.tf` — look up the Talos amd64 hcloud snapshot (built once out-of-band).
2. `network.tf` — hcloud network + the kube-cp cloud subnet.
3. `controlplane.tf` — 3 CP VMs (config via `user_data`) + the API LB.
4. `talos.tf` — `talos_machine_bootstrap` against cp-1's public IP → kubeconfig.
5. `workers.tf` — `talos_machine_configuration_apply` to each worker over apid
(its fabric IP, reached via NetBird→mgmt→fabric).
6. `cilium.tf` — Cilium via Helm → post-CNI health gate → Ready cluster.
1. `image.tf` — register the schematic (one, metal, **no qemu-guest-agent**).
2. `controlplane.tf` — apid config apply to each CP (maintenance `maint_ip` on the
first pass) → install to disk (by serial) + reboot onto bond0.11.
3. `talos.tf` — `talos_machine_bootstrap` against cp-1 (`10.40.11.11`, reached
over the NetBird kube-cp route once cp-1's netbird is up) → kubeconfig.
4. `workers.tf` — apid apply to each worker (maintenance `maint_ip` first pass).
5. `cilium.tf` — Cilium via Helm → post-CNI health gate.
6. `flux.tf` — flux-operator + instance → GitOps takes over
(`kubernetes/clusters/prod/htz-fsn1`).
## Prerequisites (before `tf:apply`)
- **`HCLOUD_TOKEN`** — create the hcloud project + read/write token, store at
`op://yucca_tf_prod/HCLOUD_API_TOKEN` (see `tf/.env.prod`).
- **Talos schematic** — the extension set lives in `schematic.yaml` and is
registered with the factory by TF (`talos_image_factory_schematic`, `image.tf`);
the schematic id + image URLs derive from it. Edit `schematic.yaml` to change it.
- **hcloud snapshot** — build it once with `mise run hetzner:talos-image` (reads the
image URL from `tofu output`; idempotent, `FORCE=1` to rebuild). `image.tf`
resolves it by label.
- **NetBird setup key** — minted by the netbird stack; path in `tf/.env.prod`.
- **Workers in maintenance mode** — provisioned to Talos maintenance at their
`fabric_ip` (10.40.10.11/.12/.13). See the runbook below.
- **DNS** — after apply, point `api_dns_name` (output) at the LB public IPv4
(output `api_dns_record`).
- **`trusted_cidrs`** — MUST include the source the TF runner dials the CP public
IPs from (CI egress / your NetBird range), or bootstrap (apid 50000) hangs.
- **Fabric** — the fabric stack applied with `breakout_ports` port 0 = 10g, the
kube-cp VLAN/IRB, and `cp_node_lags` ae4-6. Leg pairing in `../fabric/fabric.tf`
was verified 2026-07-15 by MAC-learning against the maintenance-mode nodes
(QSFP+ 4×10G breakout cables installed; all six legs link at 10G). Re-verify if
anything is re-cabled — LACP won't aggregate legs facing different nodes.
- **Nodes in maintenance mode** — every node with `provisioned = false` must be
in Talos maintenance at its `maint_ip`. Rescue → dd runbook below.
- **NetBird setup keys** — minted by the netbird stack; paths in `tf/.env.prod`.
- **Operator/CI NetBird networks selected** — the apply host reaches the CPs via
the `yucca-fsn-father-kube-cp` NetBird network (and the switches/workers via
`htz-fsn1-mgmt` / `htz-fsn1-kube`). With client ≥0.75 lazy network selection,
`netbird networks select <name>` or routes silently don't install.
- **`trusted_cidrs`** — MUST include the source the TF runner dials the CPs from
(the NetBird range), or bootstrap (apid 50000) hangs.
> CI owns `tf:apply` (`.github/workflows/infra.yml`). Locally use `tf:plan` only.
## Phase-4: worker provisioning runbook (rescue → Talos maintenance)
## Node provisioning runbook (rescue → Talos maintenance)
The workers (Robot server numbers 3008210/11/12) must boot Talos in maintenance
mode at their fabric IP before this stack applies. Per worker:
Per node (CPs: Robot 3027819/3027863/3028524; workers: 3008210/11/12):
1. Robot → enable the **rescue system** (linux64) for the server, reboot into it.
2. `dd` the Talos **metal** image for the cluster schematic onto the boot disk:
1. Robot → enable the **rescue system** (linux64), reboot into it.
2. `dd` the Talos **metal** image for the cluster schematic onto the install disk
(`tofu output talos_metal_image_url`):
```sh
wget -O /tmp/talos.raw.xz \
"https://factory.talos.dev/image/<schematic-id>/v1.13.4/metal-amd64.raw.xz"
xz -dc /tmp/talos.raw.xz | dd of=/dev/sda bs=4M && sync
wget -O /tmp/talos.raw.xz "$(tofu output -raw talos_metal_image_url)"
xz -dc /tmp/talos.raw.xz | dd of=/dev/sda bs=4M conv=fsync && sync
```
3. Reboot off the rescue system → Talos comes up in maintenance mode.
4. Bring up `bond0` over the two 25G NICs with the tagged **kube VLAN 10** carrying
the node's `fabric_ip` (matching `clusters.auto.tfvars`), so the TF runner can
reach apid (50000) over the fabric. (This stack then pins the same config.)
Record the disk's SERIAL (`lsblk -d -o NAME,SERIAL`) — it pins
`install_serial` in `clusters.auto.tfvars`.
3. Reboot off the rescue system → Talos comes up in maintenance mode on the
public NIC (DHCP) = the node's `maint_ip`. (If it lands back in rescue, the
rescue flag didn't clear — just reboot again.)
4. Set the node's `provisioned = false` in tfvars → apply → flip to `true` once
it has joined.
This mirrors the mgmt-host reprovision pattern (`../mgmt-hosts.yaml` + the fabric
stack's `mgmt.tf`); a future iteration can drive it from TF/Ansible.
## Cutover runbook (hybrid → all-bare-metal REBUILD) — EXECUTED 2026-07-15
The rebuild keeps `talos_machine_secrets` (cluster PKI) but re-bootstraps etcd on
the new CPs and re-installs the workers. **In-cluster state (Mayastor/localpv) is
lost**; Flux redeploys everything. The steps below were executed 2026-07-15 (all
stacks now plan clean); kept as the reference for any future rebuild. Order
matters:
1. **Fabric first** (CI orders fabric before talos): apply lands the kube-cp
VLAN/IRB + ae4-6. Port-0 10g channelization is already live on the spine
(set 2026-07-15, identical to the TF config) and the leg pairing is verified —
after the apply, `show lacp interfaces` should show ae4-6 collecting once the
CPs boot their bonds.
2. **New CPs in maintenance mode** (done 2026-07-15): rescue → dd → maintenance
at 178.63.124.20/.21/.22.
3. **State surgery** (forgets, no destroys — safe with prevent_destroy):
```sh
tofu state rm talos_machine_bootstrap.this # re-bootstrap on the new cp-1
tofu state rm talos_cluster_kubeconfig.this
tofu state rm helm_release.cilium helm_release.flux_operator helm_release.flux_instance
tofu state rm kubernetes_secret_v1.github_app kubernetes_namespace_v1.cert_manager kubernetes_secret_v1.cloudflare_api_token
```
4. **Reset the workers** to maintenance mode (wipes them — deliberate):
```sh
talosctl -n 10.40.10.11 reset --graceful=false --reboot \
--system-labels-to-wipe STATE --system-labels-to-wipe EPHEMERAL # × each worker
```
They come back in maintenance at their `maint_ip` (public DHCP).
5. **Merge/apply this stack**: destroys the hcloud CP VMs + LB + network +
firewall (their prevent_destroy left with the deleted config), applies CP
configs → bootstrap → workers → Cilium → Flux.
6. Flip every node's `provisioned = true` once joined; rotate operator
kubeconfigs (`op read`, secrets.tf rewrote them).
Rebuild gotchas hit on 2026-07-15 (expect them again):
- **The first apply fails partway** — helm dials the apiserver seconds after
bootstrap (connection refused) and the kubernetes/1P resources throw
"inconsistent final plan" (provider config unknowable at plan). Just re-apply;
nothing is damaged. A helm wait-timeout can strand a `failed` release
("cannot re-use a name that is still in use") — `helm uninstall` it first.
- **flux-operator waits on the first worker** (CPs are unschedulable) — worker
install+join takes longer than helm's 5m wait. Re-apply once workers are Ready.
- **CoreDNS chicken-and-egg**: `cluster.coreDNS.disabled=true` from t=0 means NO
cluster DNS until Flux deploys ours — but flux-operator needs DNS to fetch its
manifests from ghcr.io. Break the cycle once per rebuild:
`kubectl apply -f kubernetes/apps/prod/htz-fsn1/coredns.yaml` (the exact
objects Flux owns — it adopts them unchanged).
- ~~CRD deadlock (flat kustomization)~~ — fixed structurally after the rebuild:
the tree is layered (`cluster-infra` = operators/CRD providers with
`wait: true`; `cluster-apps` dependsOn it — see clusters/prod/htz-fsn1/
apps.yaml). A fresh cluster converges without manual CRD pre-installs; the
only remaining hand-step is the CoreDNS one above (it predates Flux itself).
- **Stale NetBird peers**: re-provisioned nodes join as NEW peers; the old
same-named peers linger disconnected and break the netbird stack's
`data.netbird_peer` lookups ("cannot match multiple peers"). Delete the
disconnected duplicates (API/console) before applying the netbird stack.
- **`talosctl reset --wait=false`** — the default wait can never complete (the
node comes back at a different IP, in maintenance mode).
- ~~Hand-created netops secrets die with the cluster~~ — fixed: the netops
namespace + netops-ssh/grafana-admin/hyperglass-devices Secrets are TF-owned
now (netops-secrets.tf, sourced from 1P), restored by the normal apply.
## Notes
- **Bootstrap is one-shot.** Re-applying does not re-bootstrap; tainting
`talos_machine_bootstrap.this` re-rolls cluster identity — don't.
- CP VMs have `ignore_changes = [user_data, image]` so re-applies don't recycle
live nodes; change them deliberately (cordon/drain first).
- Flux activates later from `kubernetes/clusters/prod/htz-fsn1` (already scaffolded).
- **Bootstrap is one-shot.** Re-applying does not re-bootstrap; replacing
`talos_machine_bootstrap.this` re-rolls cluster identity — don't (the state-rm
in the cutover is the deliberate exception).
- The API VIP is etcd-elected: it exists only while a healthy CP holds it. The
bootstrap/operator path deliberately dials cp-1's IP, not the VIP.
- The old hcloud snapshot build task (`hetzner:talos-image`) is retired — all
nodes boot the factory **metal** image via the rescue-dd runbook.
@@ -1,16 +1,23 @@
# Site IP plan — the single source of truth (same module the fabric + netbird
# stacks read). Gives us, derived from site_id (40):
# addr_site.kube_cidr 10.40.10.0/24 — fabric VLAN 10, worker east-west (50G)
# addr_site.kube_cp_cidr 10.40.11.0/24 — isolated hcloud subnet, CP etcd + API LB
# addr_site.kube_cp_cidr 10.40.11.0/24 — fabric VLAN 11, CP etcd + API VIP
# Nothing is hardcoded here; addresses below are cidrhost() offsets into these.
module "addr_site" {
source = "../../../../shared/modules/fabric-addressing"
site_id = var.site_id
}
# Cluster-1 view — only for the leaf vme address (netops-secrets.tf hyperglass).
module "addr_cls1" {
source = "../../../../shared/modules/fabric-addressing"
site_id = var.site_id
cluster_id = 1
}
locals {
kube_cidr = module.addr_site.kube_cidr # 10.40.10.0/24 (fabric)
kube_gateway = cidrhost(module.addr_site.kube_cidr, 1) # .1 IRB on the cluster leaf
kube_cp_cidr = module.addr_site.kube_cp_cidr # 10.40.11.0/24 (hcloud)
kube_cp_gw = module.addr_site.kube_cp_gateway # .1 Hetzner Cloud Gateway
kube_cidr = module.addr_site.kube_cidr # 10.40.10.0/24 (fabric VLAN 10)
kube_gateway = cidrhost(module.addr_site.kube_cidr, 1) # .1 IRB on the spine
kube_cp_cidr = module.addr_site.kube_cp_cidr # 10.40.11.0/24 (fabric VLAN 11)
kube_cp_gw = module.addr_site.kube_cp_gateway # .1 IRB on the spine
}
@@ -1,9 +1,9 @@
# Cilium for the hybrid prod cluster. Talos sets cni:none + proxy:disabled, so
# nodes stay NotReady until this lands the datapath.
# Cilium for the prod cluster. Talos sets cni:none + proxy:disabled, so nodes
# stay NotReady until this lands the datapath.
#
# TUNNEL (geneve) routing — see the routingMode block below for the full why:
# the cluster spans two L2 domains (CPs on kube-cp hcloud, workers on the fabric
# VLAN) bridged only by NetBird, which native routing/autoDirectNodeRoutes can't
# the cluster spans two L2 domains (CPs on the kube-cp VLAN, workers on the kube
# VLAN) routed by the spine IRBs, which autoDirectNodeRoutes (same-L2-only) can't
# span. Worker↔worker east-west still rides the 50G fabric, just encapsulated.
#
# securityContext + cgroup blocks are MANDATORY on Talos. Ref: Talos "Deploying Cilium".
@@ -12,17 +12,17 @@ ipam:
mode: kubernetes
# Tunnel (geneve) routing. The cluster spans two L2 domains — CPs on the kube-cp
# hcloud net, workers on the fabric VLAN — bridged only by NetBird. Native routing
# can't span that; a geneve overlay carries pod↔pod over whatever host-to-host path
# exists (the 50G fabric for worker↔worker, NetBird for CP↔worker). East-west still
# rides the fabric, just encapsulated (~50B overhead, negligible at 50G).
# VLAN (11), workers on the kube VLAN (10) — routed by the spine's IRBs.
# autoDirectNodeRoutes only works within one L2; a geneve overlay carries pod↔pod
# over the routed node-to-node path instead. East-west still rides the fabric,
# just encapsulated (~50B overhead, negligible at 50G).
routingMode: tunnel
tunnelProtocol: geneve
# Masquerade pod traffic to non-pod destinations (internet, node IPs); pod↔pod is
# tunnelled so it isn't masqueraded. ip-masq-agent excludes only the pod CIDR, so the
# geneve underlay + node-IP traffic egressing wt0 is SNAT'd to the node's NetBird
# address (which WireGuard accepts). Requires BPF masquerade on.
# tunnelled so it isn't masqueraded. ip-masq-agent excludes only the pod CIDR, so
# node-IP traffic egressing wt0 (NetBird, the operator plane) is still SNAT'd to
# the node's mesh address (which WireGuard accepts). Requires BPF masquerade on.
enableIPv4Masquerade: true
bpf:
masquerade: true
+5 -4
View File
@@ -1,7 +1,8 @@
# Cilium CNI — installed post-bootstrap in the same apply (helm provider bound to
# the bootstrap CP, providers.tf). Talos set cni:none + proxy:disabled, so nodes
# go Ready only once this lands the datapath. Native routing + autoDirectNodeRoutes
# keeps worker east-west on the 50G fabric (cilium-values.yaml.tftpl).
# go Ready only once this lands the datapath. Geneve tunnel routing spans the two
# routed fabric VLANs (kube/kube-cp); worker east-west still rides the 50G fabric
# (cilium-values.yaml.tftpl).
resource "helm_release" "cilium" {
name = "cilium"
namespace = "kube-system"
@@ -27,9 +28,9 @@ data "talos_cluster_health" "post_cni" {
count = var.bootstrap_health_gate ? 1 : 0
client_configuration = talos_machine_secrets.this.client_configuration
control_plane_nodes = local.cp_private_ips
control_plane_nodes = local.cp_ips
worker_nodes = [for w in var.cluster.workers : w.fabric_ip]
endpoints = local.cp_private_ips
endpoints = local.cp_ips
timeouts = { read = "10m" }
@@ -1,11 +1,12 @@
# ── prod hybrid Talos cluster: `father` ──────────────────────────────────────
# Topology (see README.md): 3 Hetzner Cloud CP VMs + 3 Hetzner Robot bare-metal
# workers. CP↔worker rides the NetBird mesh; worker↔worker east-west rides the
# 50G fabric (kube VLAN 10) via Cilium BGP; etcd CP↔CP on a private hcloud subnet;
# API via a public Hetzner Cloud LB.
# ── prod bare-metal Talos cluster: `father` ──────────────────────────────────
# Topology (see README.md): 3 Hetzner Robot bare-metal CPs on the kube-cp fabric
# VLAN 11 (etcd + API VIP 10.40.11.5) + 3 Hetzner Robot bare-metal workers on the
# kube fabric VLAN 10. The spine routes kube↔kube-cp between its IRBs; NetBird
# stays on every node as the operator/backup plane (kube-cp is routed to the mesh
# via the CPs). No Hetzner Cloud anywhere — the cloud CP VMs + API LB are retired.
#
# Adding/replacing a node = edit here + `tf:plan` (CI applies). Workers must
# already be in Talos maintenance mode at their fabric_ip (see Phase-4 runbook).
# Adding/replacing a node = edit here + `tf:plan` (CI applies). Nodes must
# already be in Talos maintenance mode at their maint_ip (see the runbook).
cluster = {
name = "father" # prod K8s cluster (Star Wars; staging = luke)
@@ -14,7 +15,6 @@ cluster = {
# The node extension set lives in schematic.yaml (managed via
# talos_image_factory_schematic in image.tf) — no schematic id to paste here.
install_disk = "/dev/sda" # CP install target (hcloud VMs: virtio /dev/sda) — the ONLY consumer is cp_install_patch; workers install by NVMe serial (workers.tf)
cilium_version = "1.19.5"
hubble = true
@@ -22,23 +22,23 @@ cluster = {
# NetBird peer address range for this deployment (firewall trust for the mesh).
netbird_node_cidr = "10.254.0.0/15"
# ── Cloud control plane (Hetzner Cloud, fsn1) ──────────────────────────────
cp_count = 3
# PINNED to the live nodes (verified against discovery/cp_nodes 2026-07-07).
# Order follows cp_ip_offset: kaycee=.11, bettie=.12, ofelia=.13. A wrong name
# here RENAMES a live control plane — check twice.
cp_names = ["kaycee", "bettie", "ofelia"]
cp_server_type = "ccx23" # 4 vCPU / 16 GB, dedicated x86
cp_location = "fsn1"
cp_ip_offset = 11 # CP private IPs → 10.40.11.11 / .12 / .13 (etcd)
lb_type = "lb11"
lb_ip_offset = 5 # API LB private IP → 10.40.11.5
lb_public = false # private-only LB; the API endpoint stays on the kube-cp net
# ── Bare-metal control planes (Hetzner Robot; kube-cp VLAN 11, gw .1 = spine) ──
# `name` keys the apply resources (stable across list edits). VIP 10.40.11.5
# (= the retired hcloud LB IP, so api_dns_name carried over unchanged).
# provisioned=false → the one-time install apply dials maint_ip (maintenance
# mode); flip true per node as it comes up.
cps = [
{ name = "harlan", cp_ip = "10.40.11.11", maint_ip = "178.63.124.20", robot_id = 3027819, install_serial = "17451A00D9F8" },
{ name = "imelda", cp_ip = "10.40.11.12", maint_ip = "178.63.124.21", robot_id = 3027863, install_serial = "1708162471F6" },
{ name = "roscoe", cp_ip = "10.40.11.13", maint_ip = "178.63.124.22", robot_id = 3028524, install_serial = "18201C72C94D" },
]
# 2×10G Intel 82599ES SFP+ (ixgbe) enslaved into bond0 (tagged kube-cp VLAN 11,
# spine port-0 breakout ae4-6). The onboard 1G (e1000e) stays the DHCP
# public/egress NIC (default route + NetBird endpoint).
cp_bond_driver = "ixgbe"
vip_offset = 5 # API VIP → 10.40.11.5
# ── Bare-metal workers (Hetzner Robot dedicated; sequential after mgmt-1/2) ──
# `name` keys the apply resources (stable across list edits — removing or
# reordering an entry no longer touches the others). Names verified against
# the live nodes 2026-07-07.
# ── Bare-metal workers (Hetzner Robot; kube VLAN 10) ─────────────────────────
workers = [
{ name = "jeanne", fabric_ip = "10.40.10.11", maint_ip = "178.63.124.38", robot_id = 3008210, install_serial = "S64GNNFX503099" },
{ name = "sheron", fabric_ip = "10.40.10.12", maint_ip = "178.63.124.37", robot_id = 3008211, install_serial = "S64GNJ0WC25870" },
@@ -52,8 +52,8 @@ cluster = {
}
# Operator/CI sources allowed on the Talos host firewall (apid 50000 + apiserver
# 6443), on top of the node planes. NetBird peer range ONLY — no public IPs (the
# hcloud firewall also blocks public apiserver/apid; see hcloud-firewall.tf). A
# re-bootstrap dials apid on a CP public IP, so temporarily re-add the operator's
# /32 here (and open 50000 on the hcloud firewall) for that one step.
# 6443), on top of the node planes. NetBird peer range ONLY — no public IPs.
# Operators reach the CPs over the NetBird kube-cp route (the CPs are the route
# peers); a re-bootstrap that must dial apid before the mesh is up goes through
# a maint_ip (maintenance mode is unauthenticated — no firewall yet).
trusted_cidrs = ["10.254.0.0/15"]
+83 -101
View File
@@ -1,117 +1,99 @@
# Spread placement group — forces the 3 CPs onto DISTINCT physical hosts, so no
# single host failure can take out >1 etcd member / break quorum. (hcloud caps a
# spread group at 10 servers; 3 is fine.)
resource "hcloud_placement_group" "control_plane" {
name = "yucca-${var.region_code}-${var.cluster.name}-cp"
type = "spread"
labels = { cluster = var.cluster.name, role = "control-plane" }
# ── Bare-metal control planes ────────────────────────────────────────────────
# Applied over apid to nodes already in Talos maintenance mode at their maint_ip
# (Hetzner public DHCP on the onboard 1G NIC). After the install+reboot they hold
# their kube-cp VLAN address and join the NetBird mesh.
#
# bond0 (2×10G LACP, ixgbe) → vlan 11 (kube-cp) = cp_ip — etcd + apiserver + VIP
# route to the kube VLAN via the kube-cp IRB (10.40.11.1) — apiserver→kubelet
# default route via the onboard 1G public NIC (DHCP) — egress + NetBird endpoint
#
# The API VIP (10.40.11.5) is Talos-managed on the VLAN: etcd elects one holder,
# so it's only up while the cluster is healthy — exactly what the api_dns_name
# record points at.
#
# CPs are PROVISIONED to maintenance mode out of band — see the runbook
# (./README.md): Hetzner rescue → dd the Talos metal image → reboot. This stack
# assumes they're already there.
locals {
# Keyed by hostname (cp_node_map) — same stable key as the apply resource.
cp_node_patches = { for hostname, n in local.cp_node_map : hostname => [
# Install disk by SERIAL (never by name — enumeration swaps across boots).
yamlencode({
machine = { install = {
diskSelector = { serial = n.install_serial }
image = local.install_image
} }
}),
yamlencode({
machine = {
network = {
interfaces = [{
interface = "bond0"
dhcp = false
# Bond members selected by NIC driver (cp_bond_driver, ixgbe) — exactly
# the two 10G SFP+ ports; the onboard 1G public NIC is e1000e.
bond = {
mode = "802.3ad"
lacpRate = "fast"
xmitHashPolicy = "layer3+4"
miimon = 100
deviceSelectors = local.c.cp_bond_driver != null ? [{
driver = local.c.cp_bond_driver
}] : null
interfaces = local.c.cp_bond_driver != null ? null : local.c.cp_bond_interfaces
}
vlans = [{
vlanId = module.addr_site.kube_cp_vlan_id # 11
addresses = ["${n.cp_ip}/${local.kube_cp_prefix}"]
# The workers live one IRB away — pin the return route so
# apiserver→kubelet + geneve ride the fabric, not the mesh.
routes = [{ network = local.kube_cidr, gateway = local.kube_cp_gw }]
vip = { ip = local.api_vip }
}]
}]
}
}
}),
yamlencode({ apiVersion = "v1alpha1", kind = "HostnameConfig", auto = "off", hostname = hostname }),
] }
}
# ── Cloud control-plane VMs ──────────────────────────────────────────────────
# 3× CCX23 booted from the Talos snapshot, configured via user_data (the per-CP
# machine config from talos.tf). Public IPv4 = NetBird NAT traversal + the TF
# runner's bootstrap path; private IP (eth1, kube-cp subnet) = etcd. Talos ignores
# SSH, so no ssh_keys. ignore_changes keeps re-applies from recycling live nodes.
resource "hcloud_server" "control_plane" {
count = var.cluster.cp_count
name = local.cp_hostnames[count.index]
image = data.hcloud_image.talos.id
server_type = var.cluster.cp_server_type
location = var.cluster.cp_location
placement_group_id = hcloud_placement_group.control_plane.id
user_data = data.talos_machine_configuration.cp[count.index].machine_configuration
public_net {
ipv4_enabled = true
ipv6_enabled = false
}
network {
network_id = hcloud_network.kube_cp.id
ip = local.cp_private_ips[count.index]
}
labels = { cluster = var.cluster.name, role = "control-plane" }
depends_on = [hcloud_network_subnet.kube_cp]
lifecycle {
ignore_changes = [user_data, image]
# etcd members — replacing one rolls quorum; destroying all rolls the
# cluster. Any legitimate replace (scale-down, location change) must
# temporarily lift this flag, deliberately.
prevent_destroy = true
}
}
# Live CP config sync — user_data only configures a CP at CREATION (and is
# ignore_changes above), so config edits never reached running CPs; today they were
# hand-patched via talosctl. This applies the current rendered config to each live CP
# on every apply (mode auto: no reboot for the config we manage). Notably it keeps the
# worker /etc/hosts entries fresh: a re-provisioned worker gets a new NetBird IP, and
# without this the apiserver keeps dialing the dead one.
# Keyed by HOSTNAME, not list position: removing or reordering a CP in tfvars
# must never shift another node's resource address (a shift = replace = an etcd
# member reset). Matches the workers.tf pattern.
resource "talos_machine_configuration_apply" "cp" {
count = var.cluster.cp_count
for_each = local.cp_node_map
client_configuration = talos_machine_secrets.this.client_configuration
machine_configuration_input = data.talos_machine_configuration.cp[count.index].machine_configuration
node = local.cp_private_ips[count.index]
endpoint = local.cp_private_ips[count.index]
machine_configuration_input = data.talos_machine_configuration.cp.machine_configuration
# The FIRST apply targets the maintenance-mode node at its Hetzner public IP
# (provisioned=false); the config brings up bond0.11 at cp_ip + joins NetBird and
# the node reboots into the cluster. Every later apply targets the LIVE node at
# its kube-cp IP (over the NetBird kube-cp route). Flip provisioned in tfvars per
# CP as it comes up.
node = each.value.provisioned ? each.value.cp_ip : each.value.maint_ip
endpoint = each.value.provisioned ? each.value.cp_ip : each.value.maint_ip
config_patches = local.cp_node_patches[each.key]
apply_mode = "auto"
depends_on = [hcloud_server.control_plane]
# reset=false: decommissioning an etcd member must be a deliberate
# `talosctl reset` (after `etcd leave`), never a terraform destroy side effect.
# NB: on_destroy is read from STATE, so this protects only after it has been
# applied once.
on_destroy = {
reboot = true
reset = false
graceful = false
}
lifecycle {
# cp_netbird_patch is silently OMITTED when the setup key is "" (the
# credential-less validate default) — an env-less apply would strip NetBird
# from the live CP configs. Fail loudly instead.
# from the live CP configs, cutting the operators' kube-cp route. Fail loudly.
precondition {
condition = length(var.netbird_talos_cp_setup_key) > 0
error_message = "netbird_talos_cp_setup_key is empty — run applies through tf/op-run.sh (op run env missing or op:// ref resolved empty)."
}
}
}
# ── API load balancer ────────────────────────────────────────────────────────
# Fronts the 3 CPs on 6443. Private IP (kube-cp) is the in-cluster target; the
# public frontend (lb_public) is what operators + workers dial via api_dns_name.
# certSANs (talos.tf) already include api_dns_name + the private LB IP.
resource "hcloud_load_balancer" "kube_api" {
name = "yucca-${var.region_code}-${var.cluster.name}-kube-api"
load_balancer_type = var.cluster.lb_type
location = var.cluster.cp_location
labels = { cluster = var.cluster.name }
}
resource "hcloud_load_balancer_network" "kube_api" {
load_balancer_id = hcloud_load_balancer.kube_api.id
network_id = hcloud_network.kube_cp.id
ip = local.lb_private_ip
enable_public_interface = var.cluster.lb_public
depends_on = [hcloud_network_subnet.kube_cp]
}
resource "hcloud_load_balancer_service" "kube_api" {
load_balancer_id = hcloud_load_balancer.kube_api.id
protocol = "tcp"
listen_port = 6443
destination_port = 6443
health_check {
protocol = "tcp"
port = 6443
interval = 10
timeout = 5
retries = 3
}
}
# Target the CPs over their PRIVATE IPs (LB is attached to the same network).
resource "hcloud_load_balancer_target" "kube_api" {
count = var.cluster.cp_count
type = "server"
load_balancer_id = hcloud_load_balancer.kube_api.id
server_id = hcloud_server.control_plane[count.index].id
use_private_ip = true
depends_on = [hcloud_load_balancer_network.kube_api]
}
@@ -35,13 +35,14 @@ output "discovery" {
}
kubernetes = {
cluster_name = var.cluster.name
api_endpoint = local.cluster_endpoint # https://<api_dns_name>:6443 (LB)
api_endpoint = local.cluster_endpoint # https://<api_dns_name>:6443 (VIP)
api_vip = local.api_vip # Talos-elected VIP on kube-cp (the api_dns_name A record)
operator_endpoint = local.operator_endpoint # direct bootstrap-CP apiserver
cp_node_ips = local.cp_private_ips # kube-cp IPs; operators/yuctl reach via NetBird
cp_node_ips = local.cp_ips # kube-cp IPs; operators/yuctl reach via NetBird
worker_node_ips = [for w in var.cluster.workers : w.fabric_ip]
# Node NAME → IP maps (short wordlist names). Consumed by the netbird stack
# (yucca.futo.network records) instead of hardcoding names in two stacks.
cp_nodes = { for i, n in var.cluster.cp_names : n => local.cp_private_ips[i] }
cp_nodes = { for n in var.cluster.cps : n.name => n.cp_ip }
worker_nodes = { for w in var.cluster.workers : w.name => w.fabric_ip }
kubeconfig_ref = "op://${local._disc_vault}/${local._kubeconfig_title}/password"
talosconfig_ref = "op://${local._disc_vault}/${local._talosconfig_title}/password"
@@ -1,15 +1,16 @@
# Talos host ingress firewall (default-deny + per-service allow-lists). Governs
# HOST-network ports only; pod/ClusterIP traffic rides Cilium.
#
# Trust planes for this hybrid cluster:
# Trust planes:
# kube_cidr 10.40.10.0/24 workers' fabric IPs (east-west, BGP)
# kube_cp_cidr 10.40.11.0/24 CP private IPs + the API LB (etcd, LB health-checks)
# netbird_node_cidr 10.254.0.0/15 the NetBird mesh (apiserver↔kubelet, node control)
# kube_cp_cidr 10.40.11.0/24 CP IPs + the API VIP (etcd, apiserver)
# netbird_node_cidr 10.254.0.0/15 the NetBird mesh (operators, backup plane)
# trusted_cidrs operator/CI source ranges
#
# ⚠️ The TF runner dials the CP PUBLIC IPs for bootstrap (apid 50000) and the
# helm/kubernetes providers (apiserver 6443). Its source IP MUST be in
# trusted_cidrs (e.g. the CI runner's egress / NetBird range) or those steps hang.
# ⚠️ The TF runner dials the CP kube-cp IPs (over the NetBird kube-cp route) for
# bootstrap (apid 50000) and the helm/kubernetes providers (apiserver 6443). Its
# source IP MUST be in trusted_cidrs (e.g. the CI runner's NetBird range) or
# those steps hang.
locals {
firewall_allow = concat([local.kube_cidr, local.kube_cp_cidr, local.c.netbird_node_cidr], var.trusted_cidrs)
operator_allow = local.firewall_allow
@@ -41,7 +42,7 @@ locals {
}),
# Cilium geneve overlay (tunnel routing): pod↔pod is encapsulated node-to-node
# (UDP 6081). Required across BOTH L2 domains — worker↔worker over the fabric and
# CP↔worker over the mesh — or pod-to-pod traffic is silently dropped.
# CP↔worker routed via the spine IRBs — or pod-to-pod traffic is silently dropped.
yamlencode({
apiVersion = "v1alpha1"
kind = "NetworkRuleConfig"
+1 -3
View File
@@ -2,9 +2,7 @@
# Helm (OCI charts), then Flux reconciles this repo's kubernetes/clusters/prod
# path on its own. Lands after the cluster + CNI (providers.tf helm/kubernetes).
#
# ⚠ TEMPORARY: the sync ref is feat/prod (var.flux_git_ref) so father can be
# GitOps-managed before the branch merges. Flip flux_git_ref to "main" (the
# default once this merges) — nothing else changes.
# Sync ref = main (var.flux_git_ref default; CI guards it stays that way).
resource "helm_release" "flux_operator" {
name = "flux-operator"
@@ -1,29 +0,0 @@
# hcloud firewall for the control-plane VMs — enforces "no public access to the
# cluster" at the cloud edge. Only NetBird's WireGuard is allowed inbound from the
# internet; apiserver (6443) + apid (50000) + everything else is dropped. The
# cluster is reached ONLY over NetBird (WireGuard tunnel, arrives on 51820/udp) or
# the private kube-cp network — neither of which this filters (hcloud firewalls
# apply to the public interface; private-net + in-tunnel traffic is untouched).
#
# Egress is unrestricted (hcloud default) — image pulls, NetBird signal/relay, etc.
#
# NB: a future re-bootstrap dials apid (50000) on a CP public IP — temporarily add
# an operator-source rule for 50000, or bootstrap from a NetBird-reachable path.
resource "hcloud_firewall" "control_plane" {
name = "yucca-${var.region_code}-${var.cluster.name}-cp"
labels = { cluster = var.cluster.name, role = "control-plane" }
rule {
direction = "in"
protocol = "udp"
port = "51820"
source_ips = ["0.0.0.0/0", "::/0"]
description = "NetBird WireGuard (P2P)"
}
}
# Attach to the CPs without recreating them.
resource "hcloud_firewall_attachment" "control_plane" {
firewall_id = hcloud_firewall.control_plane.id
server_ids = hcloud_server.control_plane[*].id
}
+5 -26
View File
@@ -1,33 +1,12 @@
# Talos Image Factory schematic — the extension set, managed in TF. The resource
# registers schematic.yaml with the factory and returns its deterministic id; we
# derive the CP (hcloud-amd64) + worker (metal-amd64) image URLs from it. No
# hand-pasted schematic id, no out-of-band curl.
# registers schematic.yaml with the factory and returns its deterministic id; the
# metal installer/image URLs derive from it. No hand-pasted schematic id, no
# out-of-band curl. One schematic for every node (all bare-metal).
resource "talos_image_factory_schematic" "this" {
schematic = file("${path.module}/schematic.yaml")
}
# Worker (bare-metal) schematic — same set minus qemu-guest-agent (see the file).
resource "talos_image_factory_schematic" "worker" {
schematic = file("${path.module}/schematic-worker.yaml")
}
locals {
talos_schematic_id = talos_image_factory_schematic.this.id
talos_worker_schematic_id = talos_image_factory_schematic.worker.id
talos_hcloud_image_url = "https://factory.talos.dev/image/${local.talos_schematic_id}/v${var.cluster.talos_version}/hcloud-amd64.raw.xz"
talos_metal_image_url = "https://factory.talos.dev/image/${local.talos_worker_schematic_id}/v${var.cluster.talos_version}/metal-amd64.raw.xz"
}
# Talos amd64 image as an hcloud snapshot. hcloud can't boot the Talos ISO, so the
# snapshot is built ONCE, out of band, by:
#
# mise run hetzner:talos-image
#
# (it reads talos_hcloud_image_url from this stack's output, spins a temporary
# rescue server, dd's the image, snapshots, tears down). This data source then
# resolves it by label — rebuild only on a Talos version bump.
data "hcloud_image" "talos" {
with_selector = "os=talos,cluster=${var.cluster.name},arch=amd64"
with_architecture = "x86"
most_recent = true
talos_schematic_id = talos_image_factory_schematic.this.id
talos_metal_image_url = "https://factory.talos.dev/image/${local.talos_schematic_id}/v${var.cluster.talos_version}/metal-amd64.raw.xz"
}
@@ -1,33 +0,0 @@
# ── One-shot state migration: count → hostname-keyed workers ─────────────────
# Moves the existing worker apply instances to their stable hostname keys and
# forgets the retired node-names shuffle module WITHOUT destroying anything.
# The gate before merging: a local `mise tf:plan` must show exactly these three
# moves + one "removed from state" + the on_destroy in-place updates — zero
# destroys, zero replaces, zero diffs on the CP resources.
# DELETE this file once CI has applied it (the blocks are inert afterwards, but
# the `removed` block conflicts if module "names" is ever reintroduced).
moved {
from = talos_machine_configuration_apply.worker[0]
to = talos_machine_configuration_apply.worker["yucca-htz-fsn-father-k8s-jeanne"]
}
moved {
from = talos_machine_configuration_apply.worker[1]
to = talos_machine_configuration_apply.worker["yucca-htz-fsn-father-k8s-sheron"]
}
moved {
from = talos_machine_configuration_apply.worker[2]
to = talos_machine_configuration_apply.worker["yucca-htz-fsn-father-k8s-dianna"]
}
# Forget (don't destroy) the retired names module — its only resource is a
# random_shuffle; destroying would be inert, but "forget" keeps the plan clean.
removed {
from = module.names
lifecycle {
destroy = false
}
}
@@ -0,0 +1,91 @@
# ─── netops namespace + fabric-credential Secrets ────────────────────────────
# The netops stack (kubernetes/apps/prod/htz-fsn1/netops/) mounts fabric
# credentials that must NEVER be in git: the read-only `netops` Junos login's
# SSH key + password (fabric stack, fabric.tf netops_users) and the Grafana
# admin password. Historically these were hand-created (`kubectl create secret`)
# and DIED WITH THE CLUSTER on the 2026-07 rebuild — now they're provisioned
# here from the same 1Password items, so a rebuild restores them with the stack.
# Flux owns the workloads around them; TF owns the namespace + these Secrets
# (the namespace also carries the VictoriaMetrics hostPath PVC, so it must
# survive flux prunes — TF ownership replaces the old prune-disabled manifest).
data "onepassword_item" "netops_ssh_key" {
vault = data.onepassword_vault.prod.uuid
title = "NETOPS_FABRIC_SSH_PRIVATE_KEY" # DOCUMENT item; file id_ed25519
}
data "onepassword_item" "netops_fabric_password" {
vault = data.onepassword_vault.prod.uuid
title = "NETOPS_FABRIC_PASSWORD"
}
data "onepassword_item" "grafana_admin" {
vault = data.onepassword_vault.prod.uuid
title = "FATHER_GRAFANA_ADMIN"
}
resource "kubernetes_namespace_v1" "netops" {
metadata {
name = "netops"
labels = {
# VictoriaMetrics persists to a hostPath (/var/mnt); baseline forbids it.
"pod-security.kubernetes.io/enforce" = "privileged"
}
annotations = {
# Belt-and-braces from the flux-owned era (the namespace manifest is gone
# from the tree, but flux's GC honors this if it ever re-tracks the object).
"kustomize.toolkit.fluxcd.io/prune" = "disabled"
}
}
}
# SSH key for junos-exporter (NETCONF scrape) + oxidized (config backup) — both
# mount key `id_ed25519` and log in as the `netops` Junos user.
resource "kubernetes_secret_v1" "netops_ssh" {
metadata {
name = "netops-ssh"
namespace = kubernetes_namespace_v1.netops.metadata[0].name
}
data = {
id_ed25519 = one([for f in data.onepassword_item.netops_ssh_key.file : f.content if f.name == "id_ed25519"])
}
}
resource "kubernetes_secret_v1" "grafana_admin" {
metadata {
name = "grafana-admin"
namespace = kubernetes_namespace_v1.netops.metadata[0].name
}
data = {
password = data.onepassword_item.grafana_admin.password
}
}
# hyperglass device inventory — embeds the netops PASSWORD (netmiko can't
# key-auth through hyperglass config), hence a Secret and not the configmap.
resource "kubernetes_secret_v1" "hyperglass_devices" {
metadata {
name = "hyperglass-devices"
namespace = kubernetes_namespace_v1.netops.metadata[0].name
}
# Spine only: hyperglass's juniper directives require source4 AND source6 per
# device, and only the spine has both (lo0 + the transit v6) — the leaf has no
# public/v6 presence, so LG queries from it would be meaningless anyway.
data = {
"devices.yaml" = yamlencode({
devices = [
{
name = "corenetsw"
description = "spine VC (QFX5200-32C x2)"
address = module.addr_site.spine_mgmt_ip
platform = "juniper"
attrs = {
source4 = "69.48.224.254" # lo0 (fabric stack, transits.loopback)
source6 = "2a01:4a0:1338:226::2" # transit /64 local (fabric stack, transits.local_v6)
}
credential = { username = "netops", password = data.onepassword_item.netops_fabric_password.password }
},
]
})
}
}
@@ -1,23 +0,0 @@
# Isolated Hetzner Cloud network for the control plane. Holds ONLY the 3 CP VMs
# (etcd CP↔CP on private IPs) + the API LB's private IP. The workers do NOT join
# it — CP↔worker rides the NetBird mesh. Range = kube-cp (10.40.11.0/24), carved
# from the site supernet for collision-free IPAM (see fabric-addressing).
resource "hcloud_network" "kube_cp" {
name = "yucca-${var.region_code}-${var.cluster.name}-kube-cp"
ip_range = local.kube_cp_cidr
labels = { cluster = var.cluster.name, plane = "control" }
}
# Single cloud subnet for the CP VMs + LB (no vSwitch subnet — workers aren't here).
resource "hcloud_network_subnet" "kube_cp" {
network_id = hcloud_network.kube_cp.id
type = "cloud"
network_zone = "eu-central" # fsn1/nbg1/hel1
ip_range = local.kube_cp_cidr
}
locals {
# Deterministic private IPs in the kube-cp subnet.
cp_private_ips = [for i in range(var.cluster.cp_count) : cidrhost(local.kube_cp_cidr, var.cluster.cp_ip_offset + i)]
lb_private_ip = cidrhost(local.kube_cp_cidr, var.cluster.lb_ip_offset)
}
+8 -15
View File
@@ -4,36 +4,29 @@ output "cluster_summary" {
cluster_name = var.cluster.name
api_endpoint = local.cluster_endpoint
api_dns_name = local.api_dns_name
api_vip = local.api_vip
operator_endpoint = local.operator_endpoint
lb_private_ip = local.lb_private_ip
cp_public_ips = local.cp_public_ips
cp_private_ips = local.cp_private_ips
cp_ips = local.cp_ips
worker_fabric_ips = [for w in var.cluster.workers : w.fabric_ip]
}
}
# DNS hint: api_dns_name resolves to the PRIVATE LB IP (lb_public = false — the
# API is reachable only over the NetBird mesh; the netbird stack's
# DNS hint: api_dns_name resolves to the Talos-elected VIP on the kube-cp VLAN
# (the API is reachable only over the NetBird kube-cp route; the netbird stack's
# yucca.futo.network zone serves the record). No public A record exists.
output "api_dns_record" {
description = "The internal record: <api_dns_name> → <lb private IP> (NetBird DNS)."
value = "${local.api_dns_name} A ${local.lb_private_ip}"
description = "The internal record: <api_dns_name> → <API VIP> (NetBird DNS)."
value = "${local.api_dns_name} A ${local.api_vip}"
}
# Image Factory outputs — `mise run hetzner:talos-image` reads the hcloud URL to
# build the snapshot; the metal URL feeds the worker rescue-install runbook.
# Image Factory outputs — the metal URL feeds the rescue-install runbook (README).
output "talos_schematic_id" {
description = "Image Factory schematic id (from schematic.yaml)."
value = local.talos_schematic_id
}
output "talos_hcloud_image_url" {
description = "Factory hcloud-amd64 raw.xz URL — input to hcloud-upload-image (CP snapshot)."
value = local.talos_hcloud_image_url
}
output "talos_metal_image_url" {
description = "Factory metal-amd64 raw.xz URL — workers dd this in rescue."
description = "Factory metal-amd64 raw.xz URL — nodes dd this in rescue."
value = local.talos_metal_image_url
}
+4 -12
View File
@@ -1,11 +1,8 @@
# hcloud — the control-plane VMs, their private subnet, the API LB, and the Talos
# snapshot lookup. Token comes from HCLOUD_TOKEN (op run --env-file=tf/.env.prod).
provider "hcloud" {}
# helm + kubernetes bind to ONE cluster: the bootstrap CP's directly-reachable
# (public) apiserver — up immediately after bootstrap and in the cert SANs, unlike
# the LB which only goes healthy once an apiserver answers. Creds come from the
# Talos-minted admin kubeconfig (known after talos_cluster_kubeconfig applies).
# apiserver (over the NetBird kube-cp route) — up immediately after bootstrap and
# in the cert SANs, unlike the VIP which only settles once etcd elects a holder.
# Creds come from the Talos-minted admin kubeconfig (known after
# talos_cluster_kubeconfig applies).
provider "helm" {
kubernetes = {
host = local.operator_endpoint
@@ -25,8 +22,3 @@ provider "kubernetes" {
# 1Password — persists the kube/talosconfig into yucca_tf_prod (secrets.tf). Auth
# via OP_SERVICE_ACCOUNT_TOKEN (op run).
provider "onepassword" {}
# netbird — read-only worker peer lookups: their mesh IPs feed the CPs'
# extraHostEntries so the apiserver dials worker kubelets peer-to-peer over the
# mesh (no mgmt route in the path). PAT via NB_PAT (op run).
provider "netbird" {}
@@ -1,11 +0,0 @@
# Talos Image Factory schematic for the father WORKERS (bare-metal Hetzner Robot).
# Split from schematic.yaml (the CP/VM set): qemu-guest-agent must NOT be here — on
# metal there is no virtio port, the extension service waits forever for
# /dev/virtio-ports/org.qemu.guest_agent.0, and the boot sequence never reaches
# `running`, failing every talos health check (and with it every TF plan/apply).
customization:
systemExtensions:
officialExtensions:
- siderolabs/netbird # node-level overlay (worker↔apiserver)
- siderolabs/intel-ucode # worker Xeon microcode
- siderolabs/util-linux-tools # fstrim et al.
@@ -1,12 +1,13 @@
# Talos Image Factory schematic for the `father` cluster — the single source of
# truth for the node extension set. Registered with the factory by TF
# (talos_image_factory_schematic, image.tf); its id derives the CP (hcloud) and
# worker (metal) image URLs. One schematic covers both platforms — the extras are
# harmless no-ops on the other (qemu-guest-agent on metal, intel-ucode on a VM).
# truth for the node extension set (CPs + workers, all bare-metal Hetzner Robot).
# Registered with the factory by TF (talos_image_factory_schematic, image.tf); its
# id derives the metal installer/image URLs. NB: qemu-guest-agent must NOT be here —
# on metal there is no virtio port, the extension service waits forever for
# /dev/virtio-ports/org.qemu.guest_agent.0, and the boot sequence never reaches
# `running`, failing every talos health check (and with it every TF plan/apply).
customization:
systemExtensions:
officialExtensions:
- siderolabs/netbird # node-level overlay (CP↔fabric route)
- siderolabs/qemu-guest-agent # hcloud control-plane VMs
- siderolabs/intel-ucode # worker Xeon microcode
- siderolabs/netbird # node-level overlay (operator plane / kube-cp route)
- siderolabs/intel-ucode # Xeon microcode
- siderolabs/util-linux-tools # fstrim et al.
+84 -176
View File
@@ -1,22 +1,18 @@
# ── Talos bring-up (hybrid) ──────────────────────────────────────────────────
# Cloud CPs are configured via hcloud user_data (controlplane.tf); bare-metal
# workers via apid apply (workers.tf). Both join the SAME cluster (one set of
# machine_secrets) and the SAME NetBird mesh (node IPs are NetBird addresses).
# ── Talos bring-up (all bare-metal) ──────────────────────────────────────────
# CPs and workers are both driven over apid (controlplane.tf / workers.tf) into
# ONE cluster (one set of machine_secrets). All node planes ride the fabric:
#
# node plane (CP↔worker, etcd-client, apiserver↔kubelet) → NetBird (100.64/10)
# etcd (CP↔CP) → kube-cp hcloud subnet
# worker↔worker pod east-west → kube fabric (50G), Cilium BGP
# API endpoint → public Hetzner Cloud LB
# etcd (CP↔CP) + apiserver + API VIP → kube-cp fabric VLAN 11 (10.40.11.0/24)
# worker↔worker pod east-west → kube fabric VLAN 10 (50G), Cilium BGP
# CP↔worker (apiserver↔kubelet, geneve) → routed kube↔kube-cp via the spine IRBs
# operators/CI → NetBird mesh (kube-cp routed via the CPs)
#
# Bootstrap/kubeconfig/health dial the CP PUBLIC IPs (firewalled) — the only thing
# the TF runner can reach before NetBird/the LB settle.
# NetBird stays on every node as the operator/backup plane — node-to-node traffic
# no longer depends on it (static fabric routes are pinned in the machine configs).
# Node names are EXPLICIT in tfvars (cluster.cp_names + workers[*].name) — the
# node-names shuffle module was retired here: auto-picked names re-roll when the
# pool input changes (adding an explicit name shrinks the shuffle pool → every
# auto name changes → every node renames), and positional slotting meant a
# cp_count change renamed all workers. migrations.tf forgets the old module
# state without destroying anything.
# Node names are EXPLICIT in tfvars (cluster.cps[*].name + workers[*].name) —
# auto-picked names re-roll when the pool input changes, silently renaming (=
# replacing) live nodes.
locals {
c = var.cluster
@@ -26,49 +22,40 @@ locals {
pod_cidr = "10.250.0.0/17" # 10.250.0.0 – 10.250.127.255
service_cidr = "10.250.128.0/17" # 10.250.128.0 – 10.250.255.255
# Factory installers (keep each schematic's extensions). Workers consult theirs on
# install/upgrade; CPs boot the hcloud snapshot and only consult this on a reinstall.
# SPLIT per role: the worker schematic drops qemu-guest-agent (blocks metal boot).
cp_install_image = "factory.talos.dev/metal-installer/${local.talos_schematic_id}:v${local.c.talos_version}"
worker_install_image = "factory.talos.dev/metal-installer/${local.talos_worker_schematic_id}:v${local.c.talos_version}"
# Factory installer (keeps the schematic's extensions) — one schematic for every
# node (all metal now); consulted on install/upgrade.
install_image = "factory.talos.dev/metal-installer/${local.talos_schematic_id}:v${local.c.talos_version}"
# Private API endpoint: a NetBird DNS-zone name resolving to the PRIVATE LB IP
# (10.40.11.5). Name = kube.<cluster>.<region>.<provider>.yucca.futo.network. It's
# in the cert SANs + on each CP as a host-entry; NetBird peers resolve it via the
# yucca.futo.network zone (netbird stack) and reach the LB over the kube-cp route
# (CPs are the route peers). Node-side traffic never resolves it: kubelets dial
# KubePrism (127.0.0.1:7445), which load-balances to the CP IPs directly.
# legacy_api_dns_name (the old yucca.internal name) stays in the SANs + host
# entries so pre-migration kubeconfigs keep verifying — drop it once rotated.
api_dns_name = "kube.${local.c.name}.${var.region_code}.${var.provider_code}.yucca.futo.network"
legacy_api_dns_name = "kube.${local.c.name}.${var.region_code}.${var.provider_code}.yucca.internal"
cluster_endpoint = "https://${local.api_dns_name}:6443"
# API endpoint: a NetBird DNS-zone name resolving to the Talos-elected VIP
# (10.40.11.5, kube-cp VLAN — same IP the retired hcloud LB held, so the record
# carried over). Name = kube.<cluster>.<region>.<provider>.yucca.futo.network.
# It's in the cert SANs + on each node as a host-entry; NetBird peers resolve it
# via the yucca.futo.network zone (netbird stack) and reach the VIP over the
# kube-cp route (CPs are the route peers). Node-side traffic never resolves it:
# kubelets dial KubePrism (127.0.0.1:7445), which load-balances to the CP IPs.
api_dns_name = "kube.${local.c.name}.${var.region_code}.${var.provider_code}.yucca.futo.network"
cluster_endpoint = "https://${local.api_dns_name}:6443"
kube_cp_prefix = split("/", local.kube_cp_cidr)[1] # 24
api_vip = cidrhost(local.kube_cp_cidr, local.c.vip_offset) # 10.40.11.5
kube_cp_prefix = split("/", local.kube_cp_cidr)[1] # 24
# Hostnames: yucca-htz-fsn-father-k8s-<name>. Workers additionally get a
# hostname-keyed map — the STABLE key for the apply resources (workers.tf) and
# the netbird peer lookups, so list edits can't shift another node's identity.
node_prefix = "yucca-${var.provider_code}-${var.region_code}-${local.c.name}-k8s"
cp_hostnames = [for n in local.c.cp_names : "${local.node_prefix}-${n}"]
workers_named = [for w in local.c.workers : merge(w, { hostname = "${local.node_prefix}-${w.name}" })]
worker_hostnames = [for w in local.workers_named : w.hostname]
worker_node_map = { for w in local.workers_named : w.hostname => w }
# Hostnames: yucca-htz-fsn-father-k8s-<name>. Both roles get hostname-keyed
# maps — the STABLE key for the apply resources, so list edits can't shift
# another node's identity.
node_prefix = "yucca-${var.provider_code}-${var.region_code}-${local.c.name}-k8s"
cps_named = [for n in local.c.cps : merge(n, { hostname = "${local.node_prefix}-${n.name}" })]
cp_node_map = { for n in local.cps_named : n.hostname => n }
cp_ips = local.cps_named[*].cp_ip
workers_named = [for w in local.c.workers : merge(w, { hostname = "${local.node_prefix}-${w.name}" })]
worker_node_map = { for w in local.workers_named : w.hostname => w }
# apiserver cert SANs — the names/IPs clients dial. NOT the CP public IPs (those
# don't exist until the servers are created from this very config).
# apiserver cert SANs — the names/IPs clients dial.
apiserver_cert_sans = concat(
[local.api_dns_name, local.legacy_api_dns_name, local.lb_private_ip],
local.cp_private_ips,
[local.api_dns_name, local.api_vip],
local.cp_ips,
["127.0.0.1", "localhost"],
)
# ── Shared patches (every node) ──────────────────────────────────────────
cp_install_patch = yamlencode({
machine = { install = { disk = local.c.install_disk, image = local.cp_install_image } }
})
# Talos's default forwards coredns's upstream queries to the host DNS on a link-local
# address (169.254.116.108) — unreachable from pods under Cilium's eBPF datapath
# (bpf.masquerade), so every EXTERNAL lookup from a pod times out while cluster.local
@@ -77,20 +64,13 @@ locals {
machine = { features = { hostDNS = { forwardKubeDNSToHost = false } } }
})
# NetBird node-level overlay. CPs and workers join with DIFFERENT setup keys so they
# land in different groups: CPs → [talos, talos_cp], workers → [talos]. talos_cp is
# the CP-only router group for the kube-cp network — the workers must NOT be in it, or
# NetBird treats them as kube-cp routers and they never install the client route (their
# pods can't reach the apiserver). See the netbird stack for the group/router wiring.
# Both CP + worker run netbird in its normal (modern) mode. The pod→routed-subnet
# problem — Cilium's eBPF host-routing does its FIB lookup against the MAIN table
# only, so any route netbird parks in a policy table (it has been observed using
# both main and table 7120 across versions/restarts) is invisible to POD egress,
# and pod→apiserver via kube-cp gets "no route to host" — is fixed DETERMINISTICALLY
# on the workers by worker_netbird_route_patch (a Talos-managed main-table route),
# NOT by pinning netbird to its deprecated NB_USE_LEGACY_ROUTING mode. CPs don't
# run the eBPF pod-datapath to routed subnets (their control-plane pods are
# hostNetwork → host stack, which honors policy routing), so they need no route.
# NetBird node-level overlay — the OPERATOR plane (kube-cp routed to the mesh via
# the CPs) and a backup path; node-to-node traffic rides the fabric via the static
# routes pinned below. CPs and workers join with DIFFERENT setup keys so they land
# in different groups: CPs → [talos, talos_cp], workers → [talos]. talos_cp is the
# CP-only router group for the kube-cp network — the workers must NOT be in it, or
# NetBird treats them as kube-cp routers and they never install the client route.
# See the netbird stack for the group/router wiring.
netbird_env = ["NB_MANAGEMENT_URL=https://api.netbird.io"]
cp_netbird_patch = var.netbird_talos_cp_setup_key != "" ? yamlencode({
apiVersion = "v1alpha1"
@@ -105,30 +85,10 @@ locals {
environment = concat(["NB_SETUP_KEY=${var.netbird_talos_setup_key}"], local.netbird_env)
}) : ""
# DETERMINISTIC pod→apiserver fix: a Talos-managed route for the kube-cp subnet
# (apiserver + private API LB, reachable only over the mesh) into the MAIN table
# via wt0 — exactly where Cilium's eBPF FIB lookup reads. netbird still installs
# its own route (its table is version-dependent); ours guarantees main is
# populated regardless, so pod egress to kube-cp always resolves. Declaring a
# route on wt0 does NOT disturb the netbird extension (it keeps owning wt0's
# address; Talos only adds the route). Workers only — CPs are ON kube-cp.
worker_netbird_route_patch = yamlencode({
machine = {
network = {
interfaces = [{
interface = "wt0"
routes = [{ network = local.kube_cp_cidr }]
}]
}
}
})
# nodeIP selection:
# CPs → kube-cp hcloud private subnet (apiserver↔CP-kubelet stays private;
# decoupled from NetBird readiness at boot)
# workers → kube fabric IP. All workers share VLAN-10 L2, so Cilium
# autoDirectNodeRoutes routes pod east-west directly over the 50G
# fabric — no BGP, no overlay.
# CPs → kube-cp fabric VLAN 11 (etcd + apiserver↔CP-kubelet)
# workers → kube fabric VLAN 10 (all workers share the L2, so Cilium routes pod
# east-west directly over the 50G fabric)
# clusterDNS must be set explicitly: Talos defaults it to 10.96.0.10 (the upstream
# default service CIDR's DNS) and does NOT derive it from our serviceSubnets — the
# kube-dns Service actually lands at cidrhost(service_cidr, 10). Without this every
@@ -196,13 +156,10 @@ locals {
}
})
cp_base_patches = compact([local.cp_install_patch, local.hostdns_patch, local.cp_netbird_patch, local.cp_nodeip_patch])
# Workers ARE NetBird peers: the apiserver lives on the kube-cp hcloud net, only
# reachable over the mesh, and workers resolve the API endpoint via the yucca.internal
# NetBird DNS zone. nodeIP stays on the fabric (worker_nodeip_patch) so pod east-west
# rides VLAN 10; only the API control path uses NetBird.
# (install patch is PER-WORKER — by disk serial — appended in workers.tf)
worker_base_patches = compact([local.hostdns_patch, local.worker_mayastor_patch, local.worker_volumes_patch, local.worker_netbird_patch, local.worker_netbird_route_patch, local.worker_nodeip_patch])
cp_base_patches = compact([local.hostdns_patch, local.cp_netbird_patch, local.cp_nodeip_patch])
# (install patch is PER-NODE — by disk serial — appended in controlplane.tf /
# workers.tf, along with the bond/VLAN network patches.)
worker_base_patches = compact([local.hostdns_patch, local.worker_mayastor_patch, local.worker_volumes_patch, local.worker_netbird_patch, local.worker_nodeip_patch])
# ── Control-plane cluster config (same on every CP) ──────────────────────
cp_cluster_patch = yamlencode({
@@ -222,21 +179,13 @@ locals {
coreDNS = { disabled = true }
apiServer = {
certSANs = local.apiserver_cert_sans
# Dial kubelets by HOSTNAME first (Talos's default is InternalIP-first). The
# worker InternalIPs are fabric addresses only reachable via the mgmt NetBird
# routers — a single flappy bridge that intermittently broke logs/exec. Worker
# hostnames resolve (via the CPs' extraHostEntries below) to the workers' OWN
# NetBird IPs, so apiserver→kubelet is peer-to-peer over the mesh — the same
# always-on tunnels the kubelets already use to reach the apiserver. The CP
# hostnames resolve via hcloud DNS to their kube-cp IPs, unchanged.
extraArgs = { "kubelet-preferred-address-types" = "Hostname,InternalIP,ExternalIP" }
# hostNetwork pods get /etc/hosts COPIED at sandbox creation — a host-level
# extraHostEntries refresh never reaches the RUNNING apiserver. Stamping the
# entry-set hash into the pod spec forces kubelet to recreate the pod (fresh
# /etc/hosts) whenever a worker's mesh IP changes (e.g. re-provision).
env = { MESH_HOSTS_REVISION = substr(sha256(jsonencode(local.cp_host_entries)), 0, 12) }
# /etc/hosts) whenever the entry set changes (e.g. a node add).
env = { HOSTS_REVISION = substr(sha256(jsonencode(local.cp_host_entries)), 0, 12) }
}
# Pin etcd to the kube-cp hcloud subnet so CP↔CP etcd stays off the mesh.
# Pin etcd to the kube-cp VLAN so CP↔CP etcd stays off the mesh + public NICs.
etcd = { advertisedSubnets = [local.kube_cp_cidr] }
}
})
@@ -244,25 +193,20 @@ locals {
# CP node extras:
# • ip_forward — the CPs are the NetBird route peers for the kube-cp subnet
# (yucca-fsn-father-kube-cp), so they must forward overlay↔subnet traffic.
# • extraHostEntries — on the CPs, resolve api_dns_name to the 3 CP private IPs
# (round-robin, all in the cert SANs). NOT the LB VIP (CPs are LB targets →
# hcloud hairpin), and NOT 127.0.0.1 (a joining CP must reach a WORKING
# apiserver — a peer's — to register its etcd membership; its own apiserver
# isn't up until etcd joins). Off-node peers resolve api_dns_name via the
# NetBird yucca.internal zone.
# • kubelet dialing (worker_mesh_kubelet): the apiserver prefers the Hostname node
# address (cp_cluster_patch), so every node hostname must resolve on the CPs:
# CP hostnames → their kube-cp IPs (stable), worker hostnames → their NetBird
# IPs (data.netbird_peer — the peer-to-peer mesh path, no mgmt route). Talos
# host-dns can't resolve NetBird DNS zones, hence /etc/hosts, which the
# hostNetwork apiserver inherits.
# • extraHostEntries — resolve api_dns_name to the 3 CP IPs (round-robin, all in
# the cert SANs). NOT the VIP (a joining CP must reach a WORKING apiserver — a
# peer's — to register its etcd membership; the VIP may be parked on itself),
# and NOT 127.0.0.1. Every node hostname also resolves to its fabric IP so
# apiserver→kubelet dials ride the fabric (Talos host-dns can't resolve NetBird
# DNS zones, hence /etc/hosts, which the hostNetwork apiserver inherits).
# Off-node peers resolve api_dns_name via the NetBird yucca.futo.network zone.
cp_host_entries = concat(
[for ip in local.cp_private_ips : { ip = ip, aliases = [local.api_dns_name, local.legacy_api_dns_name] }],
[for i, ip in local.cp_private_ips : { ip = ip, aliases = [local.cp_hostnames[i]] }],
# Iterate the tfvars LIST (not the hostname-keyed data map, whose lexical
# order differs) — entry order is part of the rendered CP config, and
# reordering it would churn every CP's machine config for nothing.
[for w in local.workers_named : { ip = data.netbird_peer.worker[w.hostname].ip, aliases = [w.hostname] } if local.c.worker_mesh_kubelet],
[for ip in local.cp_ips : { ip = ip, aliases = [local.api_dns_name] }],
# Iterate the tfvars LISTS (not the hostname-keyed maps, whose lexical order
# differs) — entry order is part of the rendered CP config, and reordering it
# would churn every CP's machine config for nothing.
[for n in local.cps_named : { ip = n.cp_ip, aliases = [n.hostname] }],
[for w in local.workers_named : { ip = w.fabric_ip, aliases = [w.hostname] }],
)
cp_extras_patch = yamlencode({
@@ -271,28 +215,6 @@ locals {
network = { extraHostEntries = local.cp_host_entries }
}
})
# ── Per-CP patches (hostname + hcloud private NIC for etcd) ───────────────
# eth0 = hcloud public (DHCP, default route); eth1 = hcloud private (etcd).
# eth1 MUST be DHCP: hcloud private networks are SDN, not L2 — servers reach each
# other via the network gateway, and hcloud's DHCP is what installs the private
# IP (the one pinned in the hcloud_server network block) + the gateway route. A
# static /24 here makes the node try direct same-subnet ARP, which the SDN doesn't
# answer → the CPs can't reach each other → etcd never forms. VERIFY eth1 is the
# private NIC on the snapshot (else use a deviceSelector).
cp_node_patches = [for i in range(local.c.cp_count) : [
yamlencode({
machine = {
network = {
interfaces = [{
interface = "eth1"
dhcp = true
}]
}
}
}),
yamlencode({ apiVersion = "v1alpha1", kind = "HostnameConfig", auto = "off", hostname = local.cp_hostnames[i] }),
]]
}
# Cluster PKI (sensitive).
@@ -307,21 +229,9 @@ resource "talos_machine_secrets" "this" {
}
}
# Worker NetBird peers — their mesh IPs feed the CPs' /etc/hosts (cp_host_entries)
# so the apiserver dials worker kubelets peer-to-peer. Lookup is by peer name
# (= the worker hostname; the netbird stack keeps one live peer per node). Gated:
# on a greenfield bootstrap the workers aren't peers yet — set
# cluster.worker_mesh_kubelet = false, then flip it after they join.
data "netbird_peer" "worker" {
for_each = { for hostname, w in local.worker_node_map : hostname => w if local.c.worker_mesh_kubelet }
name = each.key
}
# Per-CP machine config — rendered into hcloud user_data (controlplane.tf). Each
# CP gets the shared + CP-cluster + its own per-node patches.
# CP base config — per-CP install/network/hostname patches are added at apply
# time (controlplane.tf).
data "talos_machine_configuration" "cp" {
count = local.c.cp_count
cluster_name = local.c.name
machine_type = "controlplane"
cluster_endpoint = local.cluster_endpoint
@@ -331,7 +241,6 @@ data "talos_machine_configuration" "cp" {
config_patches = concat(
local.cp_base_patches,
[local.cp_cluster_patch, local.cp_extras_patch],
local.cp_node_patches[count.index],
local.common_firewall_patches,
local.cp_firewall_patches,
)
@@ -349,16 +258,15 @@ data "talos_machine_configuration" "worker" {
config_patches = concat(local.worker_base_patches, local.common_firewall_patches)
}
# ── Bootstrap / kubeconfig / health (dial CP public IPs) ──────────────────────
# ── Bootstrap / kubeconfig / health ───────────────────────────────────────────
locals {
cp_public_ips = hcloud_server.control_plane[*].ipv4_address
# Everything the talos provider dials — bootstrap, kubeconfig, talosconfig, health,
# and the helm/kubernetes providers — uses the PRIVATE kube-cp IPs, reachable from
# the apply host over the NetBird kube-cp route (and in the cert SANs). No public
# access is required to bring the cluster up (the CPs keep public IPs only for
# NetBird NAT traversal + egress; apid/apiserver are firewalled off the internet).
bootstrap_endpoint = local.cp_private_ips[0]
operator_endpoint = "https://${local.cp_private_ips[0]}:6443"
# Everything the talos provider dials — bootstrap, kubeconfig, talosconfig,
# health, and the helm/kubernetes providers — uses the kube-cp IPs, reachable
# from the apply host over the NetBird kube-cp route (and in the cert SANs).
# During a greenfield bring-up the route appears as soon as the first CP boots
# into the cluster and joins the mesh (the CPs are the route peers).
bootstrap_endpoint = local.cp_ips[0]
operator_endpoint = "https://${local.cp_ips[0]}:6443"
}
# One-shot bootstrap against the first CP. Re-running rolls cluster identity.
@@ -368,7 +276,7 @@ resource "talos_machine_bootstrap" "this" {
endpoint = local.bootstrap_endpoint
timeouts = { create = "10m" }
depends_on = [hcloud_server.control_plane]
depends_on = [talos_machine_configuration_apply.cp]
lifecycle {
# A replace re-bootstraps a LIVE cluster (identity roll). The re-bootstrap
@@ -385,12 +293,12 @@ resource "talos_cluster_kubeconfig" "this" {
depends_on = [talos_machine_bootstrap.this]
}
# talosconfig endpoints = CP private kube-cp IPs (reached over NetBird; no public).
# talosconfig endpoints = CP kube-cp IPs (reached over NetBird; no public).
data "talos_client_configuration" "this" {
cluster_name = local.c.name
client_configuration = talos_machine_secrets.this.client_configuration
endpoints = local.cp_private_ips
nodes = concat(local.cp_private_ips, [for w in local.c.workers : w.fabric_ip])
endpoints = local.cp_ips
nodes = concat(local.cp_ips, [for w in local.c.workers : w.fabric_ip])
}
locals {
@@ -403,9 +311,9 @@ data "talos_cluster_health" "this" {
count = var.bootstrap_health_gate ? 1 : 0
client_configuration = talos_machine_secrets.this.client_configuration
control_plane_nodes = local.cp_private_ips
control_plane_nodes = local.cp_ips
worker_nodes = [for w in local.c.workers : w.fabric_ip]
endpoints = local.cp_private_ips
endpoints = local.cp_ips
skip_kubernetes_checks = true
timeouts = { read = "10m" }
@@ -4,11 +4,10 @@ include "root" {
# clusters.auto.tfvars is loaded automatically by OpenTofu in this directory.
# State backend + partition/region/stack (prod/htz-fsn1/talos) are derived by the
# root config. Unlike the austin talos stack (which talks straight to bare-metal
# nodes already in maintenance mode), this stack is HYBRID: it provisions the 3
# Hetzner Cloud control-plane VMs (+ a small private subnet for etcd + the API
# load balancer) AND drives Talos on the 3 bare-metal workers. CP↔worker traffic
# rides the NetBird mesh; worker east-west rides the 50G fabric. See ./README.md.
# root config. Like the austin talos stack, this talks straight to bare-metal
# nodes already in Talos maintenance mode: 3 CPs on the kube-cp fabric VLAN
# (etcd + API VIP) + 3 workers on the kube VLAN, routed by the spine IRBs.
# See ./README.md.
#
# Secrets (HCLOUD_TOKEN, NetBird setup key, S3 state) are injected by
# Secrets (NetBird setup keys, S3 state) are injected by
# op run --env-file=tf/.env.prod
+41 -47
View File
@@ -1,9 +1,10 @@
# Hybrid prod cluster topology — the single source of truth (clusters.auto.tfvars).
# One object, not a map: this stack's bring-up is bespoke (cloud CP via hcloud
# user_data + bare-metal workers via apid apply), so a for_each map buys nothing.
# All-bare-metal prod cluster topology — the single source of truth
# (clusters.auto.tfvars). One object, not a map: this stack's bring-up is bespoke
# (CPs + workers both driven over apid, but with different planes/volumes), so a
# for_each map buys nothing.
variable "cluster" {
description = "The prod hybrid Talos cluster (Star Wars name; prod = 'father')."
description = "The prod bare-metal Talos cluster (Star Wars name; prod = 'father')."
type = object({
name = string
talos_version = string
@@ -12,47 +13,51 @@ variable "cluster" {
# The Image Factory schematic (extension set) is managed in TF — see
# schematic.yaml + talos_image_factory_schematic in image.tf. The schematic id
# and image URLs derive from it, so they're NOT inputs here.
install_disk = string
cilium_version = string
hubble = bool
# NetBird mesh range node IPs come from (kubelet nodeIP.validSubnets). THIS
# account assigns 10.254.0.0/15 (see clusters.auto.tfvars) — not the NetBird
# Cloud default of 100.64.0.0/10. The node plane (CP↔worker) rides this mesh.
# NetBird mesh range (host firewall trust + operator plane). THIS account
# assigns 10.254.0.0/15 (see clusters.auto.tfvars) — not the NetBird Cloud
# default of 100.64.0.0/10.
netbird_node_cidr = string
# ── Cloud control plane (Hetzner Cloud) ──────────────────────────────────
cp_count = number # 3
# EXPLICIT node names (wordlist-style), one per CP, in cp_ip_offset order.
# Names are PINNED — never auto-shuffled — so node identity can't silently
# re-roll on a list edit (renaming a live node's hostname = renaming its
# Kubernetes node = effectively replacing it).
cp_names = list(string)
cp_server_type = string # ccx23 (dedicated vCPU x86)
cp_location = string # fsn1
cp_ip_offset = number # CP[i] private (kube-cp) IP = cidrhost(kube_cp, offset+i)
lb_type = string # lb11
lb_ip_offset = number # API LB private IP = cidrhost(kube_cp, offset)
lb_public = bool # also expose a public frontend (operators/workers reach it)
# ── Bare-metal workers (Hetzner Robot) ────────────────────────────────────
# maint_ip = the Hetzner public IP the node comes up on in Talos maintenance
# mode (DHCP) — the endpoint for the one-time config apply. fabric_ip = the
# post-install kube (VLAN 10) address (nodeIP + worker east-west); the apiserver
# reaches the kubelet there via NetBird→mgmt, and the node reaches the apiserver
# over its own NetBird peer.
workers = list(object({
name = string # EXPLICIT node name (see cp_names) — keys the apply resources; renaming = node replacement
# Install-disk NVMe serial — NOT a device name: nvme0/nvme1 enumeration is
# not stable across boots (observed swapping), and a name-based install
# target could point an upgrade at the DATA disk.
# ── Bare-metal control planes (Hetzner Robot; kube-cp fabric VLAN 11) ─────
# cp_ip = the post-install kube-cp (VLAN 11) address — etcd + apiserver +
# nodeIP; the spine routes kube↔kube-cp. maint_ip = the Hetzner public IP the
# node comes up on in Talos maintenance mode (DHCP on the onboard 1G NIC) —
# the endpoint for the one-time install apply.
cps = list(object({
name = string # EXPLICIT node name (wordlist-style) — keys the apply resources; renaming = node replacement
# Install-disk serial — NOT a device name: sda/sdb enumeration is not
# stable across boots, and a name-based install target could point an
# upgrade at the wrong disk.
install_serial = string
fabric_ip = string # 10.40.10.x on the kube fabric VLAN
cp_ip = string # 10.40.11.x on the kube-cp fabric VLAN
maint_ip = string # Hetzner public IP (maintenance-mode apid endpoint)
robot_id = number # Hetzner Robot server number (provisioning/doc)
provisioned = optional(bool, true) # false ONLY while first-provisioning: config
# applies then target maint_ip (maintenance mode); true = target fabric_ip (live).
# applies then target maint_ip (maintenance mode); true = target cp_ip (live).
}))
# CP fabric bond members (2×10G Intel 82599 SFP+). Selected by NIC driver —
# ixgbe matches exactly the two 10G ports (the onboard 1G public NIC is e1000e).
cp_bond_driver = optional(string)
cp_bond_interfaces = optional(list(string), [])
# API VIP = cidrhost(kube_cp, vip_offset) — Talos etcd-elected, floats between
# the CPs on VLAN 11. 5 keeps the retired hcloud LB's IP, so the api_dns_name
# record (NetBird DNS zone) carried over unchanged.
vip_offset = number
# ── Bare-metal workers (Hetzner Robot; kube fabric VLAN 10) ───────────────
# maint_ip/fabric_ip semantics as for cps; nodeIP = fabric_ip (worker east-west
# rides VLAN 10 at 50G, apiserver↔kubelet routes via the spine IRBs).
workers = list(object({
name = string
install_serial = string
fabric_ip = string # 10.40.10.x on the kube fabric VLAN
maint_ip = string
robot_id = number
provisioned = optional(bool, true)
}))
# Fabric bond members. Prefer worker_bond_driver (a Talos deviceSelector by NIC
# driver, e.g. "bnxt_en") — robust across per-node PCI naming. worker_bond_interfaces
@@ -62,21 +67,10 @@ variable "cluster" {
# Worker default route (egress for image pulls + NetBird): via the kube fabric
# IRB gateway (fabric transit) when true, else the Hetzner public NIC (DHCP).
worker_default_route_via_fabric = optional(bool, true)
# apiserver→kubelet rides the mesh peer-to-peer: the CPs get /etc/hosts entries
# mapping each worker hostname to its NetBird IP (data.netbird_peer lookups) and
# the apiserver prefers the Hostname node address. Requires the workers to BE
# NetBird peers — set false for a greenfield bootstrap (no peers to look up yet),
# flip true once the workers have joined. See cp_extras_patch in talos.tf.
worker_mesh_kubelet = optional(bool, true)
})
validation {
condition = length(var.cluster.cp_names) == var.cluster.cp_count
error_message = "cluster.cp_names must have exactly cp_count entries (one name per CP, in cp_ip_offset order)."
}
validation {
condition = length(distinct(concat(var.cluster.cp_names, var.cluster.workers[*].name))) == var.cluster.cp_count + length(var.cluster.workers)
condition = length(distinct(concat(var.cluster.cps[*].name, var.cluster.workers[*].name))) == length(var.cluster.cps) + length(var.cluster.workers)
error_message = "Node names must be unique across CPs and workers."
}
}
@@ -6,17 +6,6 @@ terraform {
source = "siderolabs/talos"
version = "~> 0.11"
}
# Hetzner Cloud — the 3 control-plane VMs, their private subnet (etcd), and the
# public API load balancer. Token via HCLOUD_TOKEN (op run --env-file).
hcloud = {
source = "hetznercloud/hcloud"
version = "~> 1.51"
}
# Hostname picks for the talos nodes (node-names module → random_shuffle).
random = {
source = "hashicorp/random"
version = "~> 3.6"
}
# Cilium install (CNI) post-bootstrap, in the same apply.
helm = {
source = "hashicorp/helm"
@@ -33,11 +22,5 @@ terraform {
source = "1Password/onepassword"
version = "~> 2.1"
}
# Worker NetBird peer lookups — the mesh addresses the CP apiserver dials for
# worker kubelets (see cp_extras_patch). Auth via NB_PAT (op run).
netbird = {
source = "registry.terraform.io/futo-org/netbird"
version = "1.0.2"
}
}
}
+18 -15
View File
@@ -1,18 +1,16 @@
# ── Bare-metal workers ───────────────────────────────────────────────────────
# Applied over apid to nodes already in Talos maintenance mode at their fabric_ip
# (reachable from the TF runner via NetBird → mgmt → fabric `kube` net). After the
# install+reboot they keep that fabric IP and join the NetBird mesh.
# or maint_ip (see below). After the install+reboot they keep that fabric IP and
# join the NetBird mesh (operator/backup plane only).
#
# bond0 (2×25G LACP) → vlan 10 (kube) = fabric_ip — nodeIP + worker east-west (50G)
# default route via the kube IRB gateway (fabric transit) for egress
# bond0 (2×25G LACP) → vlan 10 (kube) = fabric_ip — nodeIP + worker east-west (50G)
# route to kube-cp (apiserver + VIP) via the kube IRB (10.40.10.1) — the fabric
# path to the control plane; the old wt0 (NetBird) route is retired
# default route via the Hetzner public NIC (DHCP) for egress
#
# Workers are NOT NetBird peers: nodeIP = fabric_ip, so Cilium autoDirectNodeRoutes
# routes pod east-west directly over the shared VLAN-10 L2 (no BGP). The apiserver
# reaches worker kubelets via the CPs' NetBird route to the kube net (mgmt routers).
#
# Workers are PROVISIONED to maintenance mode out of band — see the Phase-4 runbook
# (./README.md): Hetzner rescue → dd the Talos metal image → bring up bond0.10 at
# fabric_ip. This stack assumes they're already there.
# Workers are PROVISIONED to maintenance mode out of band — see the runbook
# (./README.md): Hetzner rescue → dd the Talos metal image → reboot. This stack
# assumes they're already there.
locals {
kube_prefix = split("/", local.kube_cidr)[1] # 24
@@ -23,7 +21,7 @@ locals {
yamlencode({
machine = { install = {
diskSelector = { serial = w.install_serial }
image = local.worker_install_image
image = local.install_image
} }
}),
yamlencode({
@@ -47,9 +45,14 @@ locals {
vlans = [{
vlanId = module.addr_site.kube_vlan_id # 10
addresses = ["${w.fabric_ip}/${local.kube_prefix}"]
routes = var.cluster.worker_default_route_via_fabric ? [
{ network = "0.0.0.0/0", gateway = local.kube_gateway },
] : []
# kube-cp (apiserver + API VIP) lives one IRB away — pin the route so
# kubelet→apiserver + geneve to the CPs ride the fabric, not the mesh.
routes = concat(
[{ network = local.kube_cp_cidr, gateway = local.kube_gateway }],
var.cluster.worker_default_route_via_fabric ? [
{ network = "0.0.0.0/0", gateway = local.kube_gateway },
] : [],
)
}]
}]
}
+5 -5
View File
@@ -1,6 +1,6 @@
# Preprovisioned spine VC + the 100G->4x25G breakout. Breakout (channel-speed)
# has no typed jeremmfr resource, so it's pushed as raw set-config. The
# `aggregated-devices ethernet device-count` line is auto-managed by
# Preprovisioned spine VC + the per-port breakout channelization. Breakout
# (channel-speed) has no typed jeremmfr resource, so it's pushed as raw set-config.
# The `aggregated-devices ethernet device-count` line is auto-managed by
# junos_interface_physical (computed from the ae interfaces) — not set here.
resource "junos_virtual_chassis" "spine" {
preprovisioned = true
@@ -19,8 +19,8 @@ resource "junos_null_load_config" "breakout" {
action = "set"
config = join("\n", flatten([
for fpc in [0, 1] : [
for p in var.breakout_ports :
"set chassis fpc ${fpc} pic 0 port ${p} channel-speed ${var.breakout_speed}"
for p, speed in var.breakout_ports :
"set chassis fpc ${fpc} pic 0 port ${p} channel-speed ${speed}"
]
]))
}
@@ -74,6 +74,35 @@ resource "junos_interface_physical" "node_lag" {
vlan_members = ["vlan${var.kube_vlan_id}"]
}
# Control-plane node bonds — same pattern as node_lags, but the trunk carries the
# kube-cp VLAN (the CPs' only fabric presence; kube↔kube-cp routes via the IRBs).
locals {
cp_node_lag_members = merge([for ae, ports in var.cp_node_lags : { for p in ports : p => ae }]...)
}
resource "junos_interface_physical" "cp_node_lag_member" {
for_each = local.cp_node_lag_members
name = each.key
ether_opts {
ae_8023ad = each.value
}
}
resource "junos_interface_physical" "cp_node_lag" {
for_each = var.cp_node_lags
name = each.key
mtu = 9216
parent_ether_opts {
lacp {
mode = "active"
}
}
trunk = true
vlan_members = ["vlan${var.kube_cp.vlan_id}"]
depends_on = [junos_vlan.this]
}
# Management-node ports (mgmt-1, mgmt-2) — one channelized port-3 leg per VC member,
# each a single-port trunk of the stretched VLANs. Identical config per node.
resource "junos_interface_physical" "mgmt_node" {
+33 -9
View File
@@ -27,15 +27,14 @@ variable "vc_member_serials" {
}
variable "breakout_ports" {
type = list(number)
default = [0, 1, 2, 3]
description = "QSFP28 ports channelized 100G->4x25G on each VC member."
}
variable "breakout_speed" {
type = string
default = "25g"
description = "Per-channel speed for the breakout ports."
type = map(string)
default = { 0 = "25g", 1 = "25g", 2 = "25g", 3 = "25g" }
description = <<-EOT
QSFP28 ports channelized on each VC member: port number -> per-channel speed.
25g -> et-<fpc>/0/<port>:0..3 legs; 10g -> xe-<fpc>/0/<port>:0..3. NB: 10g
channelization needs a QSFP+ (40G-class) breakout cable — the QFX5200 silently
falls back to unchannelized 100G on a QSFP28 cable.
EOT
}
variable "kube_vlan_id" {
@@ -72,6 +71,31 @@ variable "node_lags" {
EOT
}
variable "cp_node_lags" {
type = map(list(string))
default = {}
description = <<-EOT
Control-plane node LACP bonds terminated on the core — same shape and rules as
node_lags (key = ae name; value = the two member sub-ports, one per VC member,
cabled to the SAME node), but the trunk carries the kube-cp VLAN instead of
kube. Requires var.kube_cp. Members are 10G breakout legs (xe-…).
EOT
}
variable "kube_cp" {
type = object({
vlan_id = number
cidr = string
})
default = null
description = <<-EOT
Kubernetes control-plane network on the fabric: creates the kube-cp VLAN + its
IRB (.1) on the spine — the second spine IRB, making the spine the router
between kube (workers) and kube-cp (bare-metal CPs: etcd + the API VIP).
null = no kube-cp VLAN.
EOT
}
variable "node_bgp" {
type = object({
peer_range = string # the kube CIDR — nodes dynamic-peer from it; the IRB is its .1
+18 -5
View File
@@ -1,14 +1,17 @@
# Stretched VLANs — L2 only on the spine EXCEPT the kube VLAN, which gets an IRB when
# node_bgp is set (the spine's only L3 interface, = the Cilium iBGP peer + VLAN-10
# gateway; see bgp-nodes.tf). Other gateways live on the leaves.
# Stretched VLANs — L2 only on the spine EXCEPT the kube VLAN (IRB when node_bgp
# is set: the Cilium iBGP peer + VLAN-10 gateway, see bgp-nodes.tf) and the
# kube-cp VLAN (IRB when kube_cp is set: the CPs' gateway — the spine routes
# kube↔kube-cp). Other gateways live on the leaves.
locals {
spine_vlans = {
spine_vlans = merge({
"vlan${var.public_vlan_id}" = { id = var.public_vlan_id, l3 = null }
"vlan${var.private_vlan_id}" = { id = var.private_vlan_id, l3 = null }
"vlan${var.kube_vlan_id}" = { id = var.kube_vlan_id, l3 = var.node_bgp == null ? null : "irb.${var.kube_vlan_id}" }
"vlan${var.mgmt_vlan_id}" = { id = var.mgmt_vlan_id, l3 = null }
"vlan${var.host_mgmt_vlan_id}" = { id = var.host_mgmt_vlan_id, l3 = null }
}
}, var.kube_cp == null ? {} : {
"vlan${var.kube_cp.vlan_id}" = { id = var.kube_cp.vlan_id, l3 = "irb.${var.kube_cp.vlan_id}" }
})
}
resource "junos_vlan" "this" {
@@ -17,3 +20,13 @@ resource "junos_vlan" "this" {
vlan_id = tostring(each.value.id)
l3_interface = each.value.l3
}
# kube-cp IRB — the spine is the kube-cp gateway (.1). Bare-metal CPs sit on this
# VLAN only; worker↔CP (kubelet↔apiserver, geneve) routes irb.<kube>↔irb.<kube-cp>.
resource "junos_interface_logical" "kube_cp_irb" {
count = var.kube_cp == null ? 0 : 1
name = "irb.${var.kube_cp.vlan_id}"
family_inet {
address { cidr_ip = "${cidrhost(var.kube_cp.cidr, 1)}/${split("/", var.kube_cp.cidr)[1]}" }
}
}
+8 -9
View File
@@ -11,16 +11,15 @@ locals {
kube_cidr = "10.${var.site_id}.${var.kube_octet}.0/24"
kube_vlan_id = var.kube_octet
# Site-global Kubernetes control-plane subnet ("kube-cp"). NOT a Juniper fabric
# VLAN — it's a small, isolated Hetzner Cloud private subnet holding ONLY the
# cloud control-plane VMs (for etcd CP↔CP) + the Kubernetes API LB's private IP.
# The workers do NOT join it: CP↔worker control + worker→API ride the NetBird
# WireGuard mesh (node IPs are NetBird addresses), and worker↔worker east-west
# rides the `kube` fabric net at 50G. Carved from the site supernet only for
# collision-free IPAM; it is NEVER configured on the Junos switches.
# kube-cp 10.<site>.<kube_cp_octet>.0/24 (Hetzner Cloud subnet, gw .1)
# Site-global Kubernetes control-plane VLAN ("kube-cp") — a fabric VLAN like
# `kube`: holds the bare-metal control-plane nodes (etcd CP↔CP + apiserver) and
# the cluster's API VIP. Workers do NOT join it — the spine routes kube↔kube-cp
# via its two IRBs. (Historically this was an isolated Hetzner Cloud private
# subnet for the cloud CP VMs + API LB; same CIDR, now on the switches.)
# kube-cp 10.<site>.<kube_cp_octet>.0/24 -> vlan <kube_cp_octet> (gw .1 = spine IRB)
kube_cp_cidr = "10.${var.site_id}.${var.kube_cp_octet}.0/24"
kube_cp_gateway = cidrhost(local.kube_cp_cidr, 1) # .1 — Hetzner Cloud Gateway
kube_cp_vlan_id = var.kube_cp_octet
kube_cp_gateway = cidrhost(local.kube_cp_cidr, 1) # .1 — spine IRB
# Internal (NetBird-only) Kubernetes LoadBalancer VIP range. Like kube-cp it is
# NEVER a switch VLAN: Cilium assigns VIPs from it and the workers advertise the
@@ -90,12 +90,17 @@ output "kube_vlan_id" {
output "kube_cp_cidr" {
value = local.kube_cp_cidr
description = "Site-global Kubernetes control-plane subnet 'kube-cp' (10.<site>.<kube_cp_octet>.0/24) — an isolated Hetzner Cloud private subnet for the CP VMs (etcd) + the API LB. NOT a fabric VLAN."
description = "Site-global Kubernetes control-plane network 'kube-cp' (10.<site>.<kube_cp_octet>.0/24), a fabric VLAN — bare-metal CPs (etcd) + the API VIP."
}
output "kube_cp_vlan_id" {
value = local.kube_cp_vlan_id
description = "Site-global kube-cp VLAN id (== kube_cp_octet, e.g. 11)."
}
output "kube_cp_gateway" {
value = local.kube_cp_gateway
description = "Hetzner Cloud Gateway (.1) for the kube-cp subnet."
description = "Spine IRB gateway (.1) for the kube-cp network."
}
output "lb_internal_cidr" {
+6 -4
View File
@@ -62,10 +62,12 @@ resource "netbox_prefix" "network" {
}
resource "netbox_ip_address" "gateway" {
for_each = local.networks
ip_address = "${each.value.gateway}/${split("/", each.value.prefix)[1]}"
status = "active"
dns_name = "gw-${var.site.code}-C${split("-", each.key)[0]}-${lower(each.value.role)}"
for_each = local.networks
ip_address = "${each.value.gateway}/${split("/", each.value.prefix)[1]}"
status = "active"
# lower(): NetBox normalizes dns_name to lowercase on write — mixed case here
# is a perpetual plan diff.
dns_name = lower("gw-${var.site.code}-c${split("-", each.key)[0]}-${each.value.role}")
description = "IRB gateway for ${var.site.code}-C${split("-", each.key)[0]}-${each.value.role}"
}