mirror of
https://github.com/immich-app/yucca.git
synced 2026-09-30 13:33:00 +08:00
feat(prod): continue prod (#267)
This commit is contained in:
@@ -26,12 +26,6 @@ opentofu = "1.11.5"
|
||||
terragrunt = "0.99.4"
|
||||
# ansible/mgmt convergence (mgmt:ansible task, run from CI on prod apply).
|
||||
"pipx:ansible-core" = "2.18.1"
|
||||
# Hetzner Cloud — build/upload the Talos hcloud snapshot + manage images
|
||||
# (hetzner:talos-image). hcloud-upload-image spins a temporary rescue server,
|
||||
# dd's the factory raw image, and snapshots it (hcloud can't boot the Talos ISO).
|
||||
hcloud = "1.66.0"
|
||||
"github:apricote/hcloud-upload-image" = "1.5.0"
|
||||
|
||||
[tasks.dev]
|
||||
description = "Start all services in development mode"
|
||||
depends = ["install:deps", "common:build", "docker:start"]
|
||||
|
||||
@@ -1,76 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
#MISE description="Build + upload the Talos hcloud snapshot for the prod cluster. The schematic is TF-managed (talos_image_factory_schematic); this reads its image URL from tofu output. Idempotent unless FORCE=1."
|
||||
# hcloud can't boot the Talos ISO, so the CP VMs need a Talos *snapshot*.
|
||||
# hcloud-upload-image builds it the only way possible: spin a temporary rescue
|
||||
# server, dd the factory hcloud-amd64 raw image, snapshot, tear down. The talos
|
||||
# stack's data.hcloud_image then resolves it by label.
|
||||
#
|
||||
# mise run hetzner:talos-image # build for the prod father cluster
|
||||
# FORCE=1 mise run hetzner:talos-image # rebuild even if a snapshot exists
|
||||
set -euo pipefail
|
||||
ROOT=$(git rev-parse --show-toplevel)
|
||||
STACK="${TALOS_STACK:-tf/deployment/prod/htz-fsn1/talos}"
|
||||
TFVARS="$ROOT/$STACK/clusters.auto.tfvars"
|
||||
[ -f "$TFVARS" ] || { echo "talos-image: tfvars not found: $TFVARS" >&2; exit 1; }
|
||||
|
||||
# OVH S3 backend cert verification on macOS (same shim as the infra:* tasks).
|
||||
if [ -z "${SSL_CERT_FILE:-}" ] && [ "$(uname -s)" = "Darwin" ] && [ -f /etc/ssl/cert.pem ]; then
|
||||
export SSL_CERT_FILE=/etc/ssl/cert.pem AWS_CA_BUNDLE=/etc/ssl/cert.pem
|
||||
fi
|
||||
|
||||
val() { grep -E "^[[:space:]]*$1[[:space:]]*=" "$TFVARS" | head -1 | sed -E 's/[^"]*"([^"]+)".*/\1/'; }
|
||||
NAME=$(val name); VERSION=$(val talos_version); LOCATION=$(val cp_location)
|
||||
SELECTOR="os=talos,cluster=${NAME},arch=amd64,version=${VERSION}"
|
||||
|
||||
tg() { OP_ENV_FILE=tf/.env.prod "$ROOT/tf/op-run.sh" terragrunt --working-dir "$STACK" "$@"; }
|
||||
|
||||
# The schematic is TF-managed (talos_image_factory_schematic). Pull its id straight
|
||||
# from state (grep the exact `id =` line — robust against terragrunt log noise);
|
||||
# if it isn't applied yet, create just it (targeted, dependency-free — touches no
|
||||
# servers/data sources) and re-read. Then build the factory URLs from the id.
|
||||
read_schematic_id() {
|
||||
tg state show -no-color talos_image_factory_schematic.this 2>/dev/null \
|
||||
| grep -E '^[[:space:]]+id[[:space:]]+=' | head -1 | sed -E 's/.*"([^"]+)".*/\1/'
|
||||
}
|
||||
ID=$(read_schematic_id || true)
|
||||
if [ -z "$ID" ]; then
|
||||
echo "talos-image: registering the Image Factory schematic in TF (targeted apply)…"
|
||||
tg apply -target=talos_image_factory_schematic.this -auto-approve
|
||||
ID=$(read_schematic_id)
|
||||
fi
|
||||
[ -n "$ID" ] || { echo "talos-image: could not resolve the schematic id from $STACK" >&2; exit 1; }
|
||||
URL="https://factory.talos.dev/image/${ID}/v${VERSION}/hcloud-amd64.raw.xz"
|
||||
METAL_URL="https://factory.talos.dev/image/${ID}/v${VERSION}/metal-amd64.raw.xz"
|
||||
|
||||
# Read-only hcloud token from 1Password (CI: OP_SERVICE_ACCOUNT_TOKEN; dev:
|
||||
# interactive team-futo sign-in — same pattern as infra:plan).
|
||||
if [ -z "${HCLOUD_TOKEN:-}" ]; then
|
||||
if [ -n "${OP_SERVICE_ACCOUNT_TOKEN:-}" ]; then
|
||||
HCLOUD_TOKEN=$(op read "op://yucca_tf_prod/HCLOUD_API_TOKEN/password")
|
||||
else
|
||||
HCLOUD_TOKEN=$(op read --account "${OP_ACCOUNT:-team-futo}" "op://yucca_tf_prod/HCLOUD_API_TOKEN/password")
|
||||
fi
|
||||
export HCLOUD_TOKEN
|
||||
fi
|
||||
|
||||
echo "cluster=$NAME talos=v$VERSION schematic=$ID"
|
||||
echo " hcloud image : $URL"
|
||||
echo " metal image : $METAL_URL (workers dd this in rescue)"
|
||||
|
||||
# Idempotency: skip when a matching snapshot already exists.
|
||||
EXISTING=$(hcloud image list --type snapshot --selector "$SELECTOR" \
|
||||
--output noheader --output columns=id 2>/dev/null || true)
|
||||
if [ -n "$EXISTING" ] && [ -z "${FORCE:-}" ]; then
|
||||
echo "talos-image: snapshot already exists (id ${EXISTING}). Set FORCE=1 to rebuild."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
echo "talos-image: uploading the Talos v$VERSION hcloud snapshot (spins a temporary server)…"
|
||||
hcloud-upload-image upload \
|
||||
--image-url "$URL" \
|
||||
--architecture x86 \
|
||||
--compression xz \
|
||||
--location "$LOCATION" \
|
||||
--labels "$SELECTOR"
|
||||
|
||||
echo "talos-image: done — data.hcloud_image.talos (selector os=talos,cluster=${NAME},arch=amd64) now resolves."
|
||||
@@ -86,7 +86,8 @@ kind: CiliumBGPClusterConfig
|
||||
metadata:
|
||||
name: father
|
||||
spec:
|
||||
# Workers only — the CPs live on the hcloud net, not the fabric, so they can't peer.
|
||||
# Workers only — the CPs live on the kube-cp VLAN; the spine's iBGP group peers
|
||||
# from the kube VLAN (10.40.10.0/24) only.
|
||||
nodeSelector:
|
||||
matchExpressions:
|
||||
- key: node-role.kubernetes.io/control-plane
|
||||
|
||||
@@ -0,0 +1,16 @@
|
||||
---
|
||||
# prod@htz-fsn1 INFRA layer — the CRD/operator providers (cert-manager,
|
||||
# envoy-gateway + Gateway API, OpenEBS/mayastor) + their namespaces. Applied by
|
||||
# the `cluster-infra` Flux Kustomization (clusters/prod/htz-fsn1/apps.yaml),
|
||||
# which `cluster-apps` dependsOn: everything here must be READY (healthChecks →
|
||||
# operators running → CRDs registered) before the app layer's custom resources
|
||||
# (Certificates, Gateways, EnvoyProxy, DiskPools) are even dry-run — the flat
|
||||
# single-layer tree deadlocked on exactly that during the 2026-07 rebuild.
|
||||
apiVersion: kustomize.config.k8s.io/v1beta1
|
||||
kind: Kustomization
|
||||
resources:
|
||||
- ./namespace-envoy-system.yaml
|
||||
- ./namespace-openebs.yaml
|
||||
- ./cert-manager.yaml
|
||||
- ./envoy.yaml
|
||||
- ./openebs.yaml
|
||||
@@ -1,13 +1,15 @@
|
||||
---
|
||||
# prod@htz-fsn1 cluster overlay — father.
|
||||
# prod@htz-fsn1 cluster overlay — father: the APP layer (custom resources +
|
||||
# workloads). The CRD/operator providers live in ./infra, applied by the
|
||||
# `cluster-infra` Flux Kustomization that this layer dependsOn — see
|
||||
# infra/kustomization.yaml for why (fresh-cluster dry-run deadlock).
|
||||
#
|
||||
# CURRENT SCOPE: Flux owns the cluster BASELINE (coredns, the Cilium BGP LB
|
||||
# config, the netops stack, OpenEBS pools) — adopted from the hand-applied
|
||||
# bring-up state. The platform/infra components and the yucca app set are
|
||||
# DELIBERATELY not enabled yet:
|
||||
# config, the netops stack, OpenEBS pools). The platform/infra components and
|
||||
# the yucca app set are DELIBERATELY not enabled yet:
|
||||
# - components/infra needs real cluster-settings (RGW endpoint, o11y vmauth)
|
||||
# and would collide with the in-cluster OpenEBS install (openebs/ here owns
|
||||
# it via HelmRelease instead).
|
||||
# and would collide with the in-cluster OpenEBS install (infra/openebs.yaml
|
||||
# owns it via HelmRelease instead).
|
||||
# - components/roles/primary is the yucca WORKLOAD set — explicitly held back
|
||||
# until prod launch.
|
||||
# Re-enable by uncommenting `components:` below.
|
||||
@@ -18,10 +20,7 @@ resources:
|
||||
- ./coredns.yaml
|
||||
- ./cilium-bgp.yaml
|
||||
- ./lb-return-route.yaml
|
||||
- ./cert-manager.yaml
|
||||
- ./namespace-envoy-system.yaml
|
||||
- ./envoy.yaml
|
||||
- ./openebs
|
||||
- ./diskpools.yaml
|
||||
- ./netops
|
||||
# components:
|
||||
# - ../../../components/infra
|
||||
|
||||
@@ -26,7 +26,7 @@ spec:
|
||||
app: lb-return-route
|
||||
spec:
|
||||
hostNetwork: true
|
||||
# Workers only — the CPs are on the hcloud net, not the fabric.
|
||||
# Workers only — the CPs are on the kube-cp VLAN, not the kube VLAN.
|
||||
affinity:
|
||||
nodeAffinity:
|
||||
requiredDuringSchedulingIgnoredDuringExecution:
|
||||
|
||||
@@ -3,9 +3,8 @@
|
||||
# http://lg.father.fsn.htz.yucca.futo.network
|
||||
#
|
||||
# devices.yaml is NOT here: it embeds the netops password (netmiko can't key-auth
|
||||
# through hyperglass config), so it's a Secret rendered at deploy time from
|
||||
# op://yucca_tf_prod/NETOPS_FABRIC_PASSWORD (see the deploy script / runbook):
|
||||
# kubectl -n netops create secret generic hyperglass-devices --from-file=devices.yaml
|
||||
# through hyperglass config), so it's a Secret the talos stack renders from
|
||||
# op://yucca_tf_prod/NETOPS_FABRIC_PASSWORD (netops-secrets.tf).
|
||||
apiVersion: v1
|
||||
kind: ConfigMap
|
||||
metadata:
|
||||
|
||||
@@ -1,11 +1,11 @@
|
||||
---
|
||||
# The father netops stack. Secrets (netops-ssh, grafana-admin, hyperglass-devices)
|
||||
# are NOT in git — they are created from 1Password (yucca_tf_prod: NETOPS_FABRIC_*,
|
||||
# FATHER_GRAFANA_ADMIN) at bring-up; Flux only manages the workloads around them.
|
||||
# The father netops stack. The namespace + its Secrets (netops-ssh,
|
||||
# grafana-admin, hyperglass-devices) are NOT in git — the talos stack provisions
|
||||
# them from 1Password (tf/deployment/prod/htz-fsn1/talos/netops-secrets.tf);
|
||||
# Flux only manages the workloads around them.
|
||||
apiVersion: kustomize.config.k8s.io/v1beta1
|
||||
kind: Kustomization
|
||||
resources:
|
||||
- ./namespace.yaml
|
||||
- ./networkpolicies.yaml
|
||||
- ./junos-exporter.yaml
|
||||
- ./victoria-metrics.yaml
|
||||
|
||||
@@ -1,19 +0,0 @@
|
||||
# netops — in-cluster network operations stack for the htz-fsn1 fabric:
|
||||
# junos_exporter + vmagent + VictoriaMetrics (30d high-granularity buffer) +
|
||||
# Grafana, plus hyperglass/smokeping/oxidized. Everything is exposed ONLY on
|
||||
# internal LoadBalancer VIPs (lb-internal pool, NetBird-reachable — never public).
|
||||
# Applied by hand today; to be adopted by Flux when GitOps lands on father.
|
||||
#
|
||||
# privileged PodSecurity: VictoriaMetrics persists to a hostPath (no CSI on
|
||||
# father yet — the NVMe data disks are unprovisioned); baseline forbids hostPath.
|
||||
apiVersion: v1
|
||||
kind: Namespace
|
||||
metadata:
|
||||
name: netops
|
||||
labels:
|
||||
pod-security.kubernetes.io/enforce: privileged
|
||||
annotations:
|
||||
# Never prune: this namespace holds hand-created secrets (netops-ssh,
|
||||
# grafana-admin, hyperglass-devices) and the VictoriaMetrics PVC — a prune
|
||||
# would destroy state only a manual runbook can restore.
|
||||
kustomize.toolkit.fluxcd.io/prune: disabled
|
||||
@@ -1,7 +0,0 @@
|
||||
---
|
||||
apiVersion: kustomize.config.k8s.io/v1beta1
|
||||
kind: Kustomization
|
||||
resources:
|
||||
- ./namespace.yaml
|
||||
- ./openebs.yaml
|
||||
- ./diskpools.yaml
|
||||
@@ -1,10 +1,55 @@
|
||||
# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/kustomize.toolkit.fluxcd.io/kustomization_v1.json
|
||||
---
|
||||
# cluster-apps entry point for prod@htz-fsn1. Authored ahead of the cluster (the
|
||||
# prod Talos/flux stack isn't built yet); activates once that stack provisions
|
||||
# flux and it syncs kubernetes/clusters/prod/htz-fsn1. Precedence (last wins):
|
||||
# cluster-settings-generated (TF) -> cluster-settings (human) -> image-versions
|
||||
# (the committed, CI-promoted prod tag); keys are disjoint by design.
|
||||
# prod@htz-fsn1 entry points — TWO layers, so a fresh cluster can't deadlock:
|
||||
# the 2026-07 rebuild proved a flat tree wedges itself (kustomize-controller
|
||||
# server-side dry-runs every object, so any CR whose CRD is missing blocks the
|
||||
# WHOLE apply — including the HelmReleases that would install those CRDs).
|
||||
#
|
||||
# cluster-infra operators/CRD providers (apps/prod/htz-fsn1/infra) — wait: true,
|
||||
# so Ready ⇒ operators running ⇒ CRDs registered.
|
||||
# cluster-apps everything else (CRs + workloads) — dependsOn cluster-infra.
|
||||
#
|
||||
# Substitution precedence (last wins): cluster-settings-generated (TF) ->
|
||||
# cluster-settings (human) -> image-versions (the committed, CI-promoted prod
|
||||
# tag); keys are disjoint by design. Injected into the NESTED Kustomizations via
|
||||
# the patches block (both layers' direct objects use no ${vars} themselves).
|
||||
apiVersion: kustomize.toolkit.fluxcd.io/v1
|
||||
kind: Kustomization
|
||||
metadata:
|
||||
name: cluster-infra
|
||||
namespace: flux-system
|
||||
spec:
|
||||
interval: 1h
|
||||
retryInterval: 2m
|
||||
path: ./kubernetes/apps/prod/htz-fsn1/infra
|
||||
prune: true
|
||||
sourceRef:
|
||||
kind: GitRepository
|
||||
name: flux-system
|
||||
namespace: flux-system
|
||||
# The health gate cluster-apps depends on: every nested Kustomization here
|
||||
# carries healthChecks on its HelmRelease, so wait covers operator readiness.
|
||||
wait: true
|
||||
timeout: 10m
|
||||
patches:
|
||||
- target:
|
||||
group: kustomize.toolkit.fluxcd.io
|
||||
kind: Kustomization
|
||||
patch: |-
|
||||
apiVersion: kustomize.toolkit.fluxcd.io/v1
|
||||
kind: Kustomization
|
||||
metadata:
|
||||
name: _
|
||||
spec:
|
||||
postBuild:
|
||||
substituteFrom:
|
||||
- kind: ConfigMap
|
||||
name: cluster-settings-generated
|
||||
- kind: ConfigMap
|
||||
name: cluster-settings
|
||||
- kind: ConfigMap
|
||||
name: image-versions
|
||||
---
|
||||
apiVersion: kustomize.toolkit.fluxcd.io/v1
|
||||
kind: Kustomization
|
||||
metadata:
|
||||
@@ -12,11 +57,12 @@ metadata:
|
||||
namespace: flux-system
|
||||
spec:
|
||||
interval: 1h
|
||||
# Fast retry: this Kustomization applies CRs (Certificates, Gateways,
|
||||
# DiskPools) whose CRDs its own children install — on fresh bootstrap the
|
||||
# first apply races them, and without retryInterval a failure waits the
|
||||
# full 1h interval.
|
||||
# Fast retry: CRD registration can trail the infra layer's Ready by moments
|
||||
# (mayastor's diskpool operator creates its CRD at startup) — without
|
||||
# retryInterval a dry-run failure waits the full 1h interval.
|
||||
retryInterval: 2m
|
||||
dependsOn:
|
||||
- name: cluster-infra
|
||||
path: ./kubernetes/apps/prod/htz-fsn1
|
||||
prune: true
|
||||
sourceRef:
|
||||
|
||||
@@ -32,11 +32,6 @@ export HETZNER_ROBOT_PASSWORD=op://yucca_tf_prod/HETZNER_WEBSERVICE_API_PASSWORD
|
||||
# one). The cloudflare provider reads CLOUDFLARE_API_TOKEN directly.
|
||||
export CLOUDFLARE_API_TOKEN=op://yucca_tf_prod/CLOUDFLARE_API_TOKEN/password
|
||||
|
||||
# ── Hetzner Cloud API (prod/htz-fsn1/talos — hcloud provider) ─────────────────
|
||||
# Per-project read/write token for the control-plane VMs, network, snapshot + LB.
|
||||
# TODO(prod): create the hcloud project + token, store at the path below.
|
||||
export HCLOUD_TOKEN=op://yucca_tf_prod/HCLOUD_API_TOKEN/password
|
||||
|
||||
# ── NetBird setup keys (prod/htz-fsn1/talos — node-level overlay) ─────────────
|
||||
# Minted by the netbird stack. WORKER key (group: talos) — joins the bare-metal
|
||||
# workers to the prod htz-fsn1 NetBird network.
|
||||
|
||||
@@ -2,7 +2,8 @@
|
||||
|
||||
Manages the Falkenstein (site 40) switch fabric as code:
|
||||
|
||||
- **spine** (`corenetsw` VC) — shared site core: VC, 100G→4×25G breakout, VLAN stretch.
|
||||
- **spine** (`corenetsw` VC) — shared site core: VC, per-port breakout (ports 1-3
|
||||
100G→4×25G, port 0 100G→4×10G for the father CPs), VLAN stretch.
|
||||
- **cls1** (`cls1netsw` VC) — ceph cluster 1's leaf pair: public/private VLANs, IRB
|
||||
gateways, the `NO-CROSS-VLAN` filter, and the 48 server LAGs.
|
||||
|
||||
@@ -15,22 +16,24 @@ Each ceph cluster = one leaf pair; the spine is shared across clusters.
|
||||
| site supernet | `10.<site>.0.0/16` | `10.40.0.0/16` |
|
||||
| management (vme) | `10.<site>.5.0/24` | `10.40.5.0/24` (spine `.115`, leaf `.125`) |
|
||||
| kube (VLAN) | `10.<site>.<kube_octet>.0/24` | `10.40.10.0/24` → vlan 10 |
|
||||
| kube-cp (hcloud) | `10.<site>.<kube_cp_octet>.0/24` | `10.40.11.0/24` (gw `.1`, CP VMs etcd + API LB) |
|
||||
| kube-cp (VLAN) | `10.<site>.<kube_cp_octet>.0/24` | `10.40.11.0/24` → vlan 11 (gw `.1` = spine IRB) |
|
||||
| cluster `n` /20 | `10.<site>.<n*16>.0/20` | `10.40.16.0/20` |
|
||||
| public (VLAN) | cluster /20, /23 idx 2 | `10.40.20.0/23` → vlan 20 |
|
||||
| private (VLAN) | cluster /20, /23 idx 3 | `10.40.22.0/23` → vlan 22 |
|
||||
| leaf vme | `.125 + (n-1)*10` | `.125` |
|
||||
|
||||
VLAN id == the network's third octet; gateway = `.1` (IRB on the leaf).
|
||||
VLAN id == the network's third octet; gateway = `.1` (IRB on the leaf, except the
|
||||
site-global kube/kube-cp VLANs, whose IRBs live on the spine).
|
||||
|
||||
> **`kube-cp` is not a fabric VLAN.** It's a small isolated **Hetzner Cloud private
|
||||
> subnet** holding only the cloud control-plane VMs (etcd CP↔CP) + the API LB's
|
||||
> private IP. CP↔worker control traffic and worker→API ride the **NetBird WireGuard
|
||||
> mesh** (node IPs are NetBird addresses), and worker↔worker east-west rides the
|
||||
> `kube` fabric net (`10.40.10.0/24`) at 50G via Cilium BGP. The API endpoint is a
|
||||
> **Hetzner Cloud LB** (no L2 VIP — hcloud private nets are anti-spoofed/routed).
|
||||
> `kube-cp` is carved from the site supernet only for collision-free IPAM and is
|
||||
> **never** configured on the Junos switches.
|
||||
> **`kube-cp` is the control-plane VLAN.** The father bare-metal CPs are its only
|
||||
> members (etcd CP↔CP + apiserver + the Talos-elected API **VIP** `10.40.11.5`),
|
||||
> hanging off the spine's port-0 **4×10G** breakout (ae4-6, one leg per VC member).
|
||||
> The spine routes kube↔kube-cp between its two IRBs (`10.40.10.1` / `10.40.11.1`),
|
||||
> which is how worker kubelets and the apiserver reach each other; worker↔worker
|
||||
> east-west rides the `kube` VLAN at 50G. Operators reach the API over the NetBird
|
||||
> kube-cp route (the CPs are the route peers). Historically kube-cp was an isolated
|
||||
> Hetzner Cloud subnet for the retired cloud CP VMs + API LB — same CIDR, so the
|
||||
> API DNS record and etcd addressing carried over unchanged.
|
||||
|
||||
## Layout
|
||||
|
||||
|
||||
@@ -22,6 +22,11 @@ module "core" {
|
||||
|
||||
vc_member_serials = var.spine_vc_serials
|
||||
|
||||
# Port 0 carries the father control-plane breakout at 10G (the CPs' Intel 82599
|
||||
# NICs are 10G-only; needs the QSFP+ 4x10G breakout cables — see cp_node_lags);
|
||||
# ports 1-3 stay 25G (workers + mgmt + spares).
|
||||
breakout_ports = { 0 = "10g", 1 = "25g", 2 = "25g", 3 = "25g" }
|
||||
|
||||
# father's bare-metal kube workers hang off the core (channelized 25G breakouts of
|
||||
# port 2, one leg per VC member). Each ae bundles the two ports cabled to one node
|
||||
# (pairs derived from LLDP — consecutive MACs on the node's dual-port Broadcom NIC):
|
||||
@@ -32,6 +37,23 @@ module "core" {
|
||||
ae3 = ["et-0/0/2:1", "et-1/0/2:1"]
|
||||
}
|
||||
|
||||
# father's bare-metal control planes: port-0 breakout legs at 10G (xe-), one leg
|
||||
# per VC member, trunking the kube-cp VLAN. Pairing VERIFIED 2026-07-15 via MAC
|
||||
# learning against the maintenance-mode nodes (NIC port 1 → FPC 0, port 2 →
|
||||
# FPC 1, same leg index on both members):
|
||||
# ae4 = harlan …0a:fe:c8/ca ae5 = imelda …09:68:68/6a ae6 = roscoe …65:07:40/42
|
||||
cp_node_lags = {
|
||||
ae4 = ["xe-0/0/0:2", "xe-1/0/0:2"]
|
||||
ae5 = ["xe-0/0/0:1", "xe-1/0/0:1"]
|
||||
ae6 = ["xe-0/0/0:0", "xe-1/0/0:0"]
|
||||
}
|
||||
|
||||
# kube-cp VLAN + its spine IRB (10.40.11.1) — the spine routes kube↔kube-cp.
|
||||
kube_cp = {
|
||||
vlan_id = module.addr_site.kube_cp_vlan_id
|
||||
cidr = module.addr_site.kube_cp_cidr
|
||||
}
|
||||
|
||||
# Cilium node iBGP for LoadBalancer VIPs — the spine gets its first IRB (the kube net's
|
||||
# .1 gateway) and dynamic-peers the workers from the kube subnet, accepting the LB /32s
|
||||
# they advertise (covered by the transit aggregate, so reachable north-south). The
|
||||
@@ -54,6 +76,8 @@ module "core" {
|
||||
interfaces = [
|
||||
"et-0/0/2:1", "et-0/0/2:2", "et-0/0/2:3",
|
||||
"et-1/0/2:1", "et-1/0/2:2", "et-1/0/2:3",
|
||||
"xe-0/0/0:0", "xe-0/0/0:1", "xe-0/0/0:2",
|
||||
"xe-1/0/0:0", "xe-1/0/0:1", "xe-1/0/0:2",
|
||||
"et-0/0/3:0", "et-1/0/3:0",
|
||||
"et-0/0/27",
|
||||
"et-0/0/30", "et-0/0/31", "et-1/0/30", "et-1/0/31",
|
||||
|
||||
@@ -14,8 +14,9 @@ module "netbox" {
|
||||
|
||||
# Site-global VLANs (present on every cluster).
|
||||
global_vlans = {
|
||||
MGMT = { vid = module.addr_site.mgmt_vlan_id, prefix = module.addr_site.mgmt_cidr }
|
||||
KUBE = { vid = module.addr_site.kube_vlan_id, prefix = module.addr_site.kube_cidr }
|
||||
MGMT = { vid = module.addr_site.mgmt_vlan_id, prefix = module.addr_site.mgmt_cidr }
|
||||
KUBE = { vid = module.addr_site.kube_vlan_id, prefix = module.addr_site.kube_cidr }
|
||||
"KUBE-CP" = { vid = module.addr_site.kube_cp_vlan_id, prefix = module.addr_site.kube_cp_cidr }
|
||||
}
|
||||
|
||||
clusters = {
|
||||
@@ -37,7 +38,6 @@ module "netbox" {
|
||||
# Pod/service CIDRs mirror the talos stack (talos.tf locals); the public carves
|
||||
# mirror the Cilium LB pools + node-egress + transit config in this stack.
|
||||
extra_prefixes = {
|
||||
kube_cp = { prefix = module.addr_site.kube_cp_cidr, description = "Hetzner Cloud kube-cp: father CP VMs (etcd) + private API LB — not a fabric VLAN" }
|
||||
lb_internal = { prefix = module.addr_site.lb_internal_cidr, description = "father internal (NetBird-only) LoadBalancer VIPs — Cilium lb-internal pool, iBGP /32s to the spine" }
|
||||
pods = { prefix = "10.250.0.0/17", description = "father pod CIDR (Cilium, geneve over the kube VLAN)", status = "container" }
|
||||
services = { prefix = "10.250.128.0/17", description = "father service CIDR (ClusterIPs; kube-dns at .128.10)", status = "container" }
|
||||
|
||||
@@ -96,16 +96,17 @@ resource "netbird_dns_record" "father_worker" {
|
||||
ttl = 300
|
||||
}
|
||||
|
||||
# API endpoint — round-robin over the 3 CP IPs. NOT the LB: hcloud LBs refuse
|
||||
# traffic from their own targets (the CPs), so the endpoint resolves straight to
|
||||
# the CPs (reachable over the yucca-fsn-father-kube-cp route).
|
||||
# API endpoint — the Talos-elected VIP on the kube-cp VLAN (etcd parks it on a
|
||||
# healthy CP, so the record only answers where an apiserver runs). Reachable over
|
||||
# the yucca-fsn-father-kube-cp route. (Historically round-robin over the CP IPs —
|
||||
# the retired hcloud LB refused traffic from its own targets.)
|
||||
resource "netbird_dns_record" "father_kube_api" {
|
||||
for_each = local.father_cps
|
||||
zone_id = netbird_dns_zone.yucca_internal.id
|
||||
name = local.father_kube_api_fqdn
|
||||
type = "A"
|
||||
content = each.value
|
||||
ttl = 300
|
||||
count = var.talos_discovery_enabled ? 1 : 0
|
||||
zone_id = netbird_dns_zone.yucca_internal.id
|
||||
name = local.father_kube_api_fqdn
|
||||
type = "A"
|
||||
content = local.talos_kube.api_vip
|
||||
ttl = 300
|
||||
}
|
||||
|
||||
output "kube_api_fqdn" {
|
||||
|
||||
@@ -20,8 +20,8 @@ groups = {
|
||||
talos = { resource = true } # Talos cluster nodes → yucca-prod-htz-fsn1-talos
|
||||
resources = { resource = true } # routed-subnet tag → yucca-prod-htz-fsn1-resources (Network resources tag in)
|
||||
# CP-only subset of `talos` — the ROUTER peer group for the kube-cp network. Only
|
||||
# the cloud CPs sit on the kube-cp hcloud subnet, so only they can route it; if the
|
||||
# router were the whole `talos` group the bare-metal WORKERS (also `talos`) would be
|
||||
# the CPs sit on the kube-cp VLAN, so only they can route it; if the router were
|
||||
# the whole `talos` group the bare-metal WORKERS (also `talos`) would be
|
||||
# treated as routers and never install the client route to kube-cp. resource = false:
|
||||
# it's a routing peer group, not a yucca-reachable tag (the CPs are already reachable
|
||||
# via `talos`). CPs join via the talos_cp setup key below (auto_groups tags them
|
||||
|
||||
@@ -16,13 +16,9 @@ locals {
|
||||
# group (flagged `resource = true` in netbird.auto.tfvars), so the module-
|
||||
# generated yucca→resources policy governs access — and resources never appear
|
||||
# as a policy source, so they can't reach each other.
|
||||
# NB: the `kube-cp` network (the routed Hetzner Cloud Network for the cloud
|
||||
# control-plane VMs) is deliberately NOT advertised here. The router peers are the
|
||||
# mgmt nodes, which sit on the Juniper fabric and cannot reach the hcloud subnets.
|
||||
# The CP plane is reached out-of-band via hcloud public IPs + the API LB's public
|
||||
# frontend (both firewalled to the NetBird/operator ranges) — see the talos stack.
|
||||
# TODO(prod): for a fully-private control plane, attach the mgmt nodes to the
|
||||
# kube-cp vSwitch and add module.addr_site.kube_cp_cidr to this map.
|
||||
# NB: the `kube-cp` VLAN is deliberately NOT in this map — it's routed by its
|
||||
# own network below (via the CPs, the talos_cp group), keeping the API plane's
|
||||
# mesh path independent of the mgmt routers.
|
||||
routed = {
|
||||
mgmt = { address = module.addr_site.mgmt_cidr, description = "OOB / vme management network" }
|
||||
# Internal LB VIPs (Grafana + netops UIs): NetBird peer -> mgmt router -> spine
|
||||
@@ -48,24 +44,25 @@ locals {
|
||||
}
|
||||
}
|
||||
|
||||
# father's cloud control-plane subnet (kube-cp), routed via the CPs ONLY (the
|
||||
# talos_cp group — the CP-only subset of talos). They're the only peers on that
|
||||
# hcloud subnet. Router must NOT be the whole `talos` group: the bare-metal workers
|
||||
# are also `talos`, and a routing peer doesn't install a client route for its own
|
||||
# network — so if the workers were routers they'd never get the kube-cp route (and
|
||||
# their pods couldn't reach the apiserver). This is how NetBird peers (operators +
|
||||
# workers) reach the private API LB (10.40.11.5) + the CPs. masquerade so return
|
||||
# traffic is SNAT'd to the CP's kube-cp address.
|
||||
# father's control-plane VLAN (kube-cp), routed via the CPs ONLY (the talos_cp
|
||||
# group — the CP-only subset of talos). They're the only peers on that VLAN.
|
||||
# Router must NOT be the whole `talos` group: the bare-metal workers are also
|
||||
# `talos`, and a routing peer doesn't install a client route for its own
|
||||
# network — so if the workers were routers they'd never get the kube-cp mesh
|
||||
# route. (Worker→apiserver traffic itself rides the fabric — a static route via
|
||||
# the spine IRB pinned in the machine config — not this mesh route.) This is
|
||||
# how OPERATOR/CI peers reach the API VIP (10.40.11.5) + the CPs. masquerade so
|
||||
# return traffic is SNAT'd to the CP's kube-cp address.
|
||||
# CP membership comes from the talos_cp setup key (netbird.auto.tfvars, auto_groups
|
||||
# [talos, talos_cp]); the talos stack joins CPs with it and workers with the plain
|
||||
# `talos` key, so re-provisioning keeps the split.
|
||||
"yucca-fsn-father-kube-cp" = {
|
||||
description = "father control-plane subnet (kube-cp), routed via the CPs (talos_cp)."
|
||||
description = "father control-plane VLAN (kube-cp), routed via the CPs (talos_cp)."
|
||||
router = { peer_groups = ["talos_cp"], masquerade = true }
|
||||
resources = {
|
||||
kube_cp = {
|
||||
address = module.addr_site.kube_cp_cidr
|
||||
description = "kube-cp: CP VMs (etcd) + the private API LB (10.40.11.5)."
|
||||
description = "kube-cp: bare-metal CPs (etcd) + the API VIP (10.40.11.5)."
|
||||
groups = ["resources"]
|
||||
}
|
||||
}
|
||||
|
||||
@@ -64,50 +64,6 @@ provider "registry.opentofu.org/hashicorp/kubernetes" {
|
||||
]
|
||||
}
|
||||
|
||||
provider "registry.opentofu.org/hashicorp/random" {
|
||||
version = "3.9.0"
|
||||
constraints = "~> 3.6"
|
||||
hashes = [
|
||||
"h1:U8KXqGCoNI9/guYbTvzgdtVk3fRthoG0UXwm1JoEpIs=",
|
||||
"zh:03f1114cc20b8913523735ab76e0f0a2b16ce13c92923a53304bf85f07fc0dbc",
|
||||
"zh:105b678ee72322a3067f105d7e05e940f6143238f377f6e87ff4ec909246ac2a",
|
||||
"zh:55f3bbf13ea18cbace61a706566a80f25f33fe2b1780b6f3d7b582af2a05b6d2",
|
||||
"zh:63adf996db48f082f7a6351eb485e219cd88795fc71e6ec60a837263ab0d2cb1",
|
||||
"zh:7e99550738a4e3cc68b8a467714b0d69371025fe95e3326d5323d026d55653e9",
|
||||
"zh:8342b54af3a18a37e075eeae61be57f4de2ba71b35d95c5075d402dd2c1f289d",
|
||||
"zh:83ee18e32ac9dd5fc91298554b7c4cfa4c3a1db50f4c797945637cc93c0844ae",
|
||||
"zh:993ecc0adbf6bd535a59fbc9b735d8c33950e6f6eb5e621d750da9b71d65d80a",
|
||||
"zh:ad722bc59d4edbf1415e827fc007c0efe6e0e9462d5568bae20b34be1058a261",
|
||||
"zh:ae9448e1f87b2f9a6c5197a0e9862162ec6b137cb3a3835e11522995d8939e7c",
|
||||
"zh:bc9cdd3aac784f759125c6627f6f6416e8726a1c184eb9cf3e55b9edbc94c627",
|
||||
"zh:c8e35b89572ba1c40a9b20022e033a3395fb8d42e7604d50c900f193ba10382e",
|
||||
"zh:e2deaa8a9975ef81d9f62baed12c41286918b0a10908e0e031f13f69a3b730a1",
|
||||
"zh:ee39707557210a0ab1098aa357d2cdfe502e5a312d0dbdffb09d08facc4d3fc5",
|
||||
"zh:f81afe4eb63e8aa9e0ea71be6c990f0dc69cb360e7191c0742a991f4a5081b64",
|
||||
]
|
||||
}
|
||||
|
||||
provider "registry.opentofu.org/hetznercloud/hcloud" {
|
||||
version = "1.66.0"
|
||||
constraints = "~> 1.51"
|
||||
hashes = [
|
||||
"h1:iVAGP8gRbZK0kJF7SiYJRt61wz0D5AF9q+WMsrAiBI0=",
|
||||
"zh:1286cee6fb63dbcb18f53077bbb5e5d132a4e4d9f006af4e8d8edfc08d6bcdc8",
|
||||
"zh:204460dacc044bda019a4a18b398e094289500c36913c7c9457f432adf31b8b2",
|
||||
"zh:214175d50773481cbeaf9c9004e4121a3a1c9686c79424ebdc8ff189dd057d3e",
|
||||
"zh:22b17bceff61cc13ad04a399ba87521356a3a134d4687273727473ae9eccf5f1",
|
||||
"zh:368867dac5525c411de7e38f2e27de0a71854d1750867322ff2b9321128c88fb",
|
||||
"zh:5289b75f8370bdbc4c6051d55cf33d0b1bd25dc6d71bfbd39b360249a37f1501",
|
||||
"zh:81cb676aa50c5777df8fc80d4e69c9012330ae751f5e6f12bf6074bfd2e7c496",
|
||||
"zh:ab08aead10643b21aa6b51af562b50492e12b9dd0ab7dca27a05aa63209b7d66",
|
||||
"zh:af25c210d0570cf61ef767b2545bf9f3fb909178135f0e5e14bec0c1c9d07a63",
|
||||
"zh:bcad66f4830c97118fa793723e53f8a4d27ddd34ea969ff259408842c2238331",
|
||||
"zh:ce3ed323d75ae905d975925fa98c7054a7514c81276a485fc37da8232b53e39f",
|
||||
"zh:d481bc0ef0c87ab1969c17777f526b2f59f823432d676145134c41a6d29bd98e",
|
||||
"zh:ea7ef88df2c3ca154d86238920636d52a3c9066c7467543d3fa45f1e52ec2f7b",
|
||||
]
|
||||
}
|
||||
|
||||
provider "registry.opentofu.org/siderolabs/talos" {
|
||||
version = "0.11.0"
|
||||
constraints = "~> 0.11"
|
||||
@@ -130,24 +86,3 @@ provider "registry.opentofu.org/siderolabs/talos" {
|
||||
]
|
||||
}
|
||||
|
||||
provider "registry.terraform.io/futo-org/netbird" {
|
||||
version = "1.0.2"
|
||||
constraints = "1.0.2"
|
||||
hashes = [
|
||||
"h1:CE91Uvc3FhpgpwgjVR96ueerThTZSCTcKF8Rp2+lzQw=",
|
||||
"zh:1a1b727dd3971eb0f4c923c19fa95100e5b91b163ae0607f72a37d656f9d71a6",
|
||||
"zh:5169993e54c6b184cdb359fb310477d85eeb99d1d91db14d367a29bf3119dc55",
|
||||
"zh:573279b9532f916ee4bfb4e12faacf26e8e27df626ac37e1a5c42d5eab2350a4",
|
||||
"zh:5cafd8cb073fc6774d959080c06b6faf821c3a7caaf3d876a40be8b8945a70fd",
|
||||
"zh:66d40bbde6d756160a94e8c908582cb7806acccbebe0e9060b82cb4993f5adad",
|
||||
"zh:6eb6493922345ec4306d8f0e95e799a6045ff3b25fd05973828e960757df5b97",
|
||||
"zh:88c638c29d7dad339f1c4e90e54d0996be85a95574d5d9c5f125cb0bbb5b8902",
|
||||
"zh:890df766e9b839623b1f0437355032a3c006226a6c200cd911e15ee1a9014e9f",
|
||||
"zh:89cca24eeb5305cfcac51092bd0c330336af6613648f14a4c70f23a56fe4e93d",
|
||||
"zh:91186b07d2ee1494bcd7e2c132ca1c61b62b2a1f2d325d2bbc4bfe6572ce327d",
|
||||
"zh:aa6cbe8f5c8121d96861e293d3773cc210723fa410acf45512a337703d55d911",
|
||||
"zh:d38dce0bcc61ca2482ee3b9f457ebf45929c7ebc8717487d9f1270aa3dcd3cfa",
|
||||
"zh:dba9742f0a0cb5e190a10265ec00d20a4d266f8a6dd517cb4c1d8486335ed636",
|
||||
"zh:e8eaabc072208c10ab554b4c89dbe32831a7d0125ab628af4a6530a02ea5a993",
|
||||
]
|
||||
}
|
||||
|
||||
@@ -1,92 +1,156 @@
|
||||
# prod/htz-fsn1/talos — the `father` cluster
|
||||
|
||||
Hybrid production Talos Kubernetes cluster: **3 Hetzner Cloud control-plane VMs +
|
||||
3 Hetzner Robot bare-metal workers**, one cluster. Flux is **deferred** — this
|
||||
stack brings up a healthy, Cilium-networked cluster and stops there.
|
||||
All-bare-metal production Talos Kubernetes cluster: **3 Hetzner Robot control
|
||||
planes + 3 Hetzner Robot workers**, every node plane on the Juniper fabric. The
|
||||
previous hybrid topology (3 Hetzner Cloud CP VMs + hcloud API LB + NetBird as the
|
||||
CP↔worker plane) is retired; NetBird remains on every node as the **operator /
|
||||
backup plane** only.
|
||||
|
||||
## Topology
|
||||
|
||||
```
|
||||
┌──────────── Hetzner Cloud (fsn1) ────────────┐
|
||||
│ cp-1 cp-2 cp-3 (CCX23, Talos snapshot) │
|
||||
│ • public IPv4 → NetBird + bootstrap + LB │
|
||||
│ • kube-cp 10.40.11.0/24 (eth1) → etcd │
|
||||
│ • API LB (lb11) → :6443, PRIVATE (lb_public=false) │
|
||||
└───────┬───────────────────────┬───────────────┘
|
||||
NetBird mesh│ (CP route to fabric │ private LB :6443 (api_dns_name,
|
||||
via mgmt │ net, advertised) │ reached over the mesh)
|
||||
┌───────┴───────────────────────┴───────────────┐
|
||||
│ Juniper fabric — kube VLAN 10 / 10.40.10.0/24 │
|
||||
│ wk-1 .11 wk-2 .12 wk-3 .13 (bond0, 50G) │
|
||||
│ • pod↔pod rides geneve (routingMode: tunnel) over │
|
||||
│ the 50G fabric between workers │
|
||||
└─────────────────────────────────────────────────┘
|
||||
┌──────────────── Juniper fabric (site 40) ─────────────────┐
|
||||
│ kube-cp VLAN 11 — 10.40.11.0/24 (gw .1 = spine IRB) │
|
||||
│ cp-harlan .11 cp-imelda .12 cp-roscoe .13 │
|
||||
│ API VIP 10.40.11.5 (Talos etcd-elected) │
|
||||
│ bond0 2×10G (ixgbe, spine port-0 4×10G breakout, ae4-6) │
|
||||
├───────────────── spine routes irb.11 ↔ irb.10 ────────────┤
|
||||
│ kube VLAN 10 — 10.40.10.0/24 (gw .1 = spine IRB) │
|
||||
│ wk-jeanne .11 wk-sheron .12 wk-dianna .13 │
|
||||
│ bond0 2×25G (bnxt_en, spine port-2 4×25G breakout, ae1-3)│
|
||||
└───────────────────────────────────────────────────────────┘
|
||||
every node: onboard public NIC (DHCP) = default route/egress
|
||||
+ NetBird (operator plane; CPs route kube-cp)
|
||||
```
|
||||
|
||||
| Plane | Path | Carries |
|
||||
|---|---|---|
|
||||
| node / control | NetBird (CP) → mgmt routers → `kube` fabric | apiserver↔kubelet, CP→worker |
|
||||
| API endpoint | private Hetzner Cloud LB (`api_dns_name`, NetBird-reachable) | kubelet→apiserver, operators |
|
||||
| etcd | `kube-cp` hcloud private subnet (`10.40.11.0/24`) | CP↔CP |
|
||||
| pod east-west | `kube` fabric VLAN 10 (50G), geneve tunnel | worker↔worker pods |
|
||||
| etcd + API endpoint | `kube-cp` VLAN 11, VIP `10.40.11.5` (`api_dns_name` → VIP) | CP↔CP etcd, apiserver, VIP |
|
||||
| CP↔worker | routed `kube`↔`kube-cp` via the spine IRBs (static routes in machine config) | apiserver↔kubelet, geneve |
|
||||
| pod east-west | `kube` VLAN 10 (50G), geneve tunnel | worker↔worker pods |
|
||||
| operators / CI | NetBird → kube-cp route (the CPs are the route peers) | talosctl/kubectl/TF |
|
||||
|
||||
No vSwitch, no BGP. **Cilium BGP is reserved for north-south later** (advertising
|
||||
ingress/LoadBalancer VIPs to the fabric leaf, which already speaks BGP).
|
||||
**Cilium BGP** (north-south LoadBalancer VIPs) stays workers-only against the
|
||||
spine's VLAN-10 IRB.
|
||||
|
||||
## Bring-up flow (single `tf:apply`)
|
||||
|
||||
1. `image.tf` — look up the Talos amd64 hcloud snapshot (built once out-of-band).
|
||||
2. `network.tf` — hcloud network + the kube-cp cloud subnet.
|
||||
3. `controlplane.tf` — 3 CP VMs (config via `user_data`) + the API LB.
|
||||
4. `talos.tf` — `talos_machine_bootstrap` against cp-1's public IP → kubeconfig.
|
||||
5. `workers.tf` — `talos_machine_configuration_apply` to each worker over apid
|
||||
(its fabric IP, reached via NetBird→mgmt→fabric).
|
||||
6. `cilium.tf` — Cilium via Helm → post-CNI health gate → Ready cluster.
|
||||
1. `image.tf` — register the schematic (one, metal, **no qemu-guest-agent**).
|
||||
2. `controlplane.tf` — apid config apply to each CP (maintenance `maint_ip` on the
|
||||
first pass) → install to disk (by serial) + reboot onto bond0.11.
|
||||
3. `talos.tf` — `talos_machine_bootstrap` against cp-1 (`10.40.11.11`, reached
|
||||
over the NetBird kube-cp route once cp-1's netbird is up) → kubeconfig.
|
||||
4. `workers.tf` — apid apply to each worker (maintenance `maint_ip` first pass).
|
||||
5. `cilium.tf` — Cilium via Helm → post-CNI health gate.
|
||||
6. `flux.tf` — flux-operator + instance → GitOps takes over
|
||||
(`kubernetes/clusters/prod/htz-fsn1`).
|
||||
|
||||
## Prerequisites (before `tf:apply`)
|
||||
|
||||
- **`HCLOUD_TOKEN`** — create the hcloud project + read/write token, store at
|
||||
`op://yucca_tf_prod/HCLOUD_API_TOKEN` (see `tf/.env.prod`).
|
||||
- **Talos schematic** — the extension set lives in `schematic.yaml` and is
|
||||
registered with the factory by TF (`talos_image_factory_schematic`, `image.tf`);
|
||||
the schematic id + image URLs derive from it. Edit `schematic.yaml` to change it.
|
||||
- **hcloud snapshot** — build it once with `mise run hetzner:talos-image` (reads the
|
||||
image URL from `tofu output`; idempotent, `FORCE=1` to rebuild). `image.tf`
|
||||
resolves it by label.
|
||||
- **NetBird setup key** — minted by the netbird stack; path in `tf/.env.prod`.
|
||||
- **Workers in maintenance mode** — provisioned to Talos maintenance at their
|
||||
`fabric_ip` (10.40.10.11/.12/.13). See the runbook below.
|
||||
- **DNS** — after apply, point `api_dns_name` (output) at the LB public IPv4
|
||||
(output `api_dns_record`).
|
||||
- **`trusted_cidrs`** — MUST include the source the TF runner dials the CP public
|
||||
IPs from (CI egress / your NetBird range), or bootstrap (apid 50000) hangs.
|
||||
- **Fabric** — the fabric stack applied with `breakout_ports` port 0 = 10g, the
|
||||
kube-cp VLAN/IRB, and `cp_node_lags` ae4-6. Leg pairing in `../fabric/fabric.tf`
|
||||
was verified 2026-07-15 by MAC-learning against the maintenance-mode nodes
|
||||
(QSFP+ 4×10G breakout cables installed; all six legs link at 10G). Re-verify if
|
||||
anything is re-cabled — LACP won't aggregate legs facing different nodes.
|
||||
- **Nodes in maintenance mode** — every node with `provisioned = false` must be
|
||||
in Talos maintenance at its `maint_ip`. Rescue → dd runbook below.
|
||||
- **NetBird setup keys** — minted by the netbird stack; paths in `tf/.env.prod`.
|
||||
- **Operator/CI NetBird networks selected** — the apply host reaches the CPs via
|
||||
the `yucca-fsn-father-kube-cp` NetBird network (and the switches/workers via
|
||||
`htz-fsn1-mgmt` / `htz-fsn1-kube`). With client ≥0.75 lazy network selection,
|
||||
`netbird networks select <name>` or routes silently don't install.
|
||||
- **`trusted_cidrs`** — MUST include the source the TF runner dials the CPs from
|
||||
(the NetBird range), or bootstrap (apid 50000) hangs.
|
||||
|
||||
> CI owns `tf:apply` (`.github/workflows/infra.yml`). Locally use `tf:plan` only.
|
||||
|
||||
## Phase-4: worker provisioning runbook (rescue → Talos maintenance)
|
||||
## Node provisioning runbook (rescue → Talos maintenance)
|
||||
|
||||
The workers (Robot server numbers 3008210/11/12) must boot Talos in maintenance
|
||||
mode at their fabric IP before this stack applies. Per worker:
|
||||
Per node (CPs: Robot 3027819/3027863/3028524; workers: 3008210/11/12):
|
||||
|
||||
1. Robot → enable the **rescue system** (linux64) for the server, reboot into it.
|
||||
2. `dd` the Talos **metal** image for the cluster schematic onto the boot disk:
|
||||
1. Robot → enable the **rescue system** (linux64), reboot into it.
|
||||
2. `dd` the Talos **metal** image for the cluster schematic onto the install disk
|
||||
(`tofu output talos_metal_image_url`):
|
||||
```sh
|
||||
wget -O /tmp/talos.raw.xz \
|
||||
"https://factory.talos.dev/image/<schematic-id>/v1.13.4/metal-amd64.raw.xz"
|
||||
xz -dc /tmp/talos.raw.xz | dd of=/dev/sda bs=4M && sync
|
||||
wget -O /tmp/talos.raw.xz "$(tofu output -raw talos_metal_image_url)"
|
||||
xz -dc /tmp/talos.raw.xz | dd of=/dev/sda bs=4M conv=fsync && sync
|
||||
```
|
||||
3. Reboot off the rescue system → Talos comes up in maintenance mode.
|
||||
4. Bring up `bond0` over the two 25G NICs with the tagged **kube VLAN 10** carrying
|
||||
the node's `fabric_ip` (matching `clusters.auto.tfvars`), so the TF runner can
|
||||
reach apid (50000) over the fabric. (This stack then pins the same config.)
|
||||
Record the disk's SERIAL (`lsblk -d -o NAME,SERIAL`) — it pins
|
||||
`install_serial` in `clusters.auto.tfvars`.
|
||||
3. Reboot off the rescue system → Talos comes up in maintenance mode on the
|
||||
public NIC (DHCP) = the node's `maint_ip`. (If it lands back in rescue, the
|
||||
rescue flag didn't clear — just reboot again.)
|
||||
4. Set the node's `provisioned = false` in tfvars → apply → flip to `true` once
|
||||
it has joined.
|
||||
|
||||
This mirrors the mgmt-host reprovision pattern (`../mgmt-hosts.yaml` + the fabric
|
||||
stack's `mgmt.tf`); a future iteration can drive it from TF/Ansible.
|
||||
## Cutover runbook (hybrid → all-bare-metal REBUILD) — EXECUTED 2026-07-15
|
||||
|
||||
The rebuild keeps `talos_machine_secrets` (cluster PKI) but re-bootstraps etcd on
|
||||
the new CPs and re-installs the workers. **In-cluster state (Mayastor/localpv) is
|
||||
lost**; Flux redeploys everything. The steps below were executed 2026-07-15 (all
|
||||
stacks now plan clean); kept as the reference for any future rebuild. Order
|
||||
matters:
|
||||
|
||||
1. **Fabric first** (CI orders fabric before talos): apply lands the kube-cp
|
||||
VLAN/IRB + ae4-6. Port-0 10g channelization is already live on the spine
|
||||
(set 2026-07-15, identical to the TF config) and the leg pairing is verified —
|
||||
after the apply, `show lacp interfaces` should show ae4-6 collecting once the
|
||||
CPs boot their bonds.
|
||||
2. **New CPs in maintenance mode** (done 2026-07-15): rescue → dd → maintenance
|
||||
at 178.63.124.20/.21/.22.
|
||||
3. **State surgery** (forgets, no destroys — safe with prevent_destroy):
|
||||
```sh
|
||||
tofu state rm talos_machine_bootstrap.this # re-bootstrap on the new cp-1
|
||||
tofu state rm talos_cluster_kubeconfig.this
|
||||
tofu state rm helm_release.cilium helm_release.flux_operator helm_release.flux_instance
|
||||
tofu state rm kubernetes_secret_v1.github_app kubernetes_namespace_v1.cert_manager kubernetes_secret_v1.cloudflare_api_token
|
||||
```
|
||||
4. **Reset the workers** to maintenance mode (wipes them — deliberate):
|
||||
```sh
|
||||
talosctl -n 10.40.10.11 reset --graceful=false --reboot \
|
||||
--system-labels-to-wipe STATE --system-labels-to-wipe EPHEMERAL # × each worker
|
||||
```
|
||||
They come back in maintenance at their `maint_ip` (public DHCP).
|
||||
5. **Merge/apply this stack**: destroys the hcloud CP VMs + LB + network +
|
||||
firewall (their prevent_destroy left with the deleted config), applies CP
|
||||
configs → bootstrap → workers → Cilium → Flux.
|
||||
6. Flip every node's `provisioned = true` once joined; rotate operator
|
||||
kubeconfigs (`op read`, secrets.tf rewrote them).
|
||||
|
||||
Rebuild gotchas hit on 2026-07-15 (expect them again):
|
||||
- **The first apply fails partway** — helm dials the apiserver seconds after
|
||||
bootstrap (connection refused) and the kubernetes/1P resources throw
|
||||
"inconsistent final plan" (provider config unknowable at plan). Just re-apply;
|
||||
nothing is damaged. A helm wait-timeout can strand a `failed` release
|
||||
("cannot re-use a name that is still in use") — `helm uninstall` it first.
|
||||
- **flux-operator waits on the first worker** (CPs are unschedulable) — worker
|
||||
install+join takes longer than helm's 5m wait. Re-apply once workers are Ready.
|
||||
- **CoreDNS chicken-and-egg**: `cluster.coreDNS.disabled=true` from t=0 means NO
|
||||
cluster DNS until Flux deploys ours — but flux-operator needs DNS to fetch its
|
||||
manifests from ghcr.io. Break the cycle once per rebuild:
|
||||
`kubectl apply -f kubernetes/apps/prod/htz-fsn1/coredns.yaml` (the exact
|
||||
objects Flux owns — it adopts them unchanged).
|
||||
- ~~CRD deadlock (flat kustomization)~~ — fixed structurally after the rebuild:
|
||||
the tree is layered (`cluster-infra` = operators/CRD providers with
|
||||
`wait: true`; `cluster-apps` dependsOn it — see clusters/prod/htz-fsn1/
|
||||
apps.yaml). A fresh cluster converges without manual CRD pre-installs; the
|
||||
only remaining hand-step is the CoreDNS one above (it predates Flux itself).
|
||||
- **Stale NetBird peers**: re-provisioned nodes join as NEW peers; the old
|
||||
same-named peers linger disconnected and break the netbird stack's
|
||||
`data.netbird_peer` lookups ("cannot match multiple peers"). Delete the
|
||||
disconnected duplicates (API/console) before applying the netbird stack.
|
||||
- **`talosctl reset --wait=false`** — the default wait can never complete (the
|
||||
node comes back at a different IP, in maintenance mode).
|
||||
- ~~Hand-created netops secrets die with the cluster~~ — fixed: the netops
|
||||
namespace + netops-ssh/grafana-admin/hyperglass-devices Secrets are TF-owned
|
||||
now (netops-secrets.tf, sourced from 1P), restored by the normal apply.
|
||||
|
||||
## Notes
|
||||
|
||||
- **Bootstrap is one-shot.** Re-applying does not re-bootstrap; tainting
|
||||
`talos_machine_bootstrap.this` re-rolls cluster identity — don't.
|
||||
- CP VMs have `ignore_changes = [user_data, image]` so re-applies don't recycle
|
||||
live nodes; change them deliberately (cordon/drain first).
|
||||
- Flux activates later from `kubernetes/clusters/prod/htz-fsn1` (already scaffolded).
|
||||
- **Bootstrap is one-shot.** Re-applying does not re-bootstrap; replacing
|
||||
`talos_machine_bootstrap.this` re-rolls cluster identity — don't (the state-rm
|
||||
in the cutover is the deliberate exception).
|
||||
- The API VIP is etcd-elected: it exists only while a healthy CP holds it. The
|
||||
bootstrap/operator path deliberately dials cp-1's IP, not the VIP.
|
||||
- The old hcloud snapshot build task (`hetzner:talos-image`) is retired — all
|
||||
nodes boot the factory **metal** image via the rescue-dd runbook.
|
||||
|
||||
@@ -1,16 +1,23 @@
|
||||
# Site IP plan — the single source of truth (same module the fabric + netbird
|
||||
# stacks read). Gives us, derived from site_id (40):
|
||||
# addr_site.kube_cidr 10.40.10.0/24 — fabric VLAN 10, worker east-west (50G)
|
||||
# addr_site.kube_cp_cidr 10.40.11.0/24 — isolated hcloud subnet, CP etcd + API LB
|
||||
# addr_site.kube_cp_cidr 10.40.11.0/24 — fabric VLAN 11, CP etcd + API VIP
|
||||
# Nothing is hardcoded here; addresses below are cidrhost() offsets into these.
|
||||
module "addr_site" {
|
||||
source = "../../../../shared/modules/fabric-addressing"
|
||||
site_id = var.site_id
|
||||
}
|
||||
|
||||
# Cluster-1 view — only for the leaf vme address (netops-secrets.tf hyperglass).
|
||||
module "addr_cls1" {
|
||||
source = "../../../../shared/modules/fabric-addressing"
|
||||
site_id = var.site_id
|
||||
cluster_id = 1
|
||||
}
|
||||
|
||||
locals {
|
||||
kube_cidr = module.addr_site.kube_cidr # 10.40.10.0/24 (fabric)
|
||||
kube_gateway = cidrhost(module.addr_site.kube_cidr, 1) # .1 IRB on the cluster leaf
|
||||
kube_cp_cidr = module.addr_site.kube_cp_cidr # 10.40.11.0/24 (hcloud)
|
||||
kube_cp_gw = module.addr_site.kube_cp_gateway # .1 Hetzner Cloud Gateway
|
||||
kube_cidr = module.addr_site.kube_cidr # 10.40.10.0/24 (fabric VLAN 10)
|
||||
kube_gateway = cidrhost(module.addr_site.kube_cidr, 1) # .1 IRB on the spine
|
||||
kube_cp_cidr = module.addr_site.kube_cp_cidr # 10.40.11.0/24 (fabric VLAN 11)
|
||||
kube_cp_gw = module.addr_site.kube_cp_gateway # .1 IRB on the spine
|
||||
}
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
# Cilium for the hybrid prod cluster. Talos sets cni:none + proxy:disabled, so
|
||||
# nodes stay NotReady until this lands the datapath.
|
||||
# Cilium for the prod cluster. Talos sets cni:none + proxy:disabled, so nodes
|
||||
# stay NotReady until this lands the datapath.
|
||||
#
|
||||
# TUNNEL (geneve) routing — see the routingMode block below for the full why:
|
||||
# the cluster spans two L2 domains (CPs on kube-cp hcloud, workers on the fabric
|
||||
# VLAN) bridged only by NetBird, which native routing/autoDirectNodeRoutes can't
|
||||
# the cluster spans two L2 domains (CPs on the kube-cp VLAN, workers on the kube
|
||||
# VLAN) routed by the spine IRBs, which autoDirectNodeRoutes (same-L2-only) can't
|
||||
# span. Worker↔worker east-west still rides the 50G fabric, just encapsulated.
|
||||
#
|
||||
# securityContext + cgroup blocks are MANDATORY on Talos. Ref: Talos "Deploying Cilium".
|
||||
@@ -12,17 +12,17 @@ ipam:
|
||||
mode: kubernetes
|
||||
|
||||
# Tunnel (geneve) routing. The cluster spans two L2 domains — CPs on the kube-cp
|
||||
# hcloud net, workers on the fabric VLAN — bridged only by NetBird. Native routing
|
||||
# can't span that; a geneve overlay carries pod↔pod over whatever host-to-host path
|
||||
# exists (the 50G fabric for worker↔worker, NetBird for CP↔worker). East-west still
|
||||
# rides the fabric, just encapsulated (~50B overhead, negligible at 50G).
|
||||
# VLAN (11), workers on the kube VLAN (10) — routed by the spine's IRBs.
|
||||
# autoDirectNodeRoutes only works within one L2; a geneve overlay carries pod↔pod
|
||||
# over the routed node-to-node path instead. East-west still rides the fabric,
|
||||
# just encapsulated (~50B overhead, negligible at 50G).
|
||||
routingMode: tunnel
|
||||
tunnelProtocol: geneve
|
||||
|
||||
# Masquerade pod traffic to non-pod destinations (internet, node IPs); pod↔pod is
|
||||
# tunnelled so it isn't masqueraded. ip-masq-agent excludes only the pod CIDR, so the
|
||||
# geneve underlay + node-IP traffic egressing wt0 is SNAT'd to the node's NetBird
|
||||
# address (which WireGuard accepts). Requires BPF masquerade on.
|
||||
# tunnelled so it isn't masqueraded. ip-masq-agent excludes only the pod CIDR, so
|
||||
# node-IP traffic egressing wt0 (NetBird, the operator plane) is still SNAT'd to
|
||||
# the node's mesh address (which WireGuard accepts). Requires BPF masquerade on.
|
||||
enableIPv4Masquerade: true
|
||||
bpf:
|
||||
masquerade: true
|
||||
|
||||
@@ -1,7 +1,8 @@
|
||||
# Cilium CNI — installed post-bootstrap in the same apply (helm provider bound to
|
||||
# the bootstrap CP, providers.tf). Talos set cni:none + proxy:disabled, so nodes
|
||||
# go Ready only once this lands the datapath. Native routing + autoDirectNodeRoutes
|
||||
# keeps worker east-west on the 50G fabric (cilium-values.yaml.tftpl).
|
||||
# go Ready only once this lands the datapath. Geneve tunnel routing spans the two
|
||||
# routed fabric VLANs (kube/kube-cp); worker east-west still rides the 50G fabric
|
||||
# (cilium-values.yaml.tftpl).
|
||||
resource "helm_release" "cilium" {
|
||||
name = "cilium"
|
||||
namespace = "kube-system"
|
||||
@@ -27,9 +28,9 @@ data "talos_cluster_health" "post_cni" {
|
||||
count = var.bootstrap_health_gate ? 1 : 0
|
||||
|
||||
client_configuration = talos_machine_secrets.this.client_configuration
|
||||
control_plane_nodes = local.cp_private_ips
|
||||
control_plane_nodes = local.cp_ips
|
||||
worker_nodes = [for w in var.cluster.workers : w.fabric_ip]
|
||||
endpoints = local.cp_private_ips
|
||||
endpoints = local.cp_ips
|
||||
|
||||
timeouts = { read = "10m" }
|
||||
|
||||
|
||||
@@ -1,11 +1,12 @@
|
||||
# ── prod hybrid Talos cluster: `father` ──────────────────────────────────────
|
||||
# Topology (see README.md): 3 Hetzner Cloud CP VMs + 3 Hetzner Robot bare-metal
|
||||
# workers. CP↔worker rides the NetBird mesh; worker↔worker east-west rides the
|
||||
# 50G fabric (kube VLAN 10) via Cilium BGP; etcd CP↔CP on a private hcloud subnet;
|
||||
# API via a public Hetzner Cloud LB.
|
||||
# ── prod bare-metal Talos cluster: `father` ──────────────────────────────────
|
||||
# Topology (see README.md): 3 Hetzner Robot bare-metal CPs on the kube-cp fabric
|
||||
# VLAN 11 (etcd + API VIP 10.40.11.5) + 3 Hetzner Robot bare-metal workers on the
|
||||
# kube fabric VLAN 10. The spine routes kube↔kube-cp between its IRBs; NetBird
|
||||
# stays on every node as the operator/backup plane (kube-cp is routed to the mesh
|
||||
# via the CPs). No Hetzner Cloud anywhere — the cloud CP VMs + API LB are retired.
|
||||
#
|
||||
# Adding/replacing a node = edit here + `tf:plan` (CI applies). Workers must
|
||||
# already be in Talos maintenance mode at their fabric_ip (see Phase-4 runbook).
|
||||
# Adding/replacing a node = edit here + `tf:plan` (CI applies). Nodes must
|
||||
# already be in Talos maintenance mode at their maint_ip (see the runbook).
|
||||
|
||||
cluster = {
|
||||
name = "father" # prod K8s cluster (Star Wars; staging = luke)
|
||||
@@ -14,7 +15,6 @@ cluster = {
|
||||
|
||||
# The node extension set lives in schematic.yaml (managed via
|
||||
# talos_image_factory_schematic in image.tf) — no schematic id to paste here.
|
||||
install_disk = "/dev/sda" # CP install target (hcloud VMs: virtio /dev/sda) — the ONLY consumer is cp_install_patch; workers install by NVMe serial (workers.tf)
|
||||
|
||||
cilium_version = "1.19.5"
|
||||
hubble = true
|
||||
@@ -22,23 +22,23 @@ cluster = {
|
||||
# NetBird peer address range for this deployment (firewall trust for the mesh).
|
||||
netbird_node_cidr = "10.254.0.0/15"
|
||||
|
||||
# ── Cloud control plane (Hetzner Cloud, fsn1) ──────────────────────────────
|
||||
cp_count = 3
|
||||
# PINNED to the live nodes (verified against discovery/cp_nodes 2026-07-07).
|
||||
# Order follows cp_ip_offset: kaycee=.11, bettie=.12, ofelia=.13. A wrong name
|
||||
# here RENAMES a live control plane — check twice.
|
||||
cp_names = ["kaycee", "bettie", "ofelia"]
|
||||
cp_server_type = "ccx23" # 4 vCPU / 16 GB, dedicated x86
|
||||
cp_location = "fsn1"
|
||||
cp_ip_offset = 11 # CP private IPs → 10.40.11.11 / .12 / .13 (etcd)
|
||||
lb_type = "lb11"
|
||||
lb_ip_offset = 5 # API LB private IP → 10.40.11.5
|
||||
lb_public = false # private-only LB; the API endpoint stays on the kube-cp net
|
||||
# ── Bare-metal control planes (Hetzner Robot; kube-cp VLAN 11, gw .1 = spine) ──
|
||||
# `name` keys the apply resources (stable across list edits). VIP 10.40.11.5
|
||||
# (= the retired hcloud LB IP, so api_dns_name carried over unchanged).
|
||||
# provisioned=false → the one-time install apply dials maint_ip (maintenance
|
||||
# mode); flip true per node as it comes up.
|
||||
cps = [
|
||||
{ name = "harlan", cp_ip = "10.40.11.11", maint_ip = "178.63.124.20", robot_id = 3027819, install_serial = "17451A00D9F8" },
|
||||
{ name = "imelda", cp_ip = "10.40.11.12", maint_ip = "178.63.124.21", robot_id = 3027863, install_serial = "1708162471F6" },
|
||||
{ name = "roscoe", cp_ip = "10.40.11.13", maint_ip = "178.63.124.22", robot_id = 3028524, install_serial = "18201C72C94D" },
|
||||
]
|
||||
# 2×10G Intel 82599ES SFP+ (ixgbe) enslaved into bond0 (tagged kube-cp VLAN 11,
|
||||
# spine port-0 breakout ae4-6). The onboard 1G (e1000e) stays the DHCP
|
||||
# public/egress NIC (default route + NetBird endpoint).
|
||||
cp_bond_driver = "ixgbe"
|
||||
vip_offset = 5 # API VIP → 10.40.11.5
|
||||
|
||||
# ── Bare-metal workers (Hetzner Robot dedicated; sequential after mgmt-1/2) ──
|
||||
# `name` keys the apply resources (stable across list edits — removing or
|
||||
# reordering an entry no longer touches the others). Names verified against
|
||||
# the live nodes 2026-07-07.
|
||||
# ── Bare-metal workers (Hetzner Robot; kube VLAN 10) ─────────────────────────
|
||||
workers = [
|
||||
{ name = "jeanne", fabric_ip = "10.40.10.11", maint_ip = "178.63.124.38", robot_id = 3008210, install_serial = "S64GNNFX503099" },
|
||||
{ name = "sheron", fabric_ip = "10.40.10.12", maint_ip = "178.63.124.37", robot_id = 3008211, install_serial = "S64GNJ0WC25870" },
|
||||
@@ -52,8 +52,8 @@ cluster = {
|
||||
}
|
||||
|
||||
# Operator/CI sources allowed on the Talos host firewall (apid 50000 + apiserver
|
||||
# 6443), on top of the node planes. NetBird peer range ONLY — no public IPs (the
|
||||
# hcloud firewall also blocks public apiserver/apid; see hcloud-firewall.tf). A
|
||||
# re-bootstrap dials apid on a CP public IP, so temporarily re-add the operator's
|
||||
# /32 here (and open 50000 on the hcloud firewall) for that one step.
|
||||
# 6443), on top of the node planes. NetBird peer range ONLY — no public IPs.
|
||||
# Operators reach the CPs over the NetBird kube-cp route (the CPs are the route
|
||||
# peers); a re-bootstrap that must dial apid before the mesh is up goes through
|
||||
# a maint_ip (maintenance mode is unauthenticated — no firewall yet).
|
||||
trusted_cidrs = ["10.254.0.0/15"]
|
||||
|
||||
@@ -1,117 +1,99 @@
|
||||
# Spread placement group — forces the 3 CPs onto DISTINCT physical hosts, so no
|
||||
# single host failure can take out >1 etcd member / break quorum. (hcloud caps a
|
||||
# spread group at 10 servers; 3 is fine.)
|
||||
resource "hcloud_placement_group" "control_plane" {
|
||||
name = "yucca-${var.region_code}-${var.cluster.name}-cp"
|
||||
type = "spread"
|
||||
labels = { cluster = var.cluster.name, role = "control-plane" }
|
||||
# ── Bare-metal control planes ────────────────────────────────────────────────
|
||||
# Applied over apid to nodes already in Talos maintenance mode at their maint_ip
|
||||
# (Hetzner public DHCP on the onboard 1G NIC). After the install+reboot they hold
|
||||
# their kube-cp VLAN address and join the NetBird mesh.
|
||||
#
|
||||
# bond0 (2×10G LACP, ixgbe) → vlan 11 (kube-cp) = cp_ip — etcd + apiserver + VIP
|
||||
# route to the kube VLAN via the kube-cp IRB (10.40.11.1) — apiserver→kubelet
|
||||
# default route via the onboard 1G public NIC (DHCP) — egress + NetBird endpoint
|
||||
#
|
||||
# The API VIP (10.40.11.5) is Talos-managed on the VLAN: etcd elects one holder,
|
||||
# so it's only up while the cluster is healthy — exactly what the api_dns_name
|
||||
# record points at.
|
||||
#
|
||||
# CPs are PROVISIONED to maintenance mode out of band — see the runbook
|
||||
# (./README.md): Hetzner rescue → dd the Talos metal image → reboot. This stack
|
||||
# assumes they're already there.
|
||||
|
||||
locals {
|
||||
# Keyed by hostname (cp_node_map) — same stable key as the apply resource.
|
||||
cp_node_patches = { for hostname, n in local.cp_node_map : hostname => [
|
||||
# Install disk by SERIAL (never by name — enumeration swaps across boots).
|
||||
yamlencode({
|
||||
machine = { install = {
|
||||
diskSelector = { serial = n.install_serial }
|
||||
image = local.install_image
|
||||
} }
|
||||
}),
|
||||
yamlencode({
|
||||
machine = {
|
||||
network = {
|
||||
interfaces = [{
|
||||
interface = "bond0"
|
||||
dhcp = false
|
||||
# Bond members selected by NIC driver (cp_bond_driver, ixgbe) — exactly
|
||||
# the two 10G SFP+ ports; the onboard 1G public NIC is e1000e.
|
||||
bond = {
|
||||
mode = "802.3ad"
|
||||
lacpRate = "fast"
|
||||
xmitHashPolicy = "layer3+4"
|
||||
miimon = 100
|
||||
deviceSelectors = local.c.cp_bond_driver != null ? [{
|
||||
driver = local.c.cp_bond_driver
|
||||
}] : null
|
||||
interfaces = local.c.cp_bond_driver != null ? null : local.c.cp_bond_interfaces
|
||||
}
|
||||
vlans = [{
|
||||
vlanId = module.addr_site.kube_cp_vlan_id # 11
|
||||
addresses = ["${n.cp_ip}/${local.kube_cp_prefix}"]
|
||||
# The workers live one IRB away — pin the return route so
|
||||
# apiserver→kubelet + geneve ride the fabric, not the mesh.
|
||||
routes = [{ network = local.kube_cidr, gateway = local.kube_cp_gw }]
|
||||
vip = { ip = local.api_vip }
|
||||
}]
|
||||
}]
|
||||
}
|
||||
}
|
||||
}),
|
||||
yamlencode({ apiVersion = "v1alpha1", kind = "HostnameConfig", auto = "off", hostname = hostname }),
|
||||
] }
|
||||
}
|
||||
|
||||
# ── Cloud control-plane VMs ──────────────────────────────────────────────────
|
||||
# 3× CCX23 booted from the Talos snapshot, configured via user_data (the per-CP
|
||||
# machine config from talos.tf). Public IPv4 = NetBird NAT traversal + the TF
|
||||
# runner's bootstrap path; private IP (eth1, kube-cp subnet) = etcd. Talos ignores
|
||||
# SSH, so no ssh_keys. ignore_changes keeps re-applies from recycling live nodes.
|
||||
resource "hcloud_server" "control_plane" {
|
||||
count = var.cluster.cp_count
|
||||
name = local.cp_hostnames[count.index]
|
||||
image = data.hcloud_image.talos.id
|
||||
server_type = var.cluster.cp_server_type
|
||||
location = var.cluster.cp_location
|
||||
placement_group_id = hcloud_placement_group.control_plane.id
|
||||
|
||||
user_data = data.talos_machine_configuration.cp[count.index].machine_configuration
|
||||
|
||||
public_net {
|
||||
ipv4_enabled = true
|
||||
ipv6_enabled = false
|
||||
}
|
||||
|
||||
network {
|
||||
network_id = hcloud_network.kube_cp.id
|
||||
ip = local.cp_private_ips[count.index]
|
||||
}
|
||||
|
||||
labels = { cluster = var.cluster.name, role = "control-plane" }
|
||||
|
||||
depends_on = [hcloud_network_subnet.kube_cp]
|
||||
|
||||
lifecycle {
|
||||
ignore_changes = [user_data, image]
|
||||
# etcd members — replacing one rolls quorum; destroying all rolls the
|
||||
# cluster. Any legitimate replace (scale-down, location change) must
|
||||
# temporarily lift this flag, deliberately.
|
||||
prevent_destroy = true
|
||||
}
|
||||
}
|
||||
|
||||
# Live CP config sync — user_data only configures a CP at CREATION (and is
|
||||
# ignore_changes above), so config edits never reached running CPs; today they were
|
||||
# hand-patched via talosctl. This applies the current rendered config to each live CP
|
||||
# on every apply (mode auto: no reboot for the config we manage). Notably it keeps the
|
||||
# worker /etc/hosts entries fresh: a re-provisioned worker gets a new NetBird IP, and
|
||||
# without this the apiserver keeps dialing the dead one.
|
||||
# Keyed by HOSTNAME, not list position: removing or reordering a CP in tfvars
|
||||
# must never shift another node's resource address (a shift = replace = an etcd
|
||||
# member reset). Matches the workers.tf pattern.
|
||||
resource "talos_machine_configuration_apply" "cp" {
|
||||
count = var.cluster.cp_count
|
||||
for_each = local.cp_node_map
|
||||
|
||||
client_configuration = talos_machine_secrets.this.client_configuration
|
||||
machine_configuration_input = data.talos_machine_configuration.cp[count.index].machine_configuration
|
||||
node = local.cp_private_ips[count.index]
|
||||
endpoint = local.cp_private_ips[count.index]
|
||||
machine_configuration_input = data.talos_machine_configuration.cp.machine_configuration
|
||||
# The FIRST apply targets the maintenance-mode node at its Hetzner public IP
|
||||
# (provisioned=false); the config brings up bond0.11 at cp_ip + joins NetBird and
|
||||
# the node reboots into the cluster. Every later apply targets the LIVE node at
|
||||
# its kube-cp IP (over the NetBird kube-cp route). Flip provisioned in tfvars per
|
||||
# CP as it comes up.
|
||||
node = each.value.provisioned ? each.value.cp_ip : each.value.maint_ip
|
||||
endpoint = each.value.provisioned ? each.value.cp_ip : each.value.maint_ip
|
||||
config_patches = local.cp_node_patches[each.key]
|
||||
apply_mode = "auto"
|
||||
|
||||
depends_on = [hcloud_server.control_plane]
|
||||
# reset=false: decommissioning an etcd member must be a deliberate
|
||||
# `talosctl reset` (after `etcd leave`), never a terraform destroy side effect.
|
||||
# NB: on_destroy is read from STATE, so this protects only after it has been
|
||||
# applied once.
|
||||
on_destroy = {
|
||||
reboot = true
|
||||
reset = false
|
||||
graceful = false
|
||||
}
|
||||
|
||||
lifecycle {
|
||||
# cp_netbird_patch is silently OMITTED when the setup key is "" (the
|
||||
# credential-less validate default) — an env-less apply would strip NetBird
|
||||
# from the live CP configs. Fail loudly instead.
|
||||
# from the live CP configs, cutting the operators' kube-cp route. Fail loudly.
|
||||
precondition {
|
||||
condition = length(var.netbird_talos_cp_setup_key) > 0
|
||||
error_message = "netbird_talos_cp_setup_key is empty — run applies through tf/op-run.sh (op run env missing or op:// ref resolved empty)."
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# ── API load balancer ────────────────────────────────────────────────────────
|
||||
# Fronts the 3 CPs on 6443. Private IP (kube-cp) is the in-cluster target; the
|
||||
# public frontend (lb_public) is what operators + workers dial via api_dns_name.
|
||||
# certSANs (talos.tf) already include api_dns_name + the private LB IP.
|
||||
resource "hcloud_load_balancer" "kube_api" {
|
||||
name = "yucca-${var.region_code}-${var.cluster.name}-kube-api"
|
||||
load_balancer_type = var.cluster.lb_type
|
||||
location = var.cluster.cp_location
|
||||
labels = { cluster = var.cluster.name }
|
||||
}
|
||||
|
||||
resource "hcloud_load_balancer_network" "kube_api" {
|
||||
load_balancer_id = hcloud_load_balancer.kube_api.id
|
||||
network_id = hcloud_network.kube_cp.id
|
||||
ip = local.lb_private_ip
|
||||
enable_public_interface = var.cluster.lb_public
|
||||
depends_on = [hcloud_network_subnet.kube_cp]
|
||||
}
|
||||
|
||||
resource "hcloud_load_balancer_service" "kube_api" {
|
||||
load_balancer_id = hcloud_load_balancer.kube_api.id
|
||||
protocol = "tcp"
|
||||
listen_port = 6443
|
||||
destination_port = 6443
|
||||
|
||||
health_check {
|
||||
protocol = "tcp"
|
||||
port = 6443
|
||||
interval = 10
|
||||
timeout = 5
|
||||
retries = 3
|
||||
}
|
||||
}
|
||||
|
||||
# Target the CPs over their PRIVATE IPs (LB is attached to the same network).
|
||||
resource "hcloud_load_balancer_target" "kube_api" {
|
||||
count = var.cluster.cp_count
|
||||
type = "server"
|
||||
load_balancer_id = hcloud_load_balancer.kube_api.id
|
||||
server_id = hcloud_server.control_plane[count.index].id
|
||||
use_private_ip = true
|
||||
depends_on = [hcloud_load_balancer_network.kube_api]
|
||||
}
|
||||
|
||||
@@ -35,13 +35,14 @@ output "discovery" {
|
||||
}
|
||||
kubernetes = {
|
||||
cluster_name = var.cluster.name
|
||||
api_endpoint = local.cluster_endpoint # https://<api_dns_name>:6443 (LB)
|
||||
api_endpoint = local.cluster_endpoint # https://<api_dns_name>:6443 (VIP)
|
||||
api_vip = local.api_vip # Talos-elected VIP on kube-cp (the api_dns_name A record)
|
||||
operator_endpoint = local.operator_endpoint # direct bootstrap-CP apiserver
|
||||
cp_node_ips = local.cp_private_ips # kube-cp IPs; operators/yuctl reach via NetBird
|
||||
cp_node_ips = local.cp_ips # kube-cp IPs; operators/yuctl reach via NetBird
|
||||
worker_node_ips = [for w in var.cluster.workers : w.fabric_ip]
|
||||
# Node NAME → IP maps (short wordlist names). Consumed by the netbird stack
|
||||
# (yucca.futo.network records) instead of hardcoding names in two stacks.
|
||||
cp_nodes = { for i, n in var.cluster.cp_names : n => local.cp_private_ips[i] }
|
||||
cp_nodes = { for n in var.cluster.cps : n.name => n.cp_ip }
|
||||
worker_nodes = { for w in var.cluster.workers : w.name => w.fabric_ip }
|
||||
kubeconfig_ref = "op://${local._disc_vault}/${local._kubeconfig_title}/password"
|
||||
talosconfig_ref = "op://${local._disc_vault}/${local._talosconfig_title}/password"
|
||||
|
||||
@@ -1,15 +1,16 @@
|
||||
# Talos host ingress firewall (default-deny + per-service allow-lists). Governs
|
||||
# HOST-network ports only; pod/ClusterIP traffic rides Cilium.
|
||||
#
|
||||
# Trust planes for this hybrid cluster:
|
||||
# Trust planes:
|
||||
# kube_cidr 10.40.10.0/24 workers' fabric IPs (east-west, BGP)
|
||||
# kube_cp_cidr 10.40.11.0/24 CP private IPs + the API LB (etcd, LB health-checks)
|
||||
# netbird_node_cidr 10.254.0.0/15 the NetBird mesh (apiserver↔kubelet, node control)
|
||||
# kube_cp_cidr 10.40.11.0/24 CP IPs + the API VIP (etcd, apiserver)
|
||||
# netbird_node_cidr 10.254.0.0/15 the NetBird mesh (operators, backup plane)
|
||||
# trusted_cidrs operator/CI source ranges
|
||||
#
|
||||
# ⚠️ The TF runner dials the CP PUBLIC IPs for bootstrap (apid 50000) and the
|
||||
# helm/kubernetes providers (apiserver 6443). Its source IP MUST be in
|
||||
# trusted_cidrs (e.g. the CI runner's egress / NetBird range) or those steps hang.
|
||||
# ⚠️ The TF runner dials the CP kube-cp IPs (over the NetBird kube-cp route) for
|
||||
# bootstrap (apid 50000) and the helm/kubernetes providers (apiserver 6443). Its
|
||||
# source IP MUST be in trusted_cidrs (e.g. the CI runner's NetBird range) or
|
||||
# those steps hang.
|
||||
locals {
|
||||
firewall_allow = concat([local.kube_cidr, local.kube_cp_cidr, local.c.netbird_node_cidr], var.trusted_cidrs)
|
||||
operator_allow = local.firewall_allow
|
||||
@@ -41,7 +42,7 @@ locals {
|
||||
}),
|
||||
# Cilium geneve overlay (tunnel routing): pod↔pod is encapsulated node-to-node
|
||||
# (UDP 6081). Required across BOTH L2 domains — worker↔worker over the fabric and
|
||||
# CP↔worker over the mesh — or pod-to-pod traffic is silently dropped.
|
||||
# CP↔worker routed via the spine IRBs — or pod-to-pod traffic is silently dropped.
|
||||
yamlencode({
|
||||
apiVersion = "v1alpha1"
|
||||
kind = "NetworkRuleConfig"
|
||||
|
||||
@@ -2,9 +2,7 @@
|
||||
# Helm (OCI charts), then Flux reconciles this repo's kubernetes/clusters/prod
|
||||
# path on its own. Lands after the cluster + CNI (providers.tf helm/kubernetes).
|
||||
#
|
||||
# ⚠ TEMPORARY: the sync ref is feat/prod (var.flux_git_ref) so father can be
|
||||
# GitOps-managed before the branch merges. Flip flux_git_ref to "main" (the
|
||||
# default once this merges) — nothing else changes.
|
||||
# Sync ref = main (var.flux_git_ref default; CI guards it stays that way).
|
||||
|
||||
resource "helm_release" "flux_operator" {
|
||||
name = "flux-operator"
|
||||
|
||||
@@ -1,29 +0,0 @@
|
||||
# hcloud firewall for the control-plane VMs — enforces "no public access to the
|
||||
# cluster" at the cloud edge. Only NetBird's WireGuard is allowed inbound from the
|
||||
# internet; apiserver (6443) + apid (50000) + everything else is dropped. The
|
||||
# cluster is reached ONLY over NetBird (WireGuard tunnel, arrives on 51820/udp) or
|
||||
# the private kube-cp network — neither of which this filters (hcloud firewalls
|
||||
# apply to the public interface; private-net + in-tunnel traffic is untouched).
|
||||
#
|
||||
# Egress is unrestricted (hcloud default) — image pulls, NetBird signal/relay, etc.
|
||||
#
|
||||
# NB: a future re-bootstrap dials apid (50000) on a CP public IP — temporarily add
|
||||
# an operator-source rule for 50000, or bootstrap from a NetBird-reachable path.
|
||||
resource "hcloud_firewall" "control_plane" {
|
||||
name = "yucca-${var.region_code}-${var.cluster.name}-cp"
|
||||
labels = { cluster = var.cluster.name, role = "control-plane" }
|
||||
|
||||
rule {
|
||||
direction = "in"
|
||||
protocol = "udp"
|
||||
port = "51820"
|
||||
source_ips = ["0.0.0.0/0", "::/0"]
|
||||
description = "NetBird WireGuard (P2P)"
|
||||
}
|
||||
}
|
||||
|
||||
# Attach to the CPs without recreating them.
|
||||
resource "hcloud_firewall_attachment" "control_plane" {
|
||||
firewall_id = hcloud_firewall.control_plane.id
|
||||
server_ids = hcloud_server.control_plane[*].id
|
||||
}
|
||||
@@ -1,33 +1,12 @@
|
||||
# Talos Image Factory schematic — the extension set, managed in TF. The resource
|
||||
# registers schematic.yaml with the factory and returns its deterministic id; we
|
||||
# derive the CP (hcloud-amd64) + worker (metal-amd64) image URLs from it. No
|
||||
# hand-pasted schematic id, no out-of-band curl.
|
||||
# registers schematic.yaml with the factory and returns its deterministic id; the
|
||||
# metal installer/image URLs derive from it. No hand-pasted schematic id, no
|
||||
# out-of-band curl. One schematic for every node (all bare-metal).
|
||||
resource "talos_image_factory_schematic" "this" {
|
||||
schematic = file("${path.module}/schematic.yaml")
|
||||
}
|
||||
|
||||
# Worker (bare-metal) schematic — same set minus qemu-guest-agent (see the file).
|
||||
resource "talos_image_factory_schematic" "worker" {
|
||||
schematic = file("${path.module}/schematic-worker.yaml")
|
||||
}
|
||||
|
||||
locals {
|
||||
talos_schematic_id = talos_image_factory_schematic.this.id
|
||||
talos_worker_schematic_id = talos_image_factory_schematic.worker.id
|
||||
talos_hcloud_image_url = "https://factory.talos.dev/image/${local.talos_schematic_id}/v${var.cluster.talos_version}/hcloud-amd64.raw.xz"
|
||||
talos_metal_image_url = "https://factory.talos.dev/image/${local.talos_worker_schematic_id}/v${var.cluster.talos_version}/metal-amd64.raw.xz"
|
||||
}
|
||||
|
||||
# Talos amd64 image as an hcloud snapshot. hcloud can't boot the Talos ISO, so the
|
||||
# snapshot is built ONCE, out of band, by:
|
||||
#
|
||||
# mise run hetzner:talos-image
|
||||
#
|
||||
# (it reads talos_hcloud_image_url from this stack's output, spins a temporary
|
||||
# rescue server, dd's the image, snapshots, tears down). This data source then
|
||||
# resolves it by label — rebuild only on a Talos version bump.
|
||||
data "hcloud_image" "talos" {
|
||||
with_selector = "os=talos,cluster=${var.cluster.name},arch=amd64"
|
||||
with_architecture = "x86"
|
||||
most_recent = true
|
||||
talos_schematic_id = talos_image_factory_schematic.this.id
|
||||
talos_metal_image_url = "https://factory.talos.dev/image/${local.talos_schematic_id}/v${var.cluster.talos_version}/metal-amd64.raw.xz"
|
||||
}
|
||||
|
||||
@@ -1,33 +0,0 @@
|
||||
# ── One-shot state migration: count → hostname-keyed workers ─────────────────
|
||||
# Moves the existing worker apply instances to their stable hostname keys and
|
||||
# forgets the retired node-names shuffle module WITHOUT destroying anything.
|
||||
# The gate before merging: a local `mise tf:plan` must show exactly these three
|
||||
# moves + one "removed from state" + the on_destroy in-place updates — zero
|
||||
# destroys, zero replaces, zero diffs on the CP resources.
|
||||
# DELETE this file once CI has applied it (the blocks are inert afterwards, but
|
||||
# the `removed` block conflicts if module "names" is ever reintroduced).
|
||||
|
||||
moved {
|
||||
from = talos_machine_configuration_apply.worker[0]
|
||||
to = talos_machine_configuration_apply.worker["yucca-htz-fsn-father-k8s-jeanne"]
|
||||
}
|
||||
|
||||
moved {
|
||||
from = talos_machine_configuration_apply.worker[1]
|
||||
to = talos_machine_configuration_apply.worker["yucca-htz-fsn-father-k8s-sheron"]
|
||||
}
|
||||
|
||||
moved {
|
||||
from = talos_machine_configuration_apply.worker[2]
|
||||
to = talos_machine_configuration_apply.worker["yucca-htz-fsn-father-k8s-dianna"]
|
||||
}
|
||||
|
||||
# Forget (don't destroy) the retired names module — its only resource is a
|
||||
# random_shuffle; destroying would be inert, but "forget" keeps the plan clean.
|
||||
removed {
|
||||
from = module.names
|
||||
|
||||
lifecycle {
|
||||
destroy = false
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,91 @@
|
||||
# ─── netops namespace + fabric-credential Secrets ────────────────────────────
|
||||
# The netops stack (kubernetes/apps/prod/htz-fsn1/netops/) mounts fabric
|
||||
# credentials that must NEVER be in git: the read-only `netops` Junos login's
|
||||
# SSH key + password (fabric stack, fabric.tf netops_users) and the Grafana
|
||||
# admin password. Historically these were hand-created (`kubectl create secret`)
|
||||
# and DIED WITH THE CLUSTER on the 2026-07 rebuild — now they're provisioned
|
||||
# here from the same 1Password items, so a rebuild restores them with the stack.
|
||||
# Flux owns the workloads around them; TF owns the namespace + these Secrets
|
||||
# (the namespace also carries the VictoriaMetrics hostPath PVC, so it must
|
||||
# survive flux prunes — TF ownership replaces the old prune-disabled manifest).
|
||||
|
||||
data "onepassword_item" "netops_ssh_key" {
|
||||
vault = data.onepassword_vault.prod.uuid
|
||||
title = "NETOPS_FABRIC_SSH_PRIVATE_KEY" # DOCUMENT item; file id_ed25519
|
||||
}
|
||||
|
||||
data "onepassword_item" "netops_fabric_password" {
|
||||
vault = data.onepassword_vault.prod.uuid
|
||||
title = "NETOPS_FABRIC_PASSWORD"
|
||||
}
|
||||
|
||||
data "onepassword_item" "grafana_admin" {
|
||||
vault = data.onepassword_vault.prod.uuid
|
||||
title = "FATHER_GRAFANA_ADMIN"
|
||||
}
|
||||
|
||||
resource "kubernetes_namespace_v1" "netops" {
|
||||
metadata {
|
||||
name = "netops"
|
||||
labels = {
|
||||
# VictoriaMetrics persists to a hostPath (/var/mnt); baseline forbids it.
|
||||
"pod-security.kubernetes.io/enforce" = "privileged"
|
||||
}
|
||||
annotations = {
|
||||
# Belt-and-braces from the flux-owned era (the namespace manifest is gone
|
||||
# from the tree, but flux's GC honors this if it ever re-tracks the object).
|
||||
"kustomize.toolkit.fluxcd.io/prune" = "disabled"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# SSH key for junos-exporter (NETCONF scrape) + oxidized (config backup) — both
|
||||
# mount key `id_ed25519` and log in as the `netops` Junos user.
|
||||
resource "kubernetes_secret_v1" "netops_ssh" {
|
||||
metadata {
|
||||
name = "netops-ssh"
|
||||
namespace = kubernetes_namespace_v1.netops.metadata[0].name
|
||||
}
|
||||
data = {
|
||||
id_ed25519 = one([for f in data.onepassword_item.netops_ssh_key.file : f.content if f.name == "id_ed25519"])
|
||||
}
|
||||
}
|
||||
|
||||
resource "kubernetes_secret_v1" "grafana_admin" {
|
||||
metadata {
|
||||
name = "grafana-admin"
|
||||
namespace = kubernetes_namespace_v1.netops.metadata[0].name
|
||||
}
|
||||
data = {
|
||||
password = data.onepassword_item.grafana_admin.password
|
||||
}
|
||||
}
|
||||
|
||||
# hyperglass device inventory — embeds the netops PASSWORD (netmiko can't
|
||||
# key-auth through hyperglass config), hence a Secret and not the configmap.
|
||||
resource "kubernetes_secret_v1" "hyperglass_devices" {
|
||||
metadata {
|
||||
name = "hyperglass-devices"
|
||||
namespace = kubernetes_namespace_v1.netops.metadata[0].name
|
||||
}
|
||||
# Spine only: hyperglass's juniper directives require source4 AND source6 per
|
||||
# device, and only the spine has both (lo0 + the transit v6) — the leaf has no
|
||||
# public/v6 presence, so LG queries from it would be meaningless anyway.
|
||||
data = {
|
||||
"devices.yaml" = yamlencode({
|
||||
devices = [
|
||||
{
|
||||
name = "corenetsw"
|
||||
description = "spine VC (QFX5200-32C x2)"
|
||||
address = module.addr_site.spine_mgmt_ip
|
||||
platform = "juniper"
|
||||
attrs = {
|
||||
source4 = "69.48.224.254" # lo0 (fabric stack, transits.loopback)
|
||||
source6 = "2a01:4a0:1338:226::2" # transit /64 local (fabric stack, transits.local_v6)
|
||||
}
|
||||
credential = { username = "netops", password = data.onepassword_item.netops_fabric_password.password }
|
||||
},
|
||||
]
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -1,23 +0,0 @@
|
||||
# Isolated Hetzner Cloud network for the control plane. Holds ONLY the 3 CP VMs
|
||||
# (etcd CP↔CP on private IPs) + the API LB's private IP. The workers do NOT join
|
||||
# it — CP↔worker rides the NetBird mesh. Range = kube-cp (10.40.11.0/24), carved
|
||||
# from the site supernet for collision-free IPAM (see fabric-addressing).
|
||||
resource "hcloud_network" "kube_cp" {
|
||||
name = "yucca-${var.region_code}-${var.cluster.name}-kube-cp"
|
||||
ip_range = local.kube_cp_cidr
|
||||
labels = { cluster = var.cluster.name, plane = "control" }
|
||||
}
|
||||
|
||||
# Single cloud subnet for the CP VMs + LB (no vSwitch subnet — workers aren't here).
|
||||
resource "hcloud_network_subnet" "kube_cp" {
|
||||
network_id = hcloud_network.kube_cp.id
|
||||
type = "cloud"
|
||||
network_zone = "eu-central" # fsn1/nbg1/hel1
|
||||
ip_range = local.kube_cp_cidr
|
||||
}
|
||||
|
||||
locals {
|
||||
# Deterministic private IPs in the kube-cp subnet.
|
||||
cp_private_ips = [for i in range(var.cluster.cp_count) : cidrhost(local.kube_cp_cidr, var.cluster.cp_ip_offset + i)]
|
||||
lb_private_ip = cidrhost(local.kube_cp_cidr, var.cluster.lb_ip_offset)
|
||||
}
|
||||
@@ -4,36 +4,29 @@ output "cluster_summary" {
|
||||
cluster_name = var.cluster.name
|
||||
api_endpoint = local.cluster_endpoint
|
||||
api_dns_name = local.api_dns_name
|
||||
api_vip = local.api_vip
|
||||
operator_endpoint = local.operator_endpoint
|
||||
lb_private_ip = local.lb_private_ip
|
||||
cp_public_ips = local.cp_public_ips
|
||||
cp_private_ips = local.cp_private_ips
|
||||
cp_ips = local.cp_ips
|
||||
worker_fabric_ips = [for w in var.cluster.workers : w.fabric_ip]
|
||||
}
|
||||
}
|
||||
|
||||
# DNS hint: api_dns_name resolves to the PRIVATE LB IP (lb_public = false — the
|
||||
# API is reachable only over the NetBird mesh; the netbird stack's
|
||||
# DNS hint: api_dns_name resolves to the Talos-elected VIP on the kube-cp VLAN
|
||||
# (the API is reachable only over the NetBird kube-cp route; the netbird stack's
|
||||
# yucca.futo.network zone serves the record). No public A record exists.
|
||||
output "api_dns_record" {
|
||||
description = "The internal record: <api_dns_name> → <lb private IP> (NetBird DNS)."
|
||||
value = "${local.api_dns_name} A ${local.lb_private_ip}"
|
||||
description = "The internal record: <api_dns_name> → <API VIP> (NetBird DNS)."
|
||||
value = "${local.api_dns_name} A ${local.api_vip}"
|
||||
}
|
||||
|
||||
# Image Factory outputs — `mise run hetzner:talos-image` reads the hcloud URL to
|
||||
# build the snapshot; the metal URL feeds the worker rescue-install runbook.
|
||||
# Image Factory outputs — the metal URL feeds the rescue-install runbook (README).
|
||||
output "talos_schematic_id" {
|
||||
description = "Image Factory schematic id (from schematic.yaml)."
|
||||
value = local.talos_schematic_id
|
||||
}
|
||||
|
||||
output "talos_hcloud_image_url" {
|
||||
description = "Factory hcloud-amd64 raw.xz URL — input to hcloud-upload-image (CP snapshot)."
|
||||
value = local.talos_hcloud_image_url
|
||||
}
|
||||
|
||||
output "talos_metal_image_url" {
|
||||
description = "Factory metal-amd64 raw.xz URL — workers dd this in rescue."
|
||||
description = "Factory metal-amd64 raw.xz URL — nodes dd this in rescue."
|
||||
value = local.talos_metal_image_url
|
||||
}
|
||||
|
||||
|
||||
@@ -1,11 +1,8 @@
|
||||
# hcloud — the control-plane VMs, their private subnet, the API LB, and the Talos
|
||||
# snapshot lookup. Token comes from HCLOUD_TOKEN (op run --env-file=tf/.env.prod).
|
||||
provider "hcloud" {}
|
||||
|
||||
# helm + kubernetes bind to ONE cluster: the bootstrap CP's directly-reachable
|
||||
# (public) apiserver — up immediately after bootstrap and in the cert SANs, unlike
|
||||
# the LB which only goes healthy once an apiserver answers. Creds come from the
|
||||
# Talos-minted admin kubeconfig (known after talos_cluster_kubeconfig applies).
|
||||
# apiserver (over the NetBird kube-cp route) — up immediately after bootstrap and
|
||||
# in the cert SANs, unlike the VIP which only settles once etcd elects a holder.
|
||||
# Creds come from the Talos-minted admin kubeconfig (known after
|
||||
# talos_cluster_kubeconfig applies).
|
||||
provider "helm" {
|
||||
kubernetes = {
|
||||
host = local.operator_endpoint
|
||||
@@ -25,8 +22,3 @@ provider "kubernetes" {
|
||||
# 1Password — persists the kube/talosconfig into yucca_tf_prod (secrets.tf). Auth
|
||||
# via OP_SERVICE_ACCOUNT_TOKEN (op run).
|
||||
provider "onepassword" {}
|
||||
|
||||
# netbird — read-only worker peer lookups: their mesh IPs feed the CPs'
|
||||
# extraHostEntries so the apiserver dials worker kubelets peer-to-peer over the
|
||||
# mesh (no mgmt route in the path). PAT via NB_PAT (op run).
|
||||
provider "netbird" {}
|
||||
|
||||
@@ -1,11 +0,0 @@
|
||||
# Talos Image Factory schematic for the father WORKERS (bare-metal Hetzner Robot).
|
||||
# Split from schematic.yaml (the CP/VM set): qemu-guest-agent must NOT be here — on
|
||||
# metal there is no virtio port, the extension service waits forever for
|
||||
# /dev/virtio-ports/org.qemu.guest_agent.0, and the boot sequence never reaches
|
||||
# `running`, failing every talos health check (and with it every TF plan/apply).
|
||||
customization:
|
||||
systemExtensions:
|
||||
officialExtensions:
|
||||
- siderolabs/netbird # node-level overlay (worker↔apiserver)
|
||||
- siderolabs/intel-ucode # worker Xeon microcode
|
||||
- siderolabs/util-linux-tools # fstrim et al.
|
||||
@@ -1,12 +1,13 @@
|
||||
# Talos Image Factory schematic for the `father` cluster — the single source of
|
||||
# truth for the node extension set. Registered with the factory by TF
|
||||
# (talos_image_factory_schematic, image.tf); its id derives the CP (hcloud) and
|
||||
# worker (metal) image URLs. One schematic covers both platforms — the extras are
|
||||
# harmless no-ops on the other (qemu-guest-agent on metal, intel-ucode on a VM).
|
||||
# truth for the node extension set (CPs + workers, all bare-metal Hetzner Robot).
|
||||
# Registered with the factory by TF (talos_image_factory_schematic, image.tf); its
|
||||
# id derives the metal installer/image URLs. NB: qemu-guest-agent must NOT be here —
|
||||
# on metal there is no virtio port, the extension service waits forever for
|
||||
# /dev/virtio-ports/org.qemu.guest_agent.0, and the boot sequence never reaches
|
||||
# `running`, failing every talos health check (and with it every TF plan/apply).
|
||||
customization:
|
||||
systemExtensions:
|
||||
officialExtensions:
|
||||
- siderolabs/netbird # node-level overlay (CP↔fabric route)
|
||||
- siderolabs/qemu-guest-agent # hcloud control-plane VMs
|
||||
- siderolabs/intel-ucode # worker Xeon microcode
|
||||
- siderolabs/netbird # node-level overlay (operator plane / kube-cp route)
|
||||
- siderolabs/intel-ucode # Xeon microcode
|
||||
- siderolabs/util-linux-tools # fstrim et al.
|
||||
|
||||
@@ -1,22 +1,18 @@
|
||||
# ── Talos bring-up (hybrid) ──────────────────────────────────────────────────
|
||||
# Cloud CPs are configured via hcloud user_data (controlplane.tf); bare-metal
|
||||
# workers via apid apply (workers.tf). Both join the SAME cluster (one set of
|
||||
# machine_secrets) and the SAME NetBird mesh (node IPs are NetBird addresses).
|
||||
# ── Talos bring-up (all bare-metal) ──────────────────────────────────────────
|
||||
# CPs and workers are both driven over apid (controlplane.tf / workers.tf) into
|
||||
# ONE cluster (one set of machine_secrets). All node planes ride the fabric:
|
||||
#
|
||||
# node plane (CP↔worker, etcd-client, apiserver↔kubelet) → NetBird (100.64/10)
|
||||
# etcd (CP↔CP) → kube-cp hcloud subnet
|
||||
# worker↔worker pod east-west → kube fabric (50G), Cilium BGP
|
||||
# API endpoint → public Hetzner Cloud LB
|
||||
# etcd (CP↔CP) + apiserver + API VIP → kube-cp fabric VLAN 11 (10.40.11.0/24)
|
||||
# worker↔worker pod east-west → kube fabric VLAN 10 (50G), Cilium BGP
|
||||
# CP↔worker (apiserver↔kubelet, geneve) → routed kube↔kube-cp via the spine IRBs
|
||||
# operators/CI → NetBird mesh (kube-cp routed via the CPs)
|
||||
#
|
||||
# Bootstrap/kubeconfig/health dial the CP PUBLIC IPs (firewalled) — the only thing
|
||||
# the TF runner can reach before NetBird/the LB settle.
|
||||
# NetBird stays on every node as the operator/backup plane — node-to-node traffic
|
||||
# no longer depends on it (static fabric routes are pinned in the machine configs).
|
||||
|
||||
# Node names are EXPLICIT in tfvars (cluster.cp_names + workers[*].name) — the
|
||||
# node-names shuffle module was retired here: auto-picked names re-roll when the
|
||||
# pool input changes (adding an explicit name shrinks the shuffle pool → every
|
||||
# auto name changes → every node renames), and positional slotting meant a
|
||||
# cp_count change renamed all workers. migrations.tf forgets the old module
|
||||
# state without destroying anything.
|
||||
# Node names are EXPLICIT in tfvars (cluster.cps[*].name + workers[*].name) —
|
||||
# auto-picked names re-roll when the pool input changes, silently renaming (=
|
||||
# replacing) live nodes.
|
||||
|
||||
locals {
|
||||
c = var.cluster
|
||||
@@ -26,49 +22,40 @@ locals {
|
||||
pod_cidr = "10.250.0.0/17" # 10.250.0.0 – 10.250.127.255
|
||||
service_cidr = "10.250.128.0/17" # 10.250.128.0 – 10.250.255.255
|
||||
|
||||
# Factory installers (keep each schematic's extensions). Workers consult theirs on
|
||||
# install/upgrade; CPs boot the hcloud snapshot and only consult this on a reinstall.
|
||||
# SPLIT per role: the worker schematic drops qemu-guest-agent (blocks metal boot).
|
||||
cp_install_image = "factory.talos.dev/metal-installer/${local.talos_schematic_id}:v${local.c.talos_version}"
|
||||
worker_install_image = "factory.talos.dev/metal-installer/${local.talos_worker_schematic_id}:v${local.c.talos_version}"
|
||||
# Factory installer (keeps the schematic's extensions) — one schematic for every
|
||||
# node (all metal now); consulted on install/upgrade.
|
||||
install_image = "factory.talos.dev/metal-installer/${local.talos_schematic_id}:v${local.c.talos_version}"
|
||||
|
||||
# Private API endpoint: a NetBird DNS-zone name resolving to the PRIVATE LB IP
|
||||
# (10.40.11.5). Name = kube.<cluster>.<region>.<provider>.yucca.futo.network. It's
|
||||
# in the cert SANs + on each CP as a host-entry; NetBird peers resolve it via the
|
||||
# yucca.futo.network zone (netbird stack) and reach the LB over the kube-cp route
|
||||
# (CPs are the route peers). Node-side traffic never resolves it: kubelets dial
|
||||
# KubePrism (127.0.0.1:7445), which load-balances to the CP IPs directly.
|
||||
# legacy_api_dns_name (the old yucca.internal name) stays in the SANs + host
|
||||
# entries so pre-migration kubeconfigs keep verifying — drop it once rotated.
|
||||
api_dns_name = "kube.${local.c.name}.${var.region_code}.${var.provider_code}.yucca.futo.network"
|
||||
legacy_api_dns_name = "kube.${local.c.name}.${var.region_code}.${var.provider_code}.yucca.internal"
|
||||
cluster_endpoint = "https://${local.api_dns_name}:6443"
|
||||
# API endpoint: a NetBird DNS-zone name resolving to the Talos-elected VIP
|
||||
# (10.40.11.5, kube-cp VLAN — same IP the retired hcloud LB held, so the record
|
||||
# carried over). Name = kube.<cluster>.<region>.<provider>.yucca.futo.network.
|
||||
# It's in the cert SANs + on each node as a host-entry; NetBird peers resolve it
|
||||
# via the yucca.futo.network zone (netbird stack) and reach the VIP over the
|
||||
# kube-cp route (CPs are the route peers). Node-side traffic never resolves it:
|
||||
# kubelets dial KubePrism (127.0.0.1:7445), which load-balances to the CP IPs.
|
||||
api_dns_name = "kube.${local.c.name}.${var.region_code}.${var.provider_code}.yucca.futo.network"
|
||||
cluster_endpoint = "https://${local.api_dns_name}:6443"
|
||||
|
||||
kube_cp_prefix = split("/", local.kube_cp_cidr)[1] # 24
|
||||
api_vip = cidrhost(local.kube_cp_cidr, local.c.vip_offset) # 10.40.11.5
|
||||
kube_cp_prefix = split("/", local.kube_cp_cidr)[1] # 24
|
||||
|
||||
# Hostnames: yucca-htz-fsn-father-k8s-<name>. Workers additionally get a
|
||||
# hostname-keyed map — the STABLE key for the apply resources (workers.tf) and
|
||||
# the netbird peer lookups, so list edits can't shift another node's identity.
|
||||
node_prefix = "yucca-${var.provider_code}-${var.region_code}-${local.c.name}-k8s"
|
||||
cp_hostnames = [for n in local.c.cp_names : "${local.node_prefix}-${n}"]
|
||||
workers_named = [for w in local.c.workers : merge(w, { hostname = "${local.node_prefix}-${w.name}" })]
|
||||
worker_hostnames = [for w in local.workers_named : w.hostname]
|
||||
worker_node_map = { for w in local.workers_named : w.hostname => w }
|
||||
# Hostnames: yucca-htz-fsn-father-k8s-<name>. Both roles get hostname-keyed
|
||||
# maps — the STABLE key for the apply resources, so list edits can't shift
|
||||
# another node's identity.
|
||||
node_prefix = "yucca-${var.provider_code}-${var.region_code}-${local.c.name}-k8s"
|
||||
cps_named = [for n in local.c.cps : merge(n, { hostname = "${local.node_prefix}-${n.name}" })]
|
||||
cp_node_map = { for n in local.cps_named : n.hostname => n }
|
||||
cp_ips = local.cps_named[*].cp_ip
|
||||
workers_named = [for w in local.c.workers : merge(w, { hostname = "${local.node_prefix}-${w.name}" })]
|
||||
worker_node_map = { for w in local.workers_named : w.hostname => w }
|
||||
|
||||
# apiserver cert SANs — the names/IPs clients dial. NOT the CP public IPs (those
|
||||
# don't exist until the servers are created from this very config).
|
||||
# apiserver cert SANs — the names/IPs clients dial.
|
||||
apiserver_cert_sans = concat(
|
||||
[local.api_dns_name, local.legacy_api_dns_name, local.lb_private_ip],
|
||||
local.cp_private_ips,
|
||||
[local.api_dns_name, local.api_vip],
|
||||
local.cp_ips,
|
||||
["127.0.0.1", "localhost"],
|
||||
)
|
||||
|
||||
# ── Shared patches (every node) ──────────────────────────────────────────
|
||||
cp_install_patch = yamlencode({
|
||||
machine = { install = { disk = local.c.install_disk, image = local.cp_install_image } }
|
||||
})
|
||||
|
||||
|
||||
# Talos's default forwards coredns's upstream queries to the host DNS on a link-local
|
||||
# address (169.254.116.108) — unreachable from pods under Cilium's eBPF datapath
|
||||
# (bpf.masquerade), so every EXTERNAL lookup from a pod times out while cluster.local
|
||||
@@ -77,20 +64,13 @@ locals {
|
||||
machine = { features = { hostDNS = { forwardKubeDNSToHost = false } } }
|
||||
})
|
||||
|
||||
# NetBird node-level overlay. CPs and workers join with DIFFERENT setup keys so they
|
||||
# land in different groups: CPs → [talos, talos_cp], workers → [talos]. talos_cp is
|
||||
# the CP-only router group for the kube-cp network — the workers must NOT be in it, or
|
||||
# NetBird treats them as kube-cp routers and they never install the client route (their
|
||||
# pods can't reach the apiserver). See the netbird stack for the group/router wiring.
|
||||
# Both CP + worker run netbird in its normal (modern) mode. The pod→routed-subnet
|
||||
# problem — Cilium's eBPF host-routing does its FIB lookup against the MAIN table
|
||||
# only, so any route netbird parks in a policy table (it has been observed using
|
||||
# both main and table 7120 across versions/restarts) is invisible to POD egress,
|
||||
# and pod→apiserver via kube-cp gets "no route to host" — is fixed DETERMINISTICALLY
|
||||
# on the workers by worker_netbird_route_patch (a Talos-managed main-table route),
|
||||
# NOT by pinning netbird to its deprecated NB_USE_LEGACY_ROUTING mode. CPs don't
|
||||
# run the eBPF pod-datapath to routed subnets (their control-plane pods are
|
||||
# hostNetwork → host stack, which honors policy routing), so they need no route.
|
||||
# NetBird node-level overlay — the OPERATOR plane (kube-cp routed to the mesh via
|
||||
# the CPs) and a backup path; node-to-node traffic rides the fabric via the static
|
||||
# routes pinned below. CPs and workers join with DIFFERENT setup keys so they land
|
||||
# in different groups: CPs → [talos, talos_cp], workers → [talos]. talos_cp is the
|
||||
# CP-only router group for the kube-cp network — the workers must NOT be in it, or
|
||||
# NetBird treats them as kube-cp routers and they never install the client route.
|
||||
# See the netbird stack for the group/router wiring.
|
||||
netbird_env = ["NB_MANAGEMENT_URL=https://api.netbird.io"]
|
||||
cp_netbird_patch = var.netbird_talos_cp_setup_key != "" ? yamlencode({
|
||||
apiVersion = "v1alpha1"
|
||||
@@ -105,30 +85,10 @@ locals {
|
||||
environment = concat(["NB_SETUP_KEY=${var.netbird_talos_setup_key}"], local.netbird_env)
|
||||
}) : ""
|
||||
|
||||
# DETERMINISTIC pod→apiserver fix: a Talos-managed route for the kube-cp subnet
|
||||
# (apiserver + private API LB, reachable only over the mesh) into the MAIN table
|
||||
# via wt0 — exactly where Cilium's eBPF FIB lookup reads. netbird still installs
|
||||
# its own route (its table is version-dependent); ours guarantees main is
|
||||
# populated regardless, so pod egress to kube-cp always resolves. Declaring a
|
||||
# route on wt0 does NOT disturb the netbird extension (it keeps owning wt0's
|
||||
# address; Talos only adds the route). Workers only — CPs are ON kube-cp.
|
||||
worker_netbird_route_patch = yamlencode({
|
||||
machine = {
|
||||
network = {
|
||||
interfaces = [{
|
||||
interface = "wt0"
|
||||
routes = [{ network = local.kube_cp_cidr }]
|
||||
}]
|
||||
}
|
||||
}
|
||||
})
|
||||
|
||||
# nodeIP selection:
|
||||
# CPs → kube-cp hcloud private subnet (apiserver↔CP-kubelet stays private;
|
||||
# decoupled from NetBird readiness at boot)
|
||||
# workers → kube fabric IP. All workers share VLAN-10 L2, so Cilium
|
||||
# autoDirectNodeRoutes routes pod east-west directly over the 50G
|
||||
# fabric — no BGP, no overlay.
|
||||
# CPs → kube-cp fabric VLAN 11 (etcd + apiserver↔CP-kubelet)
|
||||
# workers → kube fabric VLAN 10 (all workers share the L2, so Cilium routes pod
|
||||
# east-west directly over the 50G fabric)
|
||||
# clusterDNS must be set explicitly: Talos defaults it to 10.96.0.10 (the upstream
|
||||
# default service CIDR's DNS) and does NOT derive it from our serviceSubnets — the
|
||||
# kube-dns Service actually lands at cidrhost(service_cidr, 10). Without this every
|
||||
@@ -196,13 +156,10 @@ locals {
|
||||
}
|
||||
})
|
||||
|
||||
cp_base_patches = compact([local.cp_install_patch, local.hostdns_patch, local.cp_netbird_patch, local.cp_nodeip_patch])
|
||||
# Workers ARE NetBird peers: the apiserver lives on the kube-cp hcloud net, only
|
||||
# reachable over the mesh, and workers resolve the API endpoint via the yucca.internal
|
||||
# NetBird DNS zone. nodeIP stays on the fabric (worker_nodeip_patch) so pod east-west
|
||||
# rides VLAN 10; only the API control path uses NetBird.
|
||||
# (install patch is PER-WORKER — by disk serial — appended in workers.tf)
|
||||
worker_base_patches = compact([local.hostdns_patch, local.worker_mayastor_patch, local.worker_volumes_patch, local.worker_netbird_patch, local.worker_netbird_route_patch, local.worker_nodeip_patch])
|
||||
cp_base_patches = compact([local.hostdns_patch, local.cp_netbird_patch, local.cp_nodeip_patch])
|
||||
# (install patch is PER-NODE — by disk serial — appended in controlplane.tf /
|
||||
# workers.tf, along with the bond/VLAN network patches.)
|
||||
worker_base_patches = compact([local.hostdns_patch, local.worker_mayastor_patch, local.worker_volumes_patch, local.worker_netbird_patch, local.worker_nodeip_patch])
|
||||
|
||||
# ── Control-plane cluster config (same on every CP) ──────────────────────
|
||||
cp_cluster_patch = yamlencode({
|
||||
@@ -222,21 +179,13 @@ locals {
|
||||
coreDNS = { disabled = true }
|
||||
apiServer = {
|
||||
certSANs = local.apiserver_cert_sans
|
||||
# Dial kubelets by HOSTNAME first (Talos's default is InternalIP-first). The
|
||||
# worker InternalIPs are fabric addresses only reachable via the mgmt NetBird
|
||||
# routers — a single flappy bridge that intermittently broke logs/exec. Worker
|
||||
# hostnames resolve (via the CPs' extraHostEntries below) to the workers' OWN
|
||||
# NetBird IPs, so apiserver→kubelet is peer-to-peer over the mesh — the same
|
||||
# always-on tunnels the kubelets already use to reach the apiserver. The CP
|
||||
# hostnames resolve via hcloud DNS to their kube-cp IPs, unchanged.
|
||||
extraArgs = { "kubelet-preferred-address-types" = "Hostname,InternalIP,ExternalIP" }
|
||||
# hostNetwork pods get /etc/hosts COPIED at sandbox creation — a host-level
|
||||
# extraHostEntries refresh never reaches the RUNNING apiserver. Stamping the
|
||||
# entry-set hash into the pod spec forces kubelet to recreate the pod (fresh
|
||||
# /etc/hosts) whenever a worker's mesh IP changes (e.g. re-provision).
|
||||
env = { MESH_HOSTS_REVISION = substr(sha256(jsonencode(local.cp_host_entries)), 0, 12) }
|
||||
# /etc/hosts) whenever the entry set changes (e.g. a node add).
|
||||
env = { HOSTS_REVISION = substr(sha256(jsonencode(local.cp_host_entries)), 0, 12) }
|
||||
}
|
||||
# Pin etcd to the kube-cp hcloud subnet so CP↔CP etcd stays off the mesh.
|
||||
# Pin etcd to the kube-cp VLAN so CP↔CP etcd stays off the mesh + public NICs.
|
||||
etcd = { advertisedSubnets = [local.kube_cp_cidr] }
|
||||
}
|
||||
})
|
||||
@@ -244,25 +193,20 @@ locals {
|
||||
# CP node extras:
|
||||
# • ip_forward — the CPs are the NetBird route peers for the kube-cp subnet
|
||||
# (yucca-fsn-father-kube-cp), so they must forward overlay↔subnet traffic.
|
||||
# • extraHostEntries — on the CPs, resolve api_dns_name to the 3 CP private IPs
|
||||
# (round-robin, all in the cert SANs). NOT the LB VIP (CPs are LB targets →
|
||||
# hcloud hairpin), and NOT 127.0.0.1 (a joining CP must reach a WORKING
|
||||
# apiserver — a peer's — to register its etcd membership; its own apiserver
|
||||
# isn't up until etcd joins). Off-node peers resolve api_dns_name via the
|
||||
# NetBird yucca.internal zone.
|
||||
# • kubelet dialing (worker_mesh_kubelet): the apiserver prefers the Hostname node
|
||||
# address (cp_cluster_patch), so every node hostname must resolve on the CPs:
|
||||
# CP hostnames → their kube-cp IPs (stable), worker hostnames → their NetBird
|
||||
# IPs (data.netbird_peer — the peer-to-peer mesh path, no mgmt route). Talos
|
||||
# host-dns can't resolve NetBird DNS zones, hence /etc/hosts, which the
|
||||
# hostNetwork apiserver inherits.
|
||||
# • extraHostEntries — resolve api_dns_name to the 3 CP IPs (round-robin, all in
|
||||
# the cert SANs). NOT the VIP (a joining CP must reach a WORKING apiserver — a
|
||||
# peer's — to register its etcd membership; the VIP may be parked on itself),
|
||||
# and NOT 127.0.0.1. Every node hostname also resolves to its fabric IP so
|
||||
# apiserver→kubelet dials ride the fabric (Talos host-dns can't resolve NetBird
|
||||
# DNS zones, hence /etc/hosts, which the hostNetwork apiserver inherits).
|
||||
# Off-node peers resolve api_dns_name via the NetBird yucca.futo.network zone.
|
||||
cp_host_entries = concat(
|
||||
[for ip in local.cp_private_ips : { ip = ip, aliases = [local.api_dns_name, local.legacy_api_dns_name] }],
|
||||
[for i, ip in local.cp_private_ips : { ip = ip, aliases = [local.cp_hostnames[i]] }],
|
||||
# Iterate the tfvars LIST (not the hostname-keyed data map, whose lexical
|
||||
# order differs) — entry order is part of the rendered CP config, and
|
||||
# reordering it would churn every CP's machine config for nothing.
|
||||
[for w in local.workers_named : { ip = data.netbird_peer.worker[w.hostname].ip, aliases = [w.hostname] } if local.c.worker_mesh_kubelet],
|
||||
[for ip in local.cp_ips : { ip = ip, aliases = [local.api_dns_name] }],
|
||||
# Iterate the tfvars LISTS (not the hostname-keyed maps, whose lexical order
|
||||
# differs) — entry order is part of the rendered CP config, and reordering it
|
||||
# would churn every CP's machine config for nothing.
|
||||
[for n in local.cps_named : { ip = n.cp_ip, aliases = [n.hostname] }],
|
||||
[for w in local.workers_named : { ip = w.fabric_ip, aliases = [w.hostname] }],
|
||||
)
|
||||
|
||||
cp_extras_patch = yamlencode({
|
||||
@@ -271,28 +215,6 @@ locals {
|
||||
network = { extraHostEntries = local.cp_host_entries }
|
||||
}
|
||||
})
|
||||
|
||||
# ── Per-CP patches (hostname + hcloud private NIC for etcd) ───────────────
|
||||
# eth0 = hcloud public (DHCP, default route); eth1 = hcloud private (etcd).
|
||||
# eth1 MUST be DHCP: hcloud private networks are SDN, not L2 — servers reach each
|
||||
# other via the network gateway, and hcloud's DHCP is what installs the private
|
||||
# IP (the one pinned in the hcloud_server network block) + the gateway route. A
|
||||
# static /24 here makes the node try direct same-subnet ARP, which the SDN doesn't
|
||||
# answer → the CPs can't reach each other → etcd never forms. VERIFY eth1 is the
|
||||
# private NIC on the snapshot (else use a deviceSelector).
|
||||
cp_node_patches = [for i in range(local.c.cp_count) : [
|
||||
yamlencode({
|
||||
machine = {
|
||||
network = {
|
||||
interfaces = [{
|
||||
interface = "eth1"
|
||||
dhcp = true
|
||||
}]
|
||||
}
|
||||
}
|
||||
}),
|
||||
yamlencode({ apiVersion = "v1alpha1", kind = "HostnameConfig", auto = "off", hostname = local.cp_hostnames[i] }),
|
||||
]]
|
||||
}
|
||||
|
||||
# Cluster PKI (sensitive).
|
||||
@@ -307,21 +229,9 @@ resource "talos_machine_secrets" "this" {
|
||||
}
|
||||
}
|
||||
|
||||
# Worker NetBird peers — their mesh IPs feed the CPs' /etc/hosts (cp_host_entries)
|
||||
# so the apiserver dials worker kubelets peer-to-peer. Lookup is by peer name
|
||||
# (= the worker hostname; the netbird stack keeps one live peer per node). Gated:
|
||||
# on a greenfield bootstrap the workers aren't peers yet — set
|
||||
# cluster.worker_mesh_kubelet = false, then flip it after they join.
|
||||
data "netbird_peer" "worker" {
|
||||
for_each = { for hostname, w in local.worker_node_map : hostname => w if local.c.worker_mesh_kubelet }
|
||||
name = each.key
|
||||
}
|
||||
|
||||
# Per-CP machine config — rendered into hcloud user_data (controlplane.tf). Each
|
||||
# CP gets the shared + CP-cluster + its own per-node patches.
|
||||
# CP base config — per-CP install/network/hostname patches are added at apply
|
||||
# time (controlplane.tf).
|
||||
data "talos_machine_configuration" "cp" {
|
||||
count = local.c.cp_count
|
||||
|
||||
cluster_name = local.c.name
|
||||
machine_type = "controlplane"
|
||||
cluster_endpoint = local.cluster_endpoint
|
||||
@@ -331,7 +241,6 @@ data "talos_machine_configuration" "cp" {
|
||||
config_patches = concat(
|
||||
local.cp_base_patches,
|
||||
[local.cp_cluster_patch, local.cp_extras_patch],
|
||||
local.cp_node_patches[count.index],
|
||||
local.common_firewall_patches,
|
||||
local.cp_firewall_patches,
|
||||
)
|
||||
@@ -349,16 +258,15 @@ data "talos_machine_configuration" "worker" {
|
||||
config_patches = concat(local.worker_base_patches, local.common_firewall_patches)
|
||||
}
|
||||
|
||||
# ── Bootstrap / kubeconfig / health (dial CP public IPs) ──────────────────────
|
||||
# ── Bootstrap / kubeconfig / health ───────────────────────────────────────────
|
||||
locals {
|
||||
cp_public_ips = hcloud_server.control_plane[*].ipv4_address
|
||||
# Everything the talos provider dials — bootstrap, kubeconfig, talosconfig, health,
|
||||
# and the helm/kubernetes providers — uses the PRIVATE kube-cp IPs, reachable from
|
||||
# the apply host over the NetBird kube-cp route (and in the cert SANs). No public
|
||||
# access is required to bring the cluster up (the CPs keep public IPs only for
|
||||
# NetBird NAT traversal + egress; apid/apiserver are firewalled off the internet).
|
||||
bootstrap_endpoint = local.cp_private_ips[0]
|
||||
operator_endpoint = "https://${local.cp_private_ips[0]}:6443"
|
||||
# Everything the talos provider dials — bootstrap, kubeconfig, talosconfig,
|
||||
# health, and the helm/kubernetes providers — uses the kube-cp IPs, reachable
|
||||
# from the apply host over the NetBird kube-cp route (and in the cert SANs).
|
||||
# During a greenfield bring-up the route appears as soon as the first CP boots
|
||||
# into the cluster and joins the mesh (the CPs are the route peers).
|
||||
bootstrap_endpoint = local.cp_ips[0]
|
||||
operator_endpoint = "https://${local.cp_ips[0]}:6443"
|
||||
}
|
||||
|
||||
# One-shot bootstrap against the first CP. Re-running rolls cluster identity.
|
||||
@@ -368,7 +276,7 @@ resource "talos_machine_bootstrap" "this" {
|
||||
endpoint = local.bootstrap_endpoint
|
||||
timeouts = { create = "10m" }
|
||||
|
||||
depends_on = [hcloud_server.control_plane]
|
||||
depends_on = [talos_machine_configuration_apply.cp]
|
||||
|
||||
lifecycle {
|
||||
# A replace re-bootstraps a LIVE cluster (identity roll). The re-bootstrap
|
||||
@@ -385,12 +293,12 @@ resource "talos_cluster_kubeconfig" "this" {
|
||||
depends_on = [talos_machine_bootstrap.this]
|
||||
}
|
||||
|
||||
# talosconfig endpoints = CP private kube-cp IPs (reached over NetBird; no public).
|
||||
# talosconfig endpoints = CP kube-cp IPs (reached over NetBird; no public).
|
||||
data "talos_client_configuration" "this" {
|
||||
cluster_name = local.c.name
|
||||
client_configuration = talos_machine_secrets.this.client_configuration
|
||||
endpoints = local.cp_private_ips
|
||||
nodes = concat(local.cp_private_ips, [for w in local.c.workers : w.fabric_ip])
|
||||
endpoints = local.cp_ips
|
||||
nodes = concat(local.cp_ips, [for w in local.c.workers : w.fabric_ip])
|
||||
}
|
||||
|
||||
locals {
|
||||
@@ -403,9 +311,9 @@ data "talos_cluster_health" "this" {
|
||||
count = var.bootstrap_health_gate ? 1 : 0
|
||||
|
||||
client_configuration = talos_machine_secrets.this.client_configuration
|
||||
control_plane_nodes = local.cp_private_ips
|
||||
control_plane_nodes = local.cp_ips
|
||||
worker_nodes = [for w in local.c.workers : w.fabric_ip]
|
||||
endpoints = local.cp_private_ips
|
||||
endpoints = local.cp_ips
|
||||
skip_kubernetes_checks = true
|
||||
timeouts = { read = "10m" }
|
||||
|
||||
|
||||
@@ -4,11 +4,10 @@ include "root" {
|
||||
|
||||
# clusters.auto.tfvars is loaded automatically by OpenTofu in this directory.
|
||||
# State backend + partition/region/stack (prod/htz-fsn1/talos) are derived by the
|
||||
# root config. Unlike the austin talos stack (which talks straight to bare-metal
|
||||
# nodes already in maintenance mode), this stack is HYBRID: it provisions the 3
|
||||
# Hetzner Cloud control-plane VMs (+ a small private subnet for etcd + the API
|
||||
# load balancer) AND drives Talos on the 3 bare-metal workers. CP↔worker traffic
|
||||
# rides the NetBird mesh; worker east-west rides the 50G fabric. See ./README.md.
|
||||
# root config. Like the austin talos stack, this talks straight to bare-metal
|
||||
# nodes already in Talos maintenance mode: 3 CPs on the kube-cp fabric VLAN
|
||||
# (etcd + API VIP) + 3 workers on the kube VLAN, routed by the spine IRBs.
|
||||
# See ./README.md.
|
||||
#
|
||||
# Secrets (HCLOUD_TOKEN, NetBird setup key, S3 state) are injected by
|
||||
# Secrets (NetBird setup keys, S3 state) are injected by
|
||||
# op run --env-file=tf/.env.prod
|
||||
|
||||
@@ -1,9 +1,10 @@
|
||||
# Hybrid prod cluster topology — the single source of truth (clusters.auto.tfvars).
|
||||
# One object, not a map: this stack's bring-up is bespoke (cloud CP via hcloud
|
||||
# user_data + bare-metal workers via apid apply), so a for_each map buys nothing.
|
||||
# All-bare-metal prod cluster topology — the single source of truth
|
||||
# (clusters.auto.tfvars). One object, not a map: this stack's bring-up is bespoke
|
||||
# (CPs + workers both driven over apid, but with different planes/volumes), so a
|
||||
# for_each map buys nothing.
|
||||
|
||||
variable "cluster" {
|
||||
description = "The prod hybrid Talos cluster (Star Wars name; prod = 'father')."
|
||||
description = "The prod bare-metal Talos cluster (Star Wars name; prod = 'father')."
|
||||
type = object({
|
||||
name = string
|
||||
talos_version = string
|
||||
@@ -12,47 +13,51 @@ variable "cluster" {
|
||||
# The Image Factory schematic (extension set) is managed in TF — see
|
||||
# schematic.yaml + talos_image_factory_schematic in image.tf. The schematic id
|
||||
# and image URLs derive from it, so they're NOT inputs here.
|
||||
install_disk = string
|
||||
|
||||
cilium_version = string
|
||||
hubble = bool
|
||||
|
||||
# NetBird mesh range node IPs come from (kubelet nodeIP.validSubnets). THIS
|
||||
# account assigns 10.254.0.0/15 (see clusters.auto.tfvars) — not the NetBird
|
||||
# Cloud default of 100.64.0.0/10. The node plane (CP↔worker) rides this mesh.
|
||||
# NetBird mesh range (host firewall trust + operator plane). THIS account
|
||||
# assigns 10.254.0.0/15 (see clusters.auto.tfvars) — not the NetBird Cloud
|
||||
# default of 100.64.0.0/10.
|
||||
netbird_node_cidr = string
|
||||
|
||||
# ── Cloud control plane (Hetzner Cloud) ──────────────────────────────────
|
||||
cp_count = number # 3
|
||||
# EXPLICIT node names (wordlist-style), one per CP, in cp_ip_offset order.
|
||||
# Names are PINNED — never auto-shuffled — so node identity can't silently
|
||||
# re-roll on a list edit (renaming a live node's hostname = renaming its
|
||||
# Kubernetes node = effectively replacing it).
|
||||
cp_names = list(string)
|
||||
cp_server_type = string # ccx23 (dedicated vCPU x86)
|
||||
cp_location = string # fsn1
|
||||
cp_ip_offset = number # CP[i] private (kube-cp) IP = cidrhost(kube_cp, offset+i)
|
||||
lb_type = string # lb11
|
||||
lb_ip_offset = number # API LB private IP = cidrhost(kube_cp, offset)
|
||||
lb_public = bool # also expose a public frontend (operators/workers reach it)
|
||||
|
||||
# ── Bare-metal workers (Hetzner Robot) ────────────────────────────────────
|
||||
# maint_ip = the Hetzner public IP the node comes up on in Talos maintenance
|
||||
# mode (DHCP) — the endpoint for the one-time config apply. fabric_ip = the
|
||||
# post-install kube (VLAN 10) address (nodeIP + worker east-west); the apiserver
|
||||
# reaches the kubelet there via NetBird→mgmt, and the node reaches the apiserver
|
||||
# over its own NetBird peer.
|
||||
workers = list(object({
|
||||
name = string # EXPLICIT node name (see cp_names) — keys the apply resources; renaming = node replacement
|
||||
# Install-disk NVMe serial — NOT a device name: nvme0/nvme1 enumeration is
|
||||
# not stable across boots (observed swapping), and a name-based install
|
||||
# target could point an upgrade at the DATA disk.
|
||||
# ── Bare-metal control planes (Hetzner Robot; kube-cp fabric VLAN 11) ─────
|
||||
# cp_ip = the post-install kube-cp (VLAN 11) address — etcd + apiserver +
|
||||
# nodeIP; the spine routes kube↔kube-cp. maint_ip = the Hetzner public IP the
|
||||
# node comes up on in Talos maintenance mode (DHCP on the onboard 1G NIC) —
|
||||
# the endpoint for the one-time install apply.
|
||||
cps = list(object({
|
||||
name = string # EXPLICIT node name (wordlist-style) — keys the apply resources; renaming = node replacement
|
||||
# Install-disk serial — NOT a device name: sda/sdb enumeration is not
|
||||
# stable across boots, and a name-based install target could point an
|
||||
# upgrade at the wrong disk.
|
||||
install_serial = string
|
||||
fabric_ip = string # 10.40.10.x on the kube fabric VLAN
|
||||
cp_ip = string # 10.40.11.x on the kube-cp fabric VLAN
|
||||
maint_ip = string # Hetzner public IP (maintenance-mode apid endpoint)
|
||||
robot_id = number # Hetzner Robot server number (provisioning/doc)
|
||||
provisioned = optional(bool, true) # false ONLY while first-provisioning: config
|
||||
# applies then target maint_ip (maintenance mode); true = target fabric_ip (live).
|
||||
# applies then target maint_ip (maintenance mode); true = target cp_ip (live).
|
||||
}))
|
||||
# CP fabric bond members (2×10G Intel 82599 SFP+). Selected by NIC driver —
|
||||
# ixgbe matches exactly the two 10G ports (the onboard 1G public NIC is e1000e).
|
||||
cp_bond_driver = optional(string)
|
||||
cp_bond_interfaces = optional(list(string), [])
|
||||
# API VIP = cidrhost(kube_cp, vip_offset) — Talos etcd-elected, floats between
|
||||
# the CPs on VLAN 11. 5 keeps the retired hcloud LB's IP, so the api_dns_name
|
||||
# record (NetBird DNS zone) carried over unchanged.
|
||||
vip_offset = number
|
||||
|
||||
# ── Bare-metal workers (Hetzner Robot; kube fabric VLAN 10) ───────────────
|
||||
# maint_ip/fabric_ip semantics as for cps; nodeIP = fabric_ip (worker east-west
|
||||
# rides VLAN 10 at 50G, apiserver↔kubelet routes via the spine IRBs).
|
||||
workers = list(object({
|
||||
name = string
|
||||
install_serial = string
|
||||
fabric_ip = string # 10.40.10.x on the kube fabric VLAN
|
||||
maint_ip = string
|
||||
robot_id = number
|
||||
provisioned = optional(bool, true)
|
||||
}))
|
||||
# Fabric bond members. Prefer worker_bond_driver (a Talos deviceSelector by NIC
|
||||
# driver, e.g. "bnxt_en") — robust across per-node PCI naming. worker_bond_interfaces
|
||||
@@ -62,21 +67,10 @@ variable "cluster" {
|
||||
# Worker default route (egress for image pulls + NetBird): via the kube fabric
|
||||
# IRB gateway (fabric transit) when true, else the Hetzner public NIC (DHCP).
|
||||
worker_default_route_via_fabric = optional(bool, true)
|
||||
|
||||
# apiserver→kubelet rides the mesh peer-to-peer: the CPs get /etc/hosts entries
|
||||
# mapping each worker hostname to its NetBird IP (data.netbird_peer lookups) and
|
||||
# the apiserver prefers the Hostname node address. Requires the workers to BE
|
||||
# NetBird peers — set false for a greenfield bootstrap (no peers to look up yet),
|
||||
# flip true once the workers have joined. See cp_extras_patch in talos.tf.
|
||||
worker_mesh_kubelet = optional(bool, true)
|
||||
})
|
||||
|
||||
validation {
|
||||
condition = length(var.cluster.cp_names) == var.cluster.cp_count
|
||||
error_message = "cluster.cp_names must have exactly cp_count entries (one name per CP, in cp_ip_offset order)."
|
||||
}
|
||||
validation {
|
||||
condition = length(distinct(concat(var.cluster.cp_names, var.cluster.workers[*].name))) == var.cluster.cp_count + length(var.cluster.workers)
|
||||
condition = length(distinct(concat(var.cluster.cps[*].name, var.cluster.workers[*].name))) == length(var.cluster.cps) + length(var.cluster.workers)
|
||||
error_message = "Node names must be unique across CPs and workers."
|
||||
}
|
||||
}
|
||||
|
||||
@@ -6,17 +6,6 @@ terraform {
|
||||
source = "siderolabs/talos"
|
||||
version = "~> 0.11"
|
||||
}
|
||||
# Hetzner Cloud — the 3 control-plane VMs, their private subnet (etcd), and the
|
||||
# public API load balancer. Token via HCLOUD_TOKEN (op run --env-file).
|
||||
hcloud = {
|
||||
source = "hetznercloud/hcloud"
|
||||
version = "~> 1.51"
|
||||
}
|
||||
# Hostname picks for the talos nodes (node-names module → random_shuffle).
|
||||
random = {
|
||||
source = "hashicorp/random"
|
||||
version = "~> 3.6"
|
||||
}
|
||||
# Cilium install (CNI) post-bootstrap, in the same apply.
|
||||
helm = {
|
||||
source = "hashicorp/helm"
|
||||
@@ -33,11 +22,5 @@ terraform {
|
||||
source = "1Password/onepassword"
|
||||
version = "~> 2.1"
|
||||
}
|
||||
# Worker NetBird peer lookups — the mesh addresses the CP apiserver dials for
|
||||
# worker kubelets (see cp_extras_patch). Auth via NB_PAT (op run).
|
||||
netbird = {
|
||||
source = "registry.terraform.io/futo-org/netbird"
|
||||
version = "1.0.2"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,18 +1,16 @@
|
||||
# ── Bare-metal workers ───────────────────────────────────────────────────────
|
||||
# Applied over apid to nodes already in Talos maintenance mode at their fabric_ip
|
||||
# (reachable from the TF runner via NetBird → mgmt → fabric `kube` net). After the
|
||||
# install+reboot they keep that fabric IP and join the NetBird mesh.
|
||||
# or maint_ip (see below). After the install+reboot they keep that fabric IP and
|
||||
# join the NetBird mesh (operator/backup plane only).
|
||||
#
|
||||
# bond0 (2×25G LACP) → vlan 10 (kube) = fabric_ip — nodeIP + worker east-west (50G)
|
||||
# default route via the kube IRB gateway (fabric transit) for egress
|
||||
# bond0 (2×25G LACP) → vlan 10 (kube) = fabric_ip — nodeIP + worker east-west (50G)
|
||||
# route to kube-cp (apiserver + VIP) via the kube IRB (10.40.10.1) — the fabric
|
||||
# path to the control plane; the old wt0 (NetBird) route is retired
|
||||
# default route via the Hetzner public NIC (DHCP) for egress
|
||||
#
|
||||
# Workers are NOT NetBird peers: nodeIP = fabric_ip, so Cilium autoDirectNodeRoutes
|
||||
# routes pod east-west directly over the shared VLAN-10 L2 (no BGP). The apiserver
|
||||
# reaches worker kubelets via the CPs' NetBird route to the kube net (mgmt routers).
|
||||
#
|
||||
# Workers are PROVISIONED to maintenance mode out of band — see the Phase-4 runbook
|
||||
# (./README.md): Hetzner rescue → dd the Talos metal image → bring up bond0.10 at
|
||||
# fabric_ip. This stack assumes they're already there.
|
||||
# Workers are PROVISIONED to maintenance mode out of band — see the runbook
|
||||
# (./README.md): Hetzner rescue → dd the Talos metal image → reboot. This stack
|
||||
# assumes they're already there.
|
||||
|
||||
locals {
|
||||
kube_prefix = split("/", local.kube_cidr)[1] # 24
|
||||
@@ -23,7 +21,7 @@ locals {
|
||||
yamlencode({
|
||||
machine = { install = {
|
||||
diskSelector = { serial = w.install_serial }
|
||||
image = local.worker_install_image
|
||||
image = local.install_image
|
||||
} }
|
||||
}),
|
||||
yamlencode({
|
||||
@@ -47,9 +45,14 @@ locals {
|
||||
vlans = [{
|
||||
vlanId = module.addr_site.kube_vlan_id # 10
|
||||
addresses = ["${w.fabric_ip}/${local.kube_prefix}"]
|
||||
routes = var.cluster.worker_default_route_via_fabric ? [
|
||||
{ network = "0.0.0.0/0", gateway = local.kube_gateway },
|
||||
] : []
|
||||
# kube-cp (apiserver + API VIP) lives one IRB away — pin the route so
|
||||
# kubelet→apiserver + geneve to the CPs ride the fabric, not the mesh.
|
||||
routes = concat(
|
||||
[{ network = local.kube_cp_cidr, gateway = local.kube_gateway }],
|
||||
var.cluster.worker_default_route_via_fabric ? [
|
||||
{ network = "0.0.0.0/0", gateway = local.kube_gateway },
|
||||
] : [],
|
||||
)
|
||||
}]
|
||||
}]
|
||||
}
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# Preprovisioned spine VC + the 100G->4x25G breakout. Breakout (channel-speed)
|
||||
# has no typed jeremmfr resource, so it's pushed as raw set-config. The
|
||||
# `aggregated-devices ethernet device-count` line is auto-managed by
|
||||
# Preprovisioned spine VC + the per-port breakout channelization. Breakout
|
||||
# (channel-speed) has no typed jeremmfr resource, so it's pushed as raw set-config.
|
||||
# The `aggregated-devices ethernet device-count` line is auto-managed by
|
||||
# junos_interface_physical (computed from the ae interfaces) — not set here.
|
||||
resource "junos_virtual_chassis" "spine" {
|
||||
preprovisioned = true
|
||||
@@ -19,8 +19,8 @@ resource "junos_null_load_config" "breakout" {
|
||||
action = "set"
|
||||
config = join("\n", flatten([
|
||||
for fpc in [0, 1] : [
|
||||
for p in var.breakout_ports :
|
||||
"set chassis fpc ${fpc} pic 0 port ${p} channel-speed ${var.breakout_speed}"
|
||||
for p, speed in var.breakout_ports :
|
||||
"set chassis fpc ${fpc} pic 0 port ${p} channel-speed ${speed}"
|
||||
]
|
||||
]))
|
||||
}
|
||||
|
||||
@@ -74,6 +74,35 @@ resource "junos_interface_physical" "node_lag" {
|
||||
vlan_members = ["vlan${var.kube_vlan_id}"]
|
||||
}
|
||||
|
||||
# Control-plane node bonds — same pattern as node_lags, but the trunk carries the
|
||||
# kube-cp VLAN (the CPs' only fabric presence; kube↔kube-cp routes via the IRBs).
|
||||
locals {
|
||||
cp_node_lag_members = merge([for ae, ports in var.cp_node_lags : { for p in ports : p => ae }]...)
|
||||
}
|
||||
|
||||
resource "junos_interface_physical" "cp_node_lag_member" {
|
||||
for_each = local.cp_node_lag_members
|
||||
name = each.key
|
||||
ether_opts {
|
||||
ae_8023ad = each.value
|
||||
}
|
||||
}
|
||||
|
||||
resource "junos_interface_physical" "cp_node_lag" {
|
||||
for_each = var.cp_node_lags
|
||||
name = each.key
|
||||
mtu = 9216
|
||||
parent_ether_opts {
|
||||
lacp {
|
||||
mode = "active"
|
||||
}
|
||||
}
|
||||
trunk = true
|
||||
vlan_members = ["vlan${var.kube_cp.vlan_id}"]
|
||||
|
||||
depends_on = [junos_vlan.this]
|
||||
}
|
||||
|
||||
# Management-node ports (mgmt-1, mgmt-2) — one channelized port-3 leg per VC member,
|
||||
# each a single-port trunk of the stretched VLANs. Identical config per node.
|
||||
resource "junos_interface_physical" "mgmt_node" {
|
||||
|
||||
@@ -27,15 +27,14 @@ variable "vc_member_serials" {
|
||||
}
|
||||
|
||||
variable "breakout_ports" {
|
||||
type = list(number)
|
||||
default = [0, 1, 2, 3]
|
||||
description = "QSFP28 ports channelized 100G->4x25G on each VC member."
|
||||
}
|
||||
|
||||
variable "breakout_speed" {
|
||||
type = string
|
||||
default = "25g"
|
||||
description = "Per-channel speed for the breakout ports."
|
||||
type = map(string)
|
||||
default = { 0 = "25g", 1 = "25g", 2 = "25g", 3 = "25g" }
|
||||
description = <<-EOT
|
||||
QSFP28 ports channelized on each VC member: port number -> per-channel speed.
|
||||
25g -> et-<fpc>/0/<port>:0..3 legs; 10g -> xe-<fpc>/0/<port>:0..3. NB: 10g
|
||||
channelization needs a QSFP+ (40G-class) breakout cable — the QFX5200 silently
|
||||
falls back to unchannelized 100G on a QSFP28 cable.
|
||||
EOT
|
||||
}
|
||||
|
||||
variable "kube_vlan_id" {
|
||||
@@ -72,6 +71,31 @@ variable "node_lags" {
|
||||
EOT
|
||||
}
|
||||
|
||||
variable "cp_node_lags" {
|
||||
type = map(list(string))
|
||||
default = {}
|
||||
description = <<-EOT
|
||||
Control-plane node LACP bonds terminated on the core — same shape and rules as
|
||||
node_lags (key = ae name; value = the two member sub-ports, one per VC member,
|
||||
cabled to the SAME node), but the trunk carries the kube-cp VLAN instead of
|
||||
kube. Requires var.kube_cp. Members are 10G breakout legs (xe-…).
|
||||
EOT
|
||||
}
|
||||
|
||||
variable "kube_cp" {
|
||||
type = object({
|
||||
vlan_id = number
|
||||
cidr = string
|
||||
})
|
||||
default = null
|
||||
description = <<-EOT
|
||||
Kubernetes control-plane network on the fabric: creates the kube-cp VLAN + its
|
||||
IRB (.1) on the spine — the second spine IRB, making the spine the router
|
||||
between kube (workers) and kube-cp (bare-metal CPs: etcd + the API VIP).
|
||||
null = no kube-cp VLAN.
|
||||
EOT
|
||||
}
|
||||
|
||||
variable "node_bgp" {
|
||||
type = object({
|
||||
peer_range = string # the kube CIDR — nodes dynamic-peer from it; the IRB is its .1
|
||||
|
||||
@@ -1,14 +1,17 @@
|
||||
# Stretched VLANs — L2 only on the spine EXCEPT the kube VLAN, which gets an IRB when
|
||||
# node_bgp is set (the spine's only L3 interface, = the Cilium iBGP peer + VLAN-10
|
||||
# gateway; see bgp-nodes.tf). Other gateways live on the leaves.
|
||||
# Stretched VLANs — L2 only on the spine EXCEPT the kube VLAN (IRB when node_bgp
|
||||
# is set: the Cilium iBGP peer + VLAN-10 gateway, see bgp-nodes.tf) and the
|
||||
# kube-cp VLAN (IRB when kube_cp is set: the CPs' gateway — the spine routes
|
||||
# kube↔kube-cp). Other gateways live on the leaves.
|
||||
locals {
|
||||
spine_vlans = {
|
||||
spine_vlans = merge({
|
||||
"vlan${var.public_vlan_id}" = { id = var.public_vlan_id, l3 = null }
|
||||
"vlan${var.private_vlan_id}" = { id = var.private_vlan_id, l3 = null }
|
||||
"vlan${var.kube_vlan_id}" = { id = var.kube_vlan_id, l3 = var.node_bgp == null ? null : "irb.${var.kube_vlan_id}" }
|
||||
"vlan${var.mgmt_vlan_id}" = { id = var.mgmt_vlan_id, l3 = null }
|
||||
"vlan${var.host_mgmt_vlan_id}" = { id = var.host_mgmt_vlan_id, l3 = null }
|
||||
}
|
||||
}, var.kube_cp == null ? {} : {
|
||||
"vlan${var.kube_cp.vlan_id}" = { id = var.kube_cp.vlan_id, l3 = "irb.${var.kube_cp.vlan_id}" }
|
||||
})
|
||||
}
|
||||
|
||||
resource "junos_vlan" "this" {
|
||||
@@ -17,3 +20,13 @@ resource "junos_vlan" "this" {
|
||||
vlan_id = tostring(each.value.id)
|
||||
l3_interface = each.value.l3
|
||||
}
|
||||
|
||||
# kube-cp IRB — the spine is the kube-cp gateway (.1). Bare-metal CPs sit on this
|
||||
# VLAN only; worker↔CP (kubelet↔apiserver, geneve) routes irb.<kube>↔irb.<kube-cp>.
|
||||
resource "junos_interface_logical" "kube_cp_irb" {
|
||||
count = var.kube_cp == null ? 0 : 1
|
||||
name = "irb.${var.kube_cp.vlan_id}"
|
||||
family_inet {
|
||||
address { cidr_ip = "${cidrhost(var.kube_cp.cidr, 1)}/${split("/", var.kube_cp.cidr)[1]}" }
|
||||
}
|
||||
}
|
||||
|
||||
@@ -11,16 +11,15 @@ locals {
|
||||
kube_cidr = "10.${var.site_id}.${var.kube_octet}.0/24"
|
||||
kube_vlan_id = var.kube_octet
|
||||
|
||||
# Site-global Kubernetes control-plane subnet ("kube-cp"). NOT a Juniper fabric
|
||||
# VLAN — it's a small, isolated Hetzner Cloud private subnet holding ONLY the
|
||||
# cloud control-plane VMs (for etcd CP↔CP) + the Kubernetes API LB's private IP.
|
||||
# The workers do NOT join it: CP↔worker control + worker→API ride the NetBird
|
||||
# WireGuard mesh (node IPs are NetBird addresses), and worker↔worker east-west
|
||||
# rides the `kube` fabric net at 50G. Carved from the site supernet only for
|
||||
# collision-free IPAM; it is NEVER configured on the Junos switches.
|
||||
# kube-cp 10.<site>.<kube_cp_octet>.0/24 (Hetzner Cloud subnet, gw .1)
|
||||
# Site-global Kubernetes control-plane VLAN ("kube-cp") — a fabric VLAN like
|
||||
# `kube`: holds the bare-metal control-plane nodes (etcd CP↔CP + apiserver) and
|
||||
# the cluster's API VIP. Workers do NOT join it — the spine routes kube↔kube-cp
|
||||
# via its two IRBs. (Historically this was an isolated Hetzner Cloud private
|
||||
# subnet for the cloud CP VMs + API LB; same CIDR, now on the switches.)
|
||||
# kube-cp 10.<site>.<kube_cp_octet>.0/24 -> vlan <kube_cp_octet> (gw .1 = spine IRB)
|
||||
kube_cp_cidr = "10.${var.site_id}.${var.kube_cp_octet}.0/24"
|
||||
kube_cp_gateway = cidrhost(local.kube_cp_cidr, 1) # .1 — Hetzner Cloud Gateway
|
||||
kube_cp_vlan_id = var.kube_cp_octet
|
||||
kube_cp_gateway = cidrhost(local.kube_cp_cidr, 1) # .1 — spine IRB
|
||||
|
||||
# Internal (NetBird-only) Kubernetes LoadBalancer VIP range. Like kube-cp it is
|
||||
# NEVER a switch VLAN: Cilium assigns VIPs from it and the workers advertise the
|
||||
|
||||
@@ -90,12 +90,17 @@ output "kube_vlan_id" {
|
||||
|
||||
output "kube_cp_cidr" {
|
||||
value = local.kube_cp_cidr
|
||||
description = "Site-global Kubernetes control-plane subnet 'kube-cp' (10.<site>.<kube_cp_octet>.0/24) — an isolated Hetzner Cloud private subnet for the CP VMs (etcd) + the API LB. NOT a fabric VLAN."
|
||||
description = "Site-global Kubernetes control-plane network 'kube-cp' (10.<site>.<kube_cp_octet>.0/24), a fabric VLAN — bare-metal CPs (etcd) + the API VIP."
|
||||
}
|
||||
|
||||
output "kube_cp_vlan_id" {
|
||||
value = local.kube_cp_vlan_id
|
||||
description = "Site-global kube-cp VLAN id (== kube_cp_octet, e.g. 11)."
|
||||
}
|
||||
|
||||
output "kube_cp_gateway" {
|
||||
value = local.kube_cp_gateway
|
||||
description = "Hetzner Cloud Gateway (.1) for the kube-cp subnet."
|
||||
description = "Spine IRB gateway (.1) for the kube-cp network."
|
||||
}
|
||||
|
||||
output "lb_internal_cidr" {
|
||||
|
||||
@@ -62,10 +62,12 @@ resource "netbox_prefix" "network" {
|
||||
}
|
||||
|
||||
resource "netbox_ip_address" "gateway" {
|
||||
for_each = local.networks
|
||||
ip_address = "${each.value.gateway}/${split("/", each.value.prefix)[1]}"
|
||||
status = "active"
|
||||
dns_name = "gw-${var.site.code}-C${split("-", each.key)[0]}-${lower(each.value.role)}"
|
||||
for_each = local.networks
|
||||
ip_address = "${each.value.gateway}/${split("/", each.value.prefix)[1]}"
|
||||
status = "active"
|
||||
# lower(): NetBox normalizes dns_name to lowercase on write — mixed case here
|
||||
# is a perpetual plan diff.
|
||||
dns_name = lower("gw-${var.site.code}-c${split("-", each.key)[0]}-${each.value.role}")
|
||||
description = "IRB gateway for ${var.site.code}-C${split("-", each.key)[0]}-${each.value.role}"
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user