From f409d321b75aab60bcb6b4e8f6d939c9105706b3 Mon Sep 17 00:00:00 2001 From: Devin Buhl Date: Wed, 1 Jul 2026 06:31:59 -0400 Subject: [PATCH] feat: migrate tailscale to netbird (#84) Signed-off-by: Devin Buhl --- .mise/config.toml | 14 ++-- README.md | 6 +- deployment/.env | 7 +- .../netbird/cluster/.terraform.lock.hcl | 38 +++++++++ deployment/modules/netbird/cluster/config.tf | 10 +++ deployment/modules/netbird/cluster/netbird.tf | 82 +++++++++++++++++++ deployment/modules/netbird/cluster/outputs.tf | 6 ++ .../modules/netbird/cluster/providers.tf | 5 ++ .../cluster}/terragrunt.hcl | 23 +++++- .../modules/netbird/cluster/variables.tf | 11 +++ deployment/modules/ovh/account/variables.tf | 12 +-- deployment/modules/ovh/account/workers.tf | 2 +- .../tailscale/account/.terraform.lock.hcl | 35 -------- .../modules/tailscale/account/config.tf | 10 --- .../modules/tailscale/account/providers.tf | 6 -- .../modules/tailscale/account/tailscale.tf | 59 ------------- .../modules/tailscale/account/variables.tf | 20 ----- .../modules/talos/cluster/.terraform.lock.hcl | 33 -------- deployment/modules/talos/cluster/config.tf | 4 - .../modules/talos/cluster/controlplane.tf | 50 +++-------- deployment/modules/talos/cluster/outputs.tf | 7 -- deployment/modules/talos/cluster/providers.tf | 7 -- .../modules/talos/cluster/terragrunt.hcl | 8 +- deployment/modules/talos/cluster/variables.tf | 17 ++-- deployment/modules/talos/cluster/workers.tf | 31 ++----- docs/01-bootstrap-guide.md | 20 ++--- docs/02-infrastructure-architecture-guide.md | 10 +-- docs/03-cluster-architecture-guide.md | 12 +-- 28 files changed, 246 insertions(+), 299 deletions(-) create mode 100644 deployment/modules/netbird/cluster/.terraform.lock.hcl create mode 100644 deployment/modules/netbird/cluster/config.tf create mode 100644 deployment/modules/netbird/cluster/netbird.tf create mode 100644 deployment/modules/netbird/cluster/outputs.tf create mode 100644 deployment/modules/netbird/cluster/providers.tf rename deployment/modules/{tailscale/account => netbird/cluster}/terragrunt.hcl (53%) create mode 100644 deployment/modules/netbird/cluster/variables.tf delete mode 100644 deployment/modules/tailscale/account/.terraform.lock.hcl delete mode 100644 deployment/modules/tailscale/account/config.tf delete mode 100644 deployment/modules/tailscale/account/providers.tf delete mode 100644 deployment/modules/tailscale/account/tailscale.tf delete mode 100644 deployment/modules/tailscale/account/variables.tf delete mode 100644 deployment/modules/talos/cluster/providers.tf diff --git a/.mise/config.toml b/.mise/config.toml index e04c027..2be0493 100644 --- a/.mise/config.toml +++ b/.mise/config.toml @@ -25,8 +25,8 @@ TF_VAR_dist_dir="{{config_root}}/dist" ENVIRONMENT_SHORT = "{% set e = get_env(name='ENVIRONMENT', default='development') %}{% if e == 'development' or e == '' %}dev{% elif e == 'production' %}prod{% else %}{{ e }}{% endif %}" # Control plane and workers boot from different Talos Factory schematics: -# control plane (Public Cloud / KVM): tailscale + qemu-guest-agent -# worker (bare metal): tailscale only — qemu-guest-agent wedges +# control plane (Public Cloud / KVM): qemu-guest-agent + netbird +# worker (bare metal): netbird only — qemu-guest-agent wedges # bare-metal boot and reboot-loops the node # Control planes need the image uploaded to OVH glance (talos:{dl,ul}:cp); workers # are OVH BYOI, so OVH fetches their raw from the Factory at order time (no upload). @@ -34,10 +34,10 @@ ENVIRONMENT_SHORT = "{% set e = get_env(name='ENVIRONMENT', default='development [tasks."talos:dl:cp"] run = """ mkdir -p {{config_root}}/.private/dist -wget https://factory.talos.dev/image/7d4c31cbd96db9f90c874990697c523482b2bae27fb4631d5583dcd9c281b1ff/v1.13.0/openstack-amd64.raw.xz -O {{config_root}}/.private/dist/talos.1.13.0-tailscale-qemu.raw.xz -unxz {{config_root}}/.private/dist/talos.1.13.0-tailscale-qemu.raw.xz --force +wget https://factory.talos.dev/image/bbfcb7053b1609712a977830952455432825890922cb6bac23cea34b980970f1/v1.13.5/openstack-amd64.raw.xz -O {{config_root}}/.private/dist/talos.1.13.5-qemu-netbird.raw.xz +unxz {{config_root}}/.private/dist/talos.1.13.5-qemu-netbird.raw.xz --force """ -description = "Download the control-plane Talos image (OpenStack, tailscale + qemu-guest-agent schematic)" +description = "Download the control-plane Talos image (OpenStack, qemu-guest-agent + netbird schematic)" dir = "{{cwd}}" [tasks."talos:ul:cp"] @@ -45,9 +45,9 @@ run = """ for region in "RBX-A" "GRA9" "EU-WEST-PAR"; do echo "Uploading to region ${region}..." source {{config_root}}/.private/openstack/${ENVIRONMENT}/openrc.sh - openstack image create "talos-1.13.0-tailscale-qemu" \ + openstack image create "talos-1.13.5-qemu-netbird" \ --os-region "${region}" \ - --file {{config_root}}/.private/dist/talos.1.13.0-tailscale-qemu.raw \ + --file {{config_root}}/.private/dist/talos.1.13.5-qemu-netbird.raw \ --disk-format raw \ --container-format bare \ --property hw_qemu_guest_agent=yes \ diff --git a/README.md b/README.md index 7e5d60b..f76627c 100644 --- a/README.md +++ b/README.md @@ -4,7 +4,7 @@ The centralized observability platform for FUTO services. A single Talos Kuberne Built for geographic resilience with single-cluster operational simplicity: three control planes in low-latency DCs hold one etcd quorum, three bare-metal workers carry the observability workload, and everything talks over a private OVH vRack. -**Status:** staging is built and running (`o11y-staging`); production is planned (`o11y-production`). +**Status:** both environments are built and running — `o11y-staging` and `o11y-production`. ## Documentation @@ -20,7 +20,7 @@ Built for geographic resilience with single-cluster operational simplicity: thre ```text deployment/modules/ ├── ovh/account/ # cloud project, vRack, private network, CPs, workers, IPLB, DNS -├── tailscale/account/ # tailnet-global ACL and tailnet settings +├── netbird/cluster/ # per-env mesh: node group, setup key, vRack network route, access policy ├── talos/cluster/ # machine secrets, CP + worker configs, bootstrap, ingress firewall └── kubernetes/helm/ # Flux Operator + Instance, env-scoped secrets @@ -34,4 +34,4 @@ kubernetes/ └── cluster-settings.yaml # per-env ConfigMap: APP_DOMAIN, CLUSTER_NAME ``` -State lives in S3 under `yucca/o11y/v3//`. Secrets and OVH/Tailscale tokens come from the environment's 1Password vault via `op run` and `deployment/.env`. +State lives in S3 under `yucca/o11y/v3//`. Secrets and OVH/NetBird tokens come from 1Password via `op run` and `deployment/.env`. diff --git a/deployment/.env b/deployment/.env index 7e44fe4..697705f 100644 --- a/deployment/.env +++ b/deployment/.env @@ -10,10 +10,9 @@ export TF_VAR_tf_state_s3_region=op://o11y_tf/TF_STATE_S3_REGION/password export TF_VAR_tf_state_s3_access_key=op://o11y_tf/TF_STATE_S3_ACCESS_KEY/password export TF_VAR_tf_state_s3_secret_key=op://o11y_tf/TF_STATE_S3_SECRET_KEY/password -export TF_VAR_tailscale_oauth_client_id=op://o11y_tf/TAILSCALE_OAUTH_CLIENT_ID/password -export TF_VAR_tailscale_oauth_client_secret=op://o11y_tf/TAILSCALE_OAUTH_CLIENT_SECRET/password -export TF_VAR_tailscale_tailnet_id=op://o11y_tf/TAILSCALE_TAILNET_ID/password - export TF_VAR_op_credentials_file=op://o11y_tf/1PASS_CONNECT_SERVER_CREDENTIALS_FILE/password export TF_VAR_op_connect_token=op://o11y_tf/1PASS_CONNECT_O11Y_SUPERUSER/password export TF_VAR_op_connect_token_env=op://o11y_tf_${ENVIRONMENT_SHORT}/1PASS_CONNECT_O11Y_READ/password + +# Netbird Cloud PAT (shared across envs) — drives the netbird/account TF provider. +export TF_VAR_netbird_tf_pat=op://shared_tf/NETBIRD_TF_PAT/password diff --git a/deployment/modules/netbird/cluster/.terraform.lock.hcl b/deployment/modules/netbird/cluster/.terraform.lock.hcl new file mode 100644 index 0000000..b5ddef2 --- /dev/null +++ b/deployment/modules/netbird/cluster/.terraform.lock.hcl @@ -0,0 +1,38 @@ +# This file is maintained automatically by "tofu init". +# Manual edits may be lost in future updates. + +provider "registry.opentofu.org/netbirdio/netbird" { + version = "0.0.9" + constraints = "0.0.9" + hashes = [ + "h1:0Vi0MLMk+K1CEuBZB+ByU55wXx87eGd5j0fGACekCDQ=", + "h1:2ska0C0jDxvbjApKlUmdkfrXQFjHiNf/KKYt355Isgo=", + "h1:H8GG0MZqAAXqnrMCw0Q3vcE3gGFEqYmDF9YY2RnD18Y=", + "h1:HKmwtSPE6k++umLH99WkpT/HcNzy69OWRTVZMsDFJh8=", + "h1:KFbgGf03fTGEgJRIpzAryWB26BHicfLQC8nuE5IiSys=", + "h1:VMUHvZOWIU/IWmK+NyEUIafKS88gGTvff+nK8N7NE3E=", + "h1:aFK3rhEjnnwiHU7Fl/nBDz5Y1b/2sRYXYomu4qewlwM=", + "h1:ba2yqgEM9Ssy+C/0nP3XHh209Y20gpYIErZtYVVb2k8=", + "h1:hC4RZOIVv5KTHcF+AuIHQe01hJmfw7yioSRTDb4ImDg=", + "h1:jaLhsZX1wOdBWKED+hJ0kyfV5TE3Laim6Kd7Uw+krzQ=", + "h1:mWLdkV/wXnK6SSMIKw+BJFJPonrts+7L5IEefRQ0rV8=", + "h1:nvVlkGxQSfphiXMNgGmvuTZQOOjhtdHMe/Upx3/HWjA=", + "h1:pjY52PZ8FJXUbeFcB0kSV8guz4h/xzlXLarWXelAwes=", + "h1:trYvM2c8h5lQ0m8+3Bgvgnbkm04+VOAL2pr6odmFfS8=", + "zh:1f34fba3ecfe0efa36d4b4bddf5714c69124e705d1e371e65abd291887d43385", + "zh:203f671af4d1f5376f4e20fb8d82cc8dc6c4fd136d9abffe8eb7b7f34e27197b", + "zh:21cb344302bdbadbc2779768116c7351d915bb7678a4847e9e2c8623d0032c48", + "zh:2368868e0f86b458f83b6598317ff5fde0d4079500d4567e99f46bf80c63f07a", + "zh:3237a426818eb3906153b1cf7fb6dfe9d52128972e7cab17c324d88788f86863", + "zh:427f8e8ba190d69cfd493b1598b4bf7940fe9a521a333df3540d2e0ea07543cf", + "zh:689cb219f3936500f5a3147299961f8740e505612b5182d93840cb8152d89b4f", + "zh:6e594e69d9e107c45ed4677a9bff68de2dd4232900b636327b24c819bee72c9e", + "zh:715634d6d0b052ac3f34aeb4c2df42960cdf44dd5c45d060867db78716aebb48", + "zh:890df766e9b839623b1f0437355032a3c006226a6c200cd911e15ee1a9014e9f", + "zh:8cbbdde5a0b59f0bb427b403da8993b215d1a6231e0d6b9b4faa89db76b10705", + "zh:af92281c3e14c6af53fb93cee584c5118c5c7bee0acb938b5a367a219d9bc2e2", + "zh:bcdeef5fadf092e33228f28f013af7d8d9553b9d2fb2cecd1894351d5ad37a7a", + "zh:d6c31cb12f18b25c9663c14e737554e06144d4e270e607337bac7963ca3b9542", + "zh:e5873e8e0b29c8cbfd98f04759579da3c0529c8b38608c0f1ea92eb566c8ddc1", + ] +} diff --git a/deployment/modules/netbird/cluster/config.tf b/deployment/modules/netbird/cluster/config.tf new file mode 100644 index 0000000..92319f0 --- /dev/null +++ b/deployment/modules/netbird/cluster/config.tf @@ -0,0 +1,10 @@ +terraform { + required_version = "~> 1.10" + + required_providers { + netbird = { + source = "netbirdio/netbird" + version = "0.0.9" + } + } +} diff --git a/deployment/modules/netbird/cluster/netbird.tf b/deployment/modules/netbird/cluster/netbird.tf new file mode 100644 index 0000000..387a42b --- /dev/null +++ b/deployment/modules/netbird/cluster/netbird.tf @@ -0,0 +1,82 @@ +# Per-env o11y NetBird objects (state key .../netbird/cluster/). Everything is +# per-env — resource groups and policies included — so access can differ by env +# (Zack: some people get dev/staging but not prod). No account-wide layer. +# +# All object names are UPPER_SNAKE to match yucca's convention (groups, setup keys, +# networks, network resources, and policies/rules). + +# Existing account-wide users group (yucca peers — populated via users' auto_groups, +# per NetBird's model). Referenced as the policy source so the same operators that +# reach yucca also reach o11y — per the team decision. +data "netbird_group" "yucca" { + name = "yucca" +} + +# Group the Talos nodes auto-join via the setup key below. +resource "netbird_group" "talos" { + name = "O11Y_${upper(var.env)}_TALOS" +} + +# Per-env tag for this cluster's routed resources (the vRack subnet below tags into +# it; the yucca->resource policy grants it). Created with its final name — the +# NetBird provider can't rename a group once network resources are tagged into it. +resource "netbird_group" "o11y_resource" { + name = "O11Y_${upper(var.env)}_RESOURCE" +} + +# Reusable, non-ephemeral enrollment key for the Talos nodes — fed to the netbird +# Talos extension as NB_SETUP_KEY. NOT ephemeral: ephemeral peers are reaped after +# 10m idle, which would delete live nodes. +resource "netbird_setup_key" "talos" { + name = "O11Y_${upper(var.env)}_TALOS" + type = "reusable" + ephemeral = false + expiry_seconds = 0 # unlimited — nodes re-enroll with the same key on reprovision + usage_limit = 0 # unlimited + auto_groups = [netbird_group.talos.id] +} + +# vRack subnet advertised to operators with the Talos nodes as routing peers — the +# Netbird "Networks" model. Every node sits on the vRack, so any can route (HA); +# masquerade NATs operator traffic to the routing peer's vRack IP, which the node +# firewall already trusts. +resource "netbird_network" "vrack" { + name = "O11Y_${upper(var.env)}_VRACK" + description = "o11y ${var.env} vRack private subnet" +} + +resource "netbird_network_router" "vrack" { + network_id = netbird_network.vrack.id + peer_groups = [netbird_group.talos.id] + masquerade = true + metric = 9999 + enabled = true +} + +resource "netbird_network_resource" "vrack" { + network_id = netbird_network.vrack.id + # Per-env: Netbird network-resource names are account-globally unique. + name = "O11Y_${upper(var.env)}_VRACK_CIDR" + address = var.private_network_cidr + groups = [netbird_group.o11y_resource.id] + enabled = true +} + +# NetBird default-denies; allow yucca operators to reach THIS env's routed subnet on +# the Talos management ports only (apid + kube-apiserver). Per-env policy so access +# can later differ by env — tighter than yucca's all-protocol yucca->yucca_resource. +resource "netbird_policy" "yucca_to_o11y_resource" { + name = "O11Y_${upper(var.env)}_YUCCA_TO_RESOURCE" + enabled = true + + rule { + name = "YUCCA_TO_O11Y_RESOURCE" + action = "accept" + protocol = "tcp" + enabled = true + bidirectional = false + sources = [data.netbird_group.yucca.id] + destinations = [netbird_group.o11y_resource.id] + ports = ["50000", "6443"] + } +} diff --git a/deployment/modules/netbird/cluster/outputs.tf b/deployment/modules/netbird/cluster/outputs.tf new file mode 100644 index 0000000..7f4cfdc --- /dev/null +++ b/deployment/modules/netbird/cluster/outputs.tf @@ -0,0 +1,6 @@ +# Plaintext setup key fed to the Talos netbird extension (NB_SETUP_KEY) by the +# talos/cluster module, which consumes this via a terragrunt dependency. +output "talos_setup_key" { + sensitive = true + value = netbird_setup_key.talos.key +} diff --git a/deployment/modules/netbird/cluster/providers.tf b/deployment/modules/netbird/cluster/providers.tf new file mode 100644 index 0000000..ee1d4b0 --- /dev/null +++ b/deployment/modules/netbird/cluster/providers.tf @@ -0,0 +1,5 @@ +provider "netbird" { + # PAT from the shared_tf vault (Netbird Cloud). management_url defaults to + # https://api.netbird.io, so it is left unset. + token = var.netbird_tf_pat +} diff --git a/deployment/modules/tailscale/account/terragrunt.hcl b/deployment/modules/netbird/cluster/terragrunt.hcl similarity index 53% rename from deployment/modules/tailscale/account/terragrunt.hcl rename to deployment/modules/netbird/cluster/terragrunt.hcl index 8788a01..a3f72cd 100644 --- a/deployment/modules/tailscale/account/terragrunt.hcl +++ b/deployment/modules/netbird/cluster/terragrunt.hcl @@ -6,7 +6,26 @@ terraform { } } -# Tailnet-global; state key intentionally not parameterised by env. +locals { + env = get_env("TF_VAR_env") + stage = get_env("TF_VAR_stage") +} + +# Per-env: the route advertises this env's vRack CIDR, sourced from the ovh module. +dependency "ovh" { + config_path = "../../ovh/account" + + mock_outputs = { + private_network_cidr = "10.150.200.0/24" + } + mock_outputs_allowed_terraform_commands = ["init", "validate", "plan"] + mock_outputs_merge_strategy_with_state = "shallow" +} + +inputs = { + private_network_cidr = dependency.ovh.outputs.private_network_cidr +} + generate "backend" { path = "backend.tf" if_exists = "overwrite_terragrunt" @@ -14,7 +33,7 @@ generate "backend" { terraform { backend "s3" { bucket = "${get_env("TF_VAR_tf_state_s3_bucket")}" - key = "yucca/o11y/v3/tailscale/account/global" + key = "yucca/o11y/v3/netbird/cluster/${local.env}${local.stage != "" ? "/${local.stage}" : ""}" region = "${get_env("TF_VAR_tf_state_s3_region")}" access_key = "${get_env("TF_VAR_tf_state_s3_access_key")}" secret_key = "${get_env("TF_VAR_tf_state_s3_secret_key")}" diff --git a/deployment/modules/netbird/cluster/variables.tf b/deployment/modules/netbird/cluster/variables.tf new file mode 100644 index 0000000..73033fa --- /dev/null +++ b/deployment/modules/netbird/cluster/variables.tf @@ -0,0 +1,11 @@ +variable "netbird_tf_pat" { + sensitive = true +} + +variable "env" {} + +# Matches the env's private_network_cidr from the ovh module — the vRack subnet +# the Talos nodes advertise as a NetBird network route for operator access. +variable "private_network_cidr" { + type = string +} diff --git a/deployment/modules/ovh/account/variables.tf b/deployment/modules/ovh/account/variables.tf index 3bca2df..d1231b3 100644 --- a/deployment/modules/ovh/account/variables.tf +++ b/deployment/modules/ovh/account/variables.tf @@ -42,28 +42,28 @@ variable "vrack_name" { variable "talos_version" { type = string - default = "v1.13.0" + default = "v1.13.5" } -# Control-plane (Public Cloud / KVM) schematic: tailscale + qemu-guest-agent. +# Control-plane (Public Cloud / KVM) schematic: qemu-guest-agent + netbird. variable "talos_schematic_id" { type = string - default = "7d4c31cbd96db9f90c874990697c523482b2bae27fb4631d5583dcd9c281b1ff" + default = "bbfcb7053b1609712a977830952455432825890922cb6bac23cea34b980970f1" } -# Worker (bare-metal) schematic: tailscale only. qemu-guest-agent must NOT be +# Worker (bare-metal) schematic: netbird only. qemu-guest-agent must NOT be # present on bare metal — it blocks on a virtio port that never appears, which # wedges the Talos boot sequence and reboots the node in a loop. variable "talos_worker_schematic_id" { type = string - default = "4a0d65c669d46663f377e7161e50cfd570c401f26fd9e7bda34a0216b6f1922b" + default = "7326f0cbca7a0e700ac1efa3f32e88df9ebe5010e6e842a8ed36fdc99ee98ead" } # Image must be pre-uploaded out-of-band (talos:dl:cp + talos:ul:cp mise tasks) — # the OVH provider doesn't upload custom images. variable "talos_public_cloud_image_name" { type = string - default = "talos-1.13.0-tailscale-qemu" + default = "talos-1.13.5-qemu-netbird" } # IPLB tier and geographic zone (public-IP location). The LB reaches the workers diff --git a/deployment/modules/ovh/account/workers.tf b/deployment/modules/ovh/account/workers.tf index 6f31c5f..ec220c8 100644 --- a/deployment/modules/ovh/account/workers.tf +++ b/deployment/modules/ovh/account/workers.tf @@ -61,7 +61,7 @@ resource "ovh_dedicated_server" "worker" { customizations = { efi_bootloader_path = "\\EFI\\BOOT\\BOOTX64.EFI" # OVH fetches this raw straight from the Talos Factory at order time (no - # OVH-side upload). The URL resolves the Tailscale-only worker schematic; + # OVH-side upload). The URL resolves the netbird-only worker schematic; # qemu-guest-agent here would reboot-loop the bare-metal node. image_url = replace(data.talos_image_factory_urls.metal.urls.iso, ".iso", ".raw") image_type = "raw" diff --git a/deployment/modules/tailscale/account/.terraform.lock.hcl b/deployment/modules/tailscale/account/.terraform.lock.hcl deleted file mode 100644 index 7241c56..0000000 --- a/deployment/modules/tailscale/account/.terraform.lock.hcl +++ /dev/null @@ -1,35 +0,0 @@ -# This file is maintained automatically by "tofu init". -# Manual edits may be lost in future updates. - -provider "registry.opentofu.org/tailscale/tailscale" { - version = "0.29.2" - constraints = "0.29.2" - hashes = [ - "h1:7D3VzQoUKr4NYJ7ZMMRdV9iQUK8d6yJBrnBhi0DtBfA=", - "h1:910l+uQ0y8nSWPo/CsGI0Ni+lM6oc/h6yckucLmivAY=", - "h1:IPFMdH5vsXeNRj8H/Y6Z6iq3Fko8zkmgcROSXeRJ7MM=", - "h1:KG/OMAOors/W0AZlHkcUD9aeGBIWDCloKa652+JlATM=", - "h1:SSZ93MdSAaJ1Xi/VIvZDz5z1sve3BIS+WqDKACvJut0=", - "h1:Vzj5bDkG9nOQbRMKHPhsol0+BrvzZ22lwkbLSIs4ycs=", - "h1:Xjuo1Cwe065i1qJfhKm3dti4eeEwIT9rNNqU3R6id0A=", - "h1:eADNOR3ZnirZXCP+3k0hy9CKQK8sgeVC0lpo08UntZY=", - "h1:gzBWWbJc4JOwQEIINtnYkbwErRkA2oLvhtRW4lAQ7UU=", - "h1:ipdqf/NJpSaP7em0/+n3hu0KVzwai3cIn2VgP4Rxdhs=", - "h1:lNitoP/DTekzHnRjZ3RBLthPRZ3aHy6lR7jo2rtoH+0=", - "h1:laqGsqlHY2/R1JQnhniKwcT0U8gb+tF0K9vr+Tovskc=", - "h1:wUSb/6AFeFi905uIodIMA11SyyboCTtDqqNhJYdHweM=", - "zh:32b453302a684584198a03c2e09d99ef1f6deac2fe26f8dc134d973b07fd0c23", - "zh:3920bb891a476f30e29248533f667a507c97e93e0a2af010b242e980d6411ca6", - "zh:630cceb40d8806945ad36e04517f313520ea5cee17bd36e26a8e999567835e1a", - "zh:6816de7f6bd3cfe341451af3be95b0af17c5539733b165f7505e88de6b730fc5", - "zh:729d75ab50efb675716ffa609358f6d9e80c66f7f64e01e11c8926c5293385be", - "zh:7d1024f621fe02b3731dc65ab20105150762b29df46e398da2c73d007bb62fb2", - "zh:81cf9cd23a70e5196b9d774482259c17e08471556932c348d3e9df0f7a482af1", - "zh:8de75b9c4b89ed1affcdb3ce32e00afaf213cfd76e93181ea4758ffd10b6eedc", - "zh:8f0b9b175ceb124c1cf01e6bbb660bf1bce0ff77416e026dc0679ed7156bc737", - "zh:8fb4cde10eb346ef9e2233591d75c39bcc2cafee578b669196192e8acdb39f12", - "zh:96798e58b8fcddc2add2da0fe8f1df9f913153d4cbccd68a7557b64633f5ea00", - "zh:a598bd615f4465b828f3b6ed2db694d8148c9469d5497922c7bc17c08b655cf4", - "zh:b2604d5067f3f259ff9be9bb0c4f9ec801d27c905cbf6ceafee50fc2a23f8331", - ] -} diff --git a/deployment/modules/tailscale/account/config.tf b/deployment/modules/tailscale/account/config.tf deleted file mode 100644 index 36accef..0000000 --- a/deployment/modules/tailscale/account/config.tf +++ /dev/null @@ -1,10 +0,0 @@ -terraform { - required_version = "~> 1.10" - - required_providers { - tailscale = { - source = "tailscale/tailscale" - version = "0.29.2" - } - } -} diff --git a/deployment/modules/tailscale/account/providers.tf b/deployment/modules/tailscale/account/providers.tf deleted file mode 100644 index ba09f01..0000000 --- a/deployment/modules/tailscale/account/providers.tf +++ /dev/null @@ -1,6 +0,0 @@ -provider "tailscale" { - # OAuth client from the o11y_tf vault. Must hold write on the tailnet policy file. - oauth_client_id = var.tailscale_oauth_client_id - oauth_client_secret = var.tailscale_oauth_client_secret - tailnet = var.tailscale_tailnet_id -} diff --git a/deployment/modules/tailscale/account/tailscale.tf b/deployment/modules/tailscale/account/tailscale.tf deleted file mode 100644 index 417b28b..0000000 --- a/deployment/modules/tailscale/account/tailscale.tf +++ /dev/null @@ -1,59 +0,0 @@ -# Tailnet-global. Declares all envs in one document so applying for one env -# doesn't wipe another env's tags. autoApprovers pre-approves each env's CP -# subnet route so kubectl/talosctl over Tailscale works without manual -# approval. -resource "tailscale_acl" "this" { - overwrite_existing_content = true - - acl = jsonencode({ - tagOwners = merge( - { - "tag:management" = [] - "tag:project-yucca" = ["autogroup:admin"] - }, - { - for env in keys(var.subnet_routes_by_env) : - "tag:env-${env}" => ["autogroup:admin"] - }, - ) - - autoApprovers = { - routes = { - for env, cidr in var.subnet_routes_by_env : - cidr => ["tag:env-${env}"] - } - } - - grants = [ - { - src = ["*"] - dst = ["*"] - ip = ["*"] - } - ] - - ssh = [ - { - action = "check" - src = ["autogroup:member"] - dst = ["autogroup:self"] - users = ["autogroup:nonroot", "root"] - }, - { - action = "accept" - src = ["autogroup:admin"] - dst = ["tag:management"] - users = ["autogroup:nonroot"] - } - ] - }) -} - -resource "tailscale_tailnet_settings" "org" { - devices_approval_on = true - devices_auto_updates_on = true - devices_key_duration_days = 5 - users_approval_on = true - users_role_allowed_to_join_external_tailnet = "member" - https_enabled = true -} diff --git a/deployment/modules/tailscale/account/variables.tf b/deployment/modules/tailscale/account/variables.tf deleted file mode 100644 index 9ffc777..0000000 --- a/deployment/modules/tailscale/account/variables.tf +++ /dev/null @@ -1,20 +0,0 @@ -variable "tailscale_oauth_client_id" { - sensitive = true -} -variable "tailscale_oauth_client_secret" { - sensitive = true -} -variable "tailscale_tailnet_id" { - sensitive = true -} - -# CIDRs must match each env's `private_network_cidr` in -# deployment/modules/ovh/account/terragrunt.hcl. -variable "subnet_routes_by_env" { - type = map(string) - default = { - development = "10.150.50.0/24" - staging = "10.150.200.0/24" - production = "10.150.100.0/24" - } -} diff --git a/deployment/modules/talos/cluster/.terraform.lock.hcl b/deployment/modules/talos/cluster/.terraform.lock.hcl index daf0653..df83ddc 100644 --- a/deployment/modules/talos/cluster/.terraform.lock.hcl +++ b/deployment/modules/talos/cluster/.terraform.lock.hcl @@ -22,36 +22,3 @@ provider "registry.opentofu.org/siderolabs/talos" { "zh:d218bab0f67a2a8b15add9b51df3d30f514b57e9a7c1d733ebe97966ea132acb", ] } - -provider "registry.opentofu.org/tailscale/tailscale" { - version = "0.29.2" - constraints = "0.29.2" - hashes = [ - "h1:7D3VzQoUKr4NYJ7ZMMRdV9iQUK8d6yJBrnBhi0DtBfA=", - "h1:910l+uQ0y8nSWPo/CsGI0Ni+lM6oc/h6yckucLmivAY=", - "h1:IPFMdH5vsXeNRj8H/Y6Z6iq3Fko8zkmgcROSXeRJ7MM=", - "h1:KG/OMAOors/W0AZlHkcUD9aeGBIWDCloKa652+JlATM=", - "h1:SSZ93MdSAaJ1Xi/VIvZDz5z1sve3BIS+WqDKACvJut0=", - "h1:Vzj5bDkG9nOQbRMKHPhsol0+BrvzZ22lwkbLSIs4ycs=", - "h1:Xjuo1Cwe065i1qJfhKm3dti4eeEwIT9rNNqU3R6id0A=", - "h1:eADNOR3ZnirZXCP+3k0hy9CKQK8sgeVC0lpo08UntZY=", - "h1:gzBWWbJc4JOwQEIINtnYkbwErRkA2oLvhtRW4lAQ7UU=", - "h1:ipdqf/NJpSaP7em0/+n3hu0KVzwai3cIn2VgP4Rxdhs=", - "h1:lNitoP/DTekzHnRjZ3RBLthPRZ3aHy6lR7jo2rtoH+0=", - "h1:laqGsqlHY2/R1JQnhniKwcT0U8gb+tF0K9vr+Tovskc=", - "h1:wUSb/6AFeFi905uIodIMA11SyyboCTtDqqNhJYdHweM=", - "zh:32b453302a684584198a03c2e09d99ef1f6deac2fe26f8dc134d973b07fd0c23", - "zh:3920bb891a476f30e29248533f667a507c97e93e0a2af010b242e980d6411ca6", - "zh:630cceb40d8806945ad36e04517f313520ea5cee17bd36e26a8e999567835e1a", - "zh:6816de7f6bd3cfe341451af3be95b0af17c5539733b165f7505e88de6b730fc5", - "zh:729d75ab50efb675716ffa609358f6d9e80c66f7f64e01e11c8926c5293385be", - "zh:7d1024f621fe02b3731dc65ab20105150762b29df46e398da2c73d007bb62fb2", - "zh:81cf9cd23a70e5196b9d774482259c17e08471556932c348d3e9df0f7a482af1", - "zh:8de75b9c4b89ed1affcdb3ce32e00afaf213cfd76e93181ea4758ffd10b6eedc", - "zh:8f0b9b175ceb124c1cf01e6bbb660bf1bce0ff77416e026dc0679ed7156bc737", - "zh:8fb4cde10eb346ef9e2233591d75c39bcc2cafee578b669196192e8acdb39f12", - "zh:96798e58b8fcddc2add2da0fe8f1df9f913153d4cbccd68a7557b64633f5ea00", - "zh:a598bd615f4465b828f3b6ed2db694d8148c9469d5497922c7bc17c08b655cf4", - "zh:b2604d5067f3f259ff9be9bb0c4f9ec801d27c905cbf6ceafee50fc2a23f8331", - ] -} diff --git a/deployment/modules/talos/cluster/config.tf b/deployment/modules/talos/cluster/config.tf index e0614b1..bcde5c1 100644 --- a/deployment/modules/talos/cluster/config.tf +++ b/deployment/modules/talos/cluster/config.tf @@ -6,9 +6,5 @@ terraform { source = "siderolabs/talos" version = "0.11.0" } - tailscale = { - source = "tailscale/tailscale" - version = "0.29.2" - } } } diff --git a/deployment/modules/talos/cluster/controlplane.tf b/deployment/modules/talos/cluster/controlplane.tf index c0aacfb..464faae 100644 --- a/deployment/modules/talos/cluster/controlplane.tf +++ b/deployment/modules/talos/cluster/controlplane.tf @@ -1,27 +1,3 @@ -resource "tailscale_tailnet_key" "controlplane" { - for_each = var.controlplane_nodes - - reusable = true - ephemeral = true - preauthorized = true - recreate_if_invalid = "always" - expiry = 7776000 - description = "Talos key ${each.value.name}" - tags = [ - "tag:project-yucca", - "tag:env-${var.env}", - ] -} - -data "tailscale_device" "controlplane" { - for_each = var.controlplane_nodes - - hostname = each.value.name - wait_for = "300s" - - depends_on = [talos_machine_bootstrap.this] -} - data "talos_machine_configuration" "controlplane" { cluster_name = local.cluster_name cluster_endpoint = local.cluster_endpoint @@ -63,8 +39,8 @@ resource "talos_machine_configuration_apply" "controlplane" { } ] # Pin the kubelet's node IP to the vRack subnet so the Kubernetes - # InternalIP is always the private IP and never falls back to the - # Tailscale CGNAT address (which happens if eth1 has no IP at kubelet + # InternalIP is always the private IP and never falls back to a mesh + # overlay address (which happens if eth1 has no IP at kubelet # start — see the static address below). kubelet = { nodeIP = { @@ -138,7 +114,7 @@ resource "talos_machine_configuration_apply" "controlplane" { } apiServer = { # VIP for in-cluster traffic; CP private IPs for operators reaching the - # apiserver over Tailscale (the floating VIP doesn't ARP reliably across DCs). + # apiserver over the mesh (the floating VIP doesn't ARP reliably across DCs). certSANs = concat( [local.controlplane_vip], [for k in local.controlplane_keys : var.controlplane_nodes[k].private_ip], @@ -155,19 +131,21 @@ resource "talos_machine_configuration_apply" "controlplane" { hostname: ${each.value.name} EOT , + # Netbird overlay — the node management mesh. Operators reach the vRack IPs + # via the Netbird route (server-side, masqueraded to a routing-peer's vRack + # IP). NB_MANAGEMENT_URL is set explicitly to mirror yucca (Cloud default). <<-EOT - name: tailscale + name: netbird apiVersion: v1alpha1 kind: ExtensionServiceConfig environment: - - TS_AUTHKEY=${tailscale_tailnet_key.controlplane[each.key].key} - - TS_HOSTNAME=${each.value.name} - - TS_ROUTES=${var.private_network_cidr} - - TS_EXTRA_ARGS=--accept-dns=false + - NB_SETUP_KEY=${var.netbird_setup_key} + - NB_MANAGEMENT_URL=https://api.netbird.io EOT , - # Talos ingress firewall — default-deny on host-bound services. Allow - # rules use Tailscale CGNAT (operators) and the vRack CIDR (intra-cluster). + # Talos ingress firewall — default-deny on host-bound services. Operators reach + # apid/apiserver via the Netbird route, masqueraded to a routing-peer's vRack IP, + # so every allow rule is the vRack CIDR (+ pod CIDR for in-cluster metrics). <<-EOT apiVersion: v1alpha1 kind: NetworkDefaultActionConfig @@ -183,8 +161,6 @@ resource "talos_machine_configuration_apply" "controlplane" { - 50000 protocol: tcp ingress: - - subnet: 100.64.0.0/10 - - subnet: fd7a:115c:a1e0::/48 - subnet: ${var.private_network_cidr} EOT , @@ -209,8 +185,6 @@ resource "talos_machine_configuration_apply" "controlplane" { - 6443 protocol: tcp ingress: - - subnet: 100.64.0.0/10 - - subnet: fd7a:115c:a1e0::/48 - subnet: ${var.private_network_cidr} EOT , diff --git a/deployment/modules/talos/cluster/outputs.tf b/deployment/modules/talos/cluster/outputs.tf index 20e7fa6..16ea1e6 100644 --- a/deployment/modules/talos/cluster/outputs.tf +++ b/deployment/modules/talos/cluster/outputs.tf @@ -11,13 +11,6 @@ output "cluster" { } } -output "controlplane_tailscale_ips" { - value = { - for k, _ in var.controlplane_nodes : - k => data.tailscale_device.controlplane[k].addresses[0] - } -} - output "talos_client_configuration" { sensitive = true value = data.talos_client_configuration.this.talos_config diff --git a/deployment/modules/talos/cluster/providers.tf b/deployment/modules/talos/cluster/providers.tf deleted file mode 100644 index 7c8598e..0000000 --- a/deployment/modules/talos/cluster/providers.tf +++ /dev/null @@ -1,7 +0,0 @@ -provider "tailscale" { - # OAuth client from the o11y_tf vault. Must hold write on auth keys + read on - # devices, and own the tag:project-yucca / tag:env-* tags it issues keys with. - oauth_client_id = var.tailscale_oauth_client_id - oauth_client_secret = var.tailscale_oauth_client_secret - tailnet = var.tailscale_tailnet_id -} diff --git a/deployment/modules/talos/cluster/terragrunt.hcl b/deployment/modules/talos/cluster/terragrunt.hcl index 5fad70f..7cb73cd 100644 --- a/deployment/modules/talos/cluster/terragrunt.hcl +++ b/deployment/modules/talos/cluster/terragrunt.hcl @@ -63,11 +63,12 @@ dependency "ovh" { mock_outputs_merge_strategy_with_state = "shallow" } -dependency "tailscale" { - config_path = "../../tailscale/account" +dependency "netbird_cluster" { + config_path = "../../netbird/cluster" mock_outputs = { - tailscale_output = "mock-tailscale-output" + talos_setup_key = "mock-netbird-setup-key" } + mock_outputs_allowed_terraform_commands = ["init", "validate", "plan"] } inputs = { @@ -78,6 +79,7 @@ inputs = { worker_data_disk_match = local.worker_data_disk_match worker_data_disk2_match = local.worker_data_disk2_match worker_nics = local.worker_nics + netbird_setup_key = dependency.netbird_cluster.outputs.talos_setup_key } generate "backend" { diff --git a/deployment/modules/talos/cluster/variables.tf b/deployment/modules/talos/cluster/variables.tf index 8fbe3aa..50b2935 100644 --- a/deployment/modules/talos/cluster/variables.tf +++ b/deployment/modules/talos/cluster/variables.tf @@ -1,13 +1,10 @@ variable "env" {} variable "stage" {} -variable "tailscale_oauth_client_id" { - sensitive = true -} -variable "tailscale_oauth_client_secret" { - sensitive = true -} -variable "tailscale_tailnet_id" { +# Netbird setup key from the netbird/cluster module (terragrunt dependency). Fed to +# every node's netbird ExtensionServiceConfig (NB_SETUP_KEY). Takes effect once the +# node runs a schematic that includes siderolabs/netbird. +variable "netbird_setup_key" { sensitive = true } @@ -43,7 +40,7 @@ variable "talos_installer_images" { variable "talos_version" { type = string - default = "v1.13.0" + default = "v1.13.5" } variable "controlplane_vip_offset" { @@ -87,9 +84,9 @@ variable "worker_nics" { })) } -# True only during initial bring-up of a brand-new env, before the Tailscale +# True only during initial bring-up of a brand-new env, before the Netbird # extension has registered any node. Drop back to false once each node is on -# the tailnet, so future applies go via the vRack and the ingress firewall +# the netbird mesh, so future applies go via the vRack and the ingress firewall # can drop public-NIC traffic without locking terraform out. variable "use_public_endpoints" { type = bool diff --git a/deployment/modules/talos/cluster/workers.tf b/deployment/modules/talos/cluster/workers.tf index 1e988cb..58b0d64 100644 --- a/deployment/modules/talos/cluster/workers.tf +++ b/deployment/modules/talos/cluster/workers.tf @@ -1,18 +1,3 @@ -resource "tailscale_tailnet_key" "worker" { - for_each = var.worker_nodes - - reusable = true - ephemeral = true - preauthorized = true - recreate_if_invalid = "always" - expiry = 7776000 - description = "Talos key ${each.value.name}" - tags = [ - "tag:project-yucca", - "tag:env-${var.env}", - ] -} - data "talos_machine_configuration" "worker" { cluster_name = local.cluster_name cluster_endpoint = local.cluster_endpoint @@ -80,16 +65,16 @@ resource "talos_machine_configuration_apply" "worker" { hostname: ${each.value.name} EOT , - # Tailscale extension is baked into the worker image's schematic; without - # this config block the service starts unauthenticated and hangs. + # Netbird overlay — the node management mesh. Operators reach the vRack IPs + # via the Netbird route (server-side, masqueraded to a routing-peer's vRack + # IP). NB_MANAGEMENT_URL is set explicitly to mirror yucca (Cloud default). <<-EOT - name: tailscale + name: netbird apiVersion: v1alpha1 kind: ExtensionServiceConfig environment: - - TS_AUTHKEY=${tailscale_tailnet_key.worker[each.key].key} - - TS_HOSTNAME=${each.value.name} - - TS_EXTRA_ARGS=--accept-dns=false + - NB_SETUP_KEY=${var.netbird_setup_key} + - NB_MANAGEMENT_URL=https://api.netbird.io EOT , <<-EOT @@ -144,8 +129,6 @@ resource "talos_machine_configuration_apply" "worker" { - 50000 protocol: tcp ingress: - - subnet: 100.64.0.0/10 - - subnet: fd7a:115c:a1e0::/48 - subnet: ${var.private_network_cidr} EOT , @@ -235,7 +218,7 @@ resource "talos_machine_configuration_apply" "worker" { - subnet: ${var.private_network_cidr} - subnet: 10.244.0.0/16 EOT - ], var.worker_data_disk2_match == "" ? [] : [ + ], var.worker_data_disk2_match == "" ? [] : [ # Second spare NVMe — production only (data2_match = "" renders nothing). <<-EOT apiVersion: v1alpha1 diff --git a/docs/01-bootstrap-guide.md b/docs/01-bootstrap-guide.md index 20a0e8d..37702b7 100644 --- a/docs/01-bootstrap-guide.md +++ b/docs/01-bootstrap-guide.md @@ -12,8 +12,8 @@ How to stand up an environment from nothing. The cluster is built by Terragrunt mise run talos:dl:cp && mise run talos:ul:cp ``` - This image carries the `qemu-guest-agent` + `tailscale` schematic. -4. Workers need **no download or upload** — they are OVH BYOI and pull the bare-metal raw straight from the Talos Factory at order time. The worker schematic must stay **tailscale-only**: `qemu-guest-agent` on bare metal blocks on a virtio port that never appears and reboot-loops the node. + This image carries the `qemu-guest-agent` + `netbird` schematic. +4. Workers need **no download or upload** — they are OVH BYOI and pull the bare-metal raw straight from the Talos Factory at order time. The worker schematic must stay **netbird-only**: `qemu-guest-agent` on bare metal blocks on a virtio port that never appears and reboot-loops the node. 5. For production: delete the apex DNS records via the OVH dashboard before applying. ## Apply order @@ -31,19 +31,19 @@ export TF_VAR_env=staging mise run tg run --working-dir deployment/modules/ovh/account apply ``` -2. **Tailscale** — tailnet-global ACL (only needs one run across all environments). +2. **NetBird** — the per-environment mesh objects: the Talos node group, a reusable setup key, the vRack network route (Talos nodes as routing peers), and the `yucca → resource` access policy. The Talos module consumes the setup key from here, so apply NetBird first. ```bash - mise run tg run --working-dir deployment/modules/tailscale/account apply + mise run tg run --working-dir deployment/modules/netbird/cluster apply ``` -3. **Talos (bootstrap)** — initial bring-up over public IPs, because the Tailscale extension isn't running yet. +3. **Talos (bootstrap)** — initial bring-up over public IPs, because the NetBird extension isn't running yet. ```bash TF_VAR_use_public_endpoints=true mise run tg run --working-dir deployment/modules/talos/cluster apply ``` -4. **Verify** the cluster is up and operator-side Tailscale routing works. Pull the configs (see [Cluster access](#cluster-access)) and hit the APIs over the tailnet: +4. **Verify** the cluster is up and operator-side NetBird routing works. Pull the configs (see [Cluster access](#cluster-access)) and hit the APIs over the NetBird network: ```bash mise run talos:kubeconfig && mise run talos:talosconfig @@ -51,7 +51,7 @@ export TF_VAR_env=staging talosctl --talosconfig .private/$ENVIRONMENT/talosconfig -n 10.150.200.10 get members ``` -5. **Talos (steady state)** — drop the public-endpoints override now that Tailscale routes work; the host firewall closes the public NIC (everything except `:30443` on workers). +5. **Talos (steady state)** — drop the public-endpoints override now that NetBird routes work; the host firewall closes the public NIC (everything except `:30443` on workers). ```bash unset TF_VAR_use_public_endpoints @@ -72,7 +72,7 @@ How to get `kubectl` / `talosctl` access to an **existing** cluster (no bootstra **Prerequisites:** -* **Tailscale** — the cluster APIs are reachable only over the tailnet, so your host needs Tailscale running with subnet-route consumption enabled: `tailscale set --accept-routes` on Linux, or the "Use Tailscale subnets" toggle in the macOS app. +* **NetBird** — the cluster APIs are reachable only over the NetBird network, so your host must be running the NetBird client (`netbird up`) and joined to the FUTO NetBird account, which places your peer in the `yucca` group. The access policy then distributes the route to the cluster's vRack CIDR, so `kubectl`/`talosctl` can reach the nodes' private IPs. * **1Password CLI (`op`)** — installed and signed in to the `team-futo.1password.com` account. `mise run talos:config` fetches the configs through `mise run tg`, which wraps `op run` to inject the Terraform state credentials; without an authenticated `op` it can't read state. **Fetch the configs.** Two tasks pull `kubeconfig` and `talosconfig` for the environment (run whichever you need): @@ -84,9 +84,9 @@ mise run talos:kubeconfig # for kubectl mise run talos:talosconfig # for talosctl ``` -Each writes to `.private/$ENVIRONMENT/` (mode 600) from the Talos module's Terraform outputs. `talos:kubeconfig` also repoints the kubeconfig `server:` from the floating VIP (`10.150.200.5`) to a control-plane private IP (`10.150.200.10`) — the VIP doesn't ARP reliably across DCs over Tailscale, and every CP IP is in the apiserver cert SANs so TLS still validates. +Each writes to `.private/$ENVIRONMENT/` (mode 600) from the Talos module's Terraform outputs. `talos:kubeconfig` also repoints the kubeconfig `server:` from the floating VIP (`10.150.200.5`) to a control-plane private IP (`10.150.200.10`) — the VIP doesn't ARP reliably across DCs over the NetBird network, and every CP IP is in the apiserver cert SANs so TLS still validates. -> **A highly-available operator API endpoint is TBD.** `kubectl` is pinned to a single control-plane IP, so if that CP is down you currently repoint to another by hand (any CP IP works — they're all cert SANs). The floating VIP is HA *inside* the cluster (kubelet and in-cluster clients use it) but doesn't ARP across DCs over Tailscale, so there's no HA endpoint for operators yet. +> **A highly-available operator API endpoint is TBD.** `kubectl` is pinned to a single control-plane IP, so if that CP is down you currently repoint to another by hand (any CP IP works — they're all cert SANs). The floating VIP is HA *inside* the cluster (kubelet and in-cluster clients use it) but doesn't ARP across DCs over the NetBird network, so there's no HA endpoint for operators yet. **Point your tools at them:** diff --git a/docs/02-infrastructure-architecture-guide.md b/docs/02-infrastructure-architecture-guide.md index d4d2cd2..62f1ee4 100644 --- a/docs/02-infrastructure-architecture-guide.md +++ b/docs/02-infrastructure-architecture-guide.md @@ -2,7 +2,7 @@ The OVH foundation under the cluster: compute, private network, and public ingress. Each environment is a fully independent build of the same shape; they differ only in worker tier and IPLB zone count. -> **Status:** staging is built and running (`o11y-staging`). Production is planned (`o11y-production`) — same shape, larger workers, multi-zone ingress. +> **Status:** both environments are built and running — `o11y-staging` and `o11y-production`. Same shape; production has larger workers and multi-zone ingress. ## Shape @@ -42,11 +42,11 @@ The worker host firewall scopes `:30443` to OVH's IPLB NAT range (`10.108.0.0/14 Because the farm targets the workers' public IPs (not the vRack), three things are required and are handled in the cluster config: NodePorts must answer on the public NIC, exactly one Envoy must run per worker, and Envoy must parse PROXY protocol. See the cluster architecture guide for those details. -## Operator access (Tailscale) +## Operator access (NetBird) -Tailscale runs as a Talos system extension on **every** node, so operators reach `talosctl` and `kubectl` over the tailnet without exposing those APIs publicly. Control planes advertise the private CIDR as a subnet route, auto-approved by the tailnet ACL; workers consume the routes. The ACL is environment-scoped (`tag:env-staging` vs `tag:env-production`) so staging operators can't pivot into production. +NetBird runs as a Talos system extension on **every** node, so operators reach `talosctl` and `kubectl` over the NetBird network without exposing those APIs publicly. The vRack subnet is published as a NetBird network route with the Talos nodes as routing peers — any node can route, so it's HA — and operator traffic is masqueraded to the routing peer's vRack IP, which the host firewall already trusts. A per-environment access policy lets the shared `yucca` operator group reach this environment's routed subnet on the management ports only (apid `50000`, kube-apiserver `6443`); the groups and policy are environment-scoped (`O11Y_STAGING_*` vs `O11Y_PRODUCTION_*`), so staging operators can't pivot into production. -Operators point `kubectl`/`talosctl` at a specific control plane's static private IP — not the floating VIP, since cross-DC ARP for the VIP over Tailscale subnet routes is unreliable. The VIP remains the in-cluster apiserver endpoint used by kubelet and other in-cluster components. +Operators point `kubectl`/`talosctl` at a specific control plane's static private IP — not the floating VIP, since cross-DC ARP for the VIP over the NetBird network route is unreliable. The VIP remains the in-cluster apiserver endpoint used by kubelet and other in-cluster components. ## Cost @@ -69,5 +69,5 @@ Staging + production run-rate ≈ **$955/mo** plus the one-time **$221** product | Workers | 3× `SYS-2` (`24sys022`) | 3× `Rise-2` (`24rise02-v1`) | | IPLB | 1 zone (`gra`) | 3 zones (`gra` + `rbx` + `sbg`), anycast | | Private CIDR | `10.150.200.0/24` | `10.150.100.0/24` | -| Tailscale tag | `tag:env-staging` | `tag:env-production` | +| NetBird objects | `O11Y_STAGING_*` | `O11Y_PRODUCTION_*` | | Flux source | `staging` overlay | `production` overlay | diff --git a/docs/03-cluster-architecture-guide.md b/docs/03-cluster-architecture-guide.md index 43eba35..fac92d4 100644 --- a/docs/03-cluster-architecture-guide.md +++ b/docs/03-cluster-architecture-guide.md @@ -12,8 +12,8 @@ Control planes and workers use **different** Talos Factory schematics, on purpos | Node type | Platform | Schematic | |-----------|----------|-----------| -| Control plane | OVH Public Cloud (KVM) | `tailscale` + `qemu-guest-agent` | -| Worker | Bare metal | `tailscale` only | +| Control plane | OVH Public Cloud (KVM) | `netbird` + `qemu-guest-agent` | +| Worker | Bare metal | `netbird` only | `qemu-guest-agent` on bare metal wedges boot — it waits on a virtio-serial port that isn't present and reboot-loops the node. The control-plane image is an OpenStack image uploaded to OVH glance once; workers are BYOI and fetch their raw image from the Factory at order time. @@ -36,9 +36,9 @@ Default-deny ingress on every node; anything not listed is dropped at the host. | Service | Port(s) | Allowed sources | |---------|---------|-----------------| -| apid | 50000/tcp | Tailscale (`100.64.0.0/10`, `fd7a:115c:a1e0::/48`), vRack | -| trustd | 50001/tcp | Tailscale, vRack | -| kube-apiserver (CPs) | 6443/tcp | Tailscale, vRack | +| apid | 50000/tcp | vRack | +| trustd | 50001/tcp | vRack | +| kube-apiserver (CPs) | 6443/tcp | vRack | | etcd (CPs) | 2379–2380/tcp | vRack | | kubelet | 10250/tcp | vRack + pod CIDR `10.244.0.0/16` | | flannel VXLAN | 4789/udp | vRack | @@ -49,6 +49,8 @@ Default-deny ingress on every node; anything not listed is dropped at the host. Pod CIDR is allowed on `kubelet` and the metrics ports because pod-to-own-node-IP traffic skips flannel masquerade (a same-node scrape keeps its pod-IP source), which the vRack-only rule would otherwise drop. +Operator `talosctl`/`kubectl` traffic needs no rule of its own: it arrives over the NetBird network route masqueraded to a routing peer's vRack IP, so the vRack allow on `apid` and `kube-apiserver` already covers it. + ## Kubernetes Kubernetes with flannel CNI and kube-proxy in nftables mode. Spegel runs as a peer-to-peer image registry mirror so each node's containerd pulls layers from its peers before the upstream registry (this requires `discard_unpacked_layers = false` in the worker containerd config).