feat: migrate tailscale to netbird (#84)

Signed-off-by: Devin Buhl <devin@buhl.casa>
This commit is contained in:
Devin Buhl
2026-07-01 06:31:59 -04:00
committed by GitHub
parent 59456618d9
commit f409d321b7
28 changed files with 246 additions and 299 deletions
+7 -7
View File
@@ -25,8 +25,8 @@ TF_VAR_dist_dir="{{config_root}}/dist"
ENVIRONMENT_SHORT = "{% set e = get_env(name='ENVIRONMENT', default='development') %}{% if e == 'development' or e == '' %}dev{% elif e == 'production' %}prod{% else %}{{ e }}{% endif %}"
# Control plane and workers boot from different Talos Factory schematics:
# control plane (Public Cloud / KVM): tailscale + qemu-guest-agent
# worker (bare metal): tailscale only — qemu-guest-agent wedges
# control plane (Public Cloud / KVM): qemu-guest-agent + netbird
# worker (bare metal): netbird only — qemu-guest-agent wedges
# bare-metal boot and reboot-loops the node
# Control planes need the image uploaded to OVH glance (talos:{dl,ul}:cp); workers
# are OVH BYOI, so OVH fetches their raw from the Factory at order time (no upload).
@@ -34,10 +34,10 @@ ENVIRONMENT_SHORT = "{% set e = get_env(name='ENVIRONMENT', default='development
[tasks."talos:dl:cp"]
run = """
mkdir -p {{config_root}}/.private/dist
wget https://factory.talos.dev/image/7d4c31cbd96db9f90c874990697c523482b2bae27fb4631d5583dcd9c281b1ff/v1.13.0/openstack-amd64.raw.xz -O {{config_root}}/.private/dist/talos.1.13.0-tailscale-qemu.raw.xz
unxz {{config_root}}/.private/dist/talos.1.13.0-tailscale-qemu.raw.xz --force
wget https://factory.talos.dev/image/bbfcb7053b1609712a977830952455432825890922cb6bac23cea34b980970f1/v1.13.5/openstack-amd64.raw.xz -O {{config_root}}/.private/dist/talos.1.13.5-qemu-netbird.raw.xz
unxz {{config_root}}/.private/dist/talos.1.13.5-qemu-netbird.raw.xz --force
"""
description = "Download the control-plane Talos image (OpenStack, tailscale + qemu-guest-agent schematic)"
description = "Download the control-plane Talos image (OpenStack, qemu-guest-agent + netbird schematic)"
dir = "{{cwd}}"
[tasks."talos:ul:cp"]
@@ -45,9 +45,9 @@ run = """
for region in "RBX-A" "GRA9" "EU-WEST-PAR"; do
echo "Uploading to region ${region}..."
source {{config_root}}/.private/openstack/${ENVIRONMENT}/openrc.sh
openstack image create "talos-1.13.0-tailscale-qemu" \
openstack image create "talos-1.13.5-qemu-netbird" \
--os-region "${region}" \
--file {{config_root}}/.private/dist/talos.1.13.0-tailscale-qemu.raw \
--file {{config_root}}/.private/dist/talos.1.13.5-qemu-netbird.raw \
--disk-format raw \
--container-format bare \
--property hw_qemu_guest_agent=yes \
+3 -3
View File
@@ -4,7 +4,7 @@ The centralized observability platform for FUTO services. A single Talos Kuberne
Built for geographic resilience with single-cluster operational simplicity: three control planes in low-latency DCs hold one etcd quorum, three bare-metal workers carry the observability workload, and everything talks over a private OVH vRack.
**Status:** staging is built and running (`o11y-staging`); production is planned (`o11y-production`).
**Status:** both environments are built and running — `o11y-staging` and `o11y-production`.
## Documentation
@@ -20,7 +20,7 @@ Built for geographic resilience with single-cluster operational simplicity: thre
```text
deployment/modules/
├── ovh/account/ # cloud project, vRack, private network, CPs, workers, IPLB, DNS
├── tailscale/account/ # tailnet-global ACL and tailnet settings
├── netbird/cluster/ # per-env mesh: node group, setup key, vRack network route, access policy
├── talos/cluster/ # machine secrets, CP + worker configs, bootstrap, ingress firewall
└── kubernetes/helm/ # Flux Operator + Instance, env-scoped secrets
@@ -34,4 +34,4 @@ kubernetes/
└── cluster-settings.yaml # per-env ConfigMap: APP_DOMAIN, CLUSTER_NAME
```
State lives in S3 under `yucca/o11y/v3/<module>/<env>`. Secrets and OVH/Tailscale tokens come from the environment's 1Password vault via `op run` and `deployment/.env`.
State lives in S3 under `yucca/o11y/v3/<module>/<env>`. Secrets and OVH/NetBird tokens come from 1Password via `op run` and `deployment/.env`.
+3 -4
View File
@@ -10,10 +10,9 @@ export TF_VAR_tf_state_s3_region=op://o11y_tf/TF_STATE_S3_REGION/password
export TF_VAR_tf_state_s3_access_key=op://o11y_tf/TF_STATE_S3_ACCESS_KEY/password
export TF_VAR_tf_state_s3_secret_key=op://o11y_tf/TF_STATE_S3_SECRET_KEY/password
export TF_VAR_tailscale_oauth_client_id=op://o11y_tf/TAILSCALE_OAUTH_CLIENT_ID/password
export TF_VAR_tailscale_oauth_client_secret=op://o11y_tf/TAILSCALE_OAUTH_CLIENT_SECRET/password
export TF_VAR_tailscale_tailnet_id=op://o11y_tf/TAILSCALE_TAILNET_ID/password
export TF_VAR_op_credentials_file=op://o11y_tf/1PASS_CONNECT_SERVER_CREDENTIALS_FILE/password
export TF_VAR_op_connect_token=op://o11y_tf/1PASS_CONNECT_O11Y_SUPERUSER/password
export TF_VAR_op_connect_token_env=op://o11y_tf_${ENVIRONMENT_SHORT}/1PASS_CONNECT_O11Y_READ/password
# Netbird Cloud PAT (shared across envs) — drives the netbird/account TF provider.
export TF_VAR_netbird_tf_pat=op://shared_tf/NETBIRD_TF_PAT/password
+38
View File
@@ -0,0 +1,38 @@
# This file is maintained automatically by "tofu init".
# Manual edits may be lost in future updates.
provider "registry.opentofu.org/netbirdio/netbird" {
version = "0.0.9"
constraints = "0.0.9"
hashes = [
"h1:0Vi0MLMk+K1CEuBZB+ByU55wXx87eGd5j0fGACekCDQ=",
"h1:2ska0C0jDxvbjApKlUmdkfrXQFjHiNf/KKYt355Isgo=",
"h1:H8GG0MZqAAXqnrMCw0Q3vcE3gGFEqYmDF9YY2RnD18Y=",
"h1:HKmwtSPE6k++umLH99WkpT/HcNzy69OWRTVZMsDFJh8=",
"h1:KFbgGf03fTGEgJRIpzAryWB26BHicfLQC8nuE5IiSys=",
"h1:VMUHvZOWIU/IWmK+NyEUIafKS88gGTvff+nK8N7NE3E=",
"h1:aFK3rhEjnnwiHU7Fl/nBDz5Y1b/2sRYXYomu4qewlwM=",
"h1:ba2yqgEM9Ssy+C/0nP3XHh209Y20gpYIErZtYVVb2k8=",
"h1:hC4RZOIVv5KTHcF+AuIHQe01hJmfw7yioSRTDb4ImDg=",
"h1:jaLhsZX1wOdBWKED+hJ0kyfV5TE3Laim6Kd7Uw+krzQ=",
"h1:mWLdkV/wXnK6SSMIKw+BJFJPonrts+7L5IEefRQ0rV8=",
"h1:nvVlkGxQSfphiXMNgGmvuTZQOOjhtdHMe/Upx3/HWjA=",
"h1:pjY52PZ8FJXUbeFcB0kSV8guz4h/xzlXLarWXelAwes=",
"h1:trYvM2c8h5lQ0m8+3Bgvgnbkm04+VOAL2pr6odmFfS8=",
"zh:1f34fba3ecfe0efa36d4b4bddf5714c69124e705d1e371e65abd291887d43385",
"zh:203f671af4d1f5376f4e20fb8d82cc8dc6c4fd136d9abffe8eb7b7f34e27197b",
"zh:21cb344302bdbadbc2779768116c7351d915bb7678a4847e9e2c8623d0032c48",
"zh:2368868e0f86b458f83b6598317ff5fde0d4079500d4567e99f46bf80c63f07a",
"zh:3237a426818eb3906153b1cf7fb6dfe9d52128972e7cab17c324d88788f86863",
"zh:427f8e8ba190d69cfd493b1598b4bf7940fe9a521a333df3540d2e0ea07543cf",
"zh:689cb219f3936500f5a3147299961f8740e505612b5182d93840cb8152d89b4f",
"zh:6e594e69d9e107c45ed4677a9bff68de2dd4232900b636327b24c819bee72c9e",
"zh:715634d6d0b052ac3f34aeb4c2df42960cdf44dd5c45d060867db78716aebb48",
"zh:890df766e9b839623b1f0437355032a3c006226a6c200cd911e15ee1a9014e9f",
"zh:8cbbdde5a0b59f0bb427b403da8993b215d1a6231e0d6b9b4faa89db76b10705",
"zh:af92281c3e14c6af53fb93cee584c5118c5c7bee0acb938b5a367a219d9bc2e2",
"zh:bcdeef5fadf092e33228f28f013af7d8d9553b9d2fb2cecd1894351d5ad37a7a",
"zh:d6c31cb12f18b25c9663c14e737554e06144d4e270e607337bac7963ca3b9542",
"zh:e5873e8e0b29c8cbfd98f04759579da3c0529c8b38608c0f1ea92eb566c8ddc1",
]
}
@@ -0,0 +1,10 @@
terraform {
required_version = "~> 1.10"
required_providers {
netbird = {
source = "netbirdio/netbird"
version = "0.0.9"
}
}
}
@@ -0,0 +1,82 @@
# Per-env o11y NetBird objects (state key .../netbird/cluster/<env>). Everything is
# per-env — resource groups and policies included — so access can differ by env
# (Zack: some people get dev/staging but not prod). No account-wide layer.
#
# All object names are UPPER_SNAKE to match yucca's convention (groups, setup keys,
# networks, network resources, and policies/rules).
# Existing account-wide users group (yucca peers — populated via users' auto_groups,
# per NetBird's model). Referenced as the policy source so the same operators that
# reach yucca also reach o11y — per the team decision.
data "netbird_group" "yucca" {
name = "yucca"
}
# Group the Talos nodes auto-join via the setup key below.
resource "netbird_group" "talos" {
name = "O11Y_${upper(var.env)}_TALOS"
}
# Per-env tag for this cluster's routed resources (the vRack subnet below tags into
# it; the yucca->resource policy grants it). Created with its final name — the
# NetBird provider can't rename a group once network resources are tagged into it.
resource "netbird_group" "o11y_resource" {
name = "O11Y_${upper(var.env)}_RESOURCE"
}
# Reusable, non-ephemeral enrollment key for the Talos nodes — fed to the netbird
# Talos extension as NB_SETUP_KEY. NOT ephemeral: ephemeral peers are reaped after
# 10m idle, which would delete live nodes.
resource "netbird_setup_key" "talos" {
name = "O11Y_${upper(var.env)}_TALOS"
type = "reusable"
ephemeral = false
expiry_seconds = 0 # unlimited — nodes re-enroll with the same key on reprovision
usage_limit = 0 # unlimited
auto_groups = [netbird_group.talos.id]
}
# vRack subnet advertised to operators with the Talos nodes as routing peers — the
# Netbird "Networks" model. Every node sits on the vRack, so any can route (HA);
# masquerade NATs operator traffic to the routing peer's vRack IP, which the node
# firewall already trusts.
resource "netbird_network" "vrack" {
name = "O11Y_${upper(var.env)}_VRACK"
description = "o11y ${var.env} vRack private subnet"
}
resource "netbird_network_router" "vrack" {
network_id = netbird_network.vrack.id
peer_groups = [netbird_group.talos.id]
masquerade = true
metric = 9999
enabled = true
}
resource "netbird_network_resource" "vrack" {
network_id = netbird_network.vrack.id
# Per-env: Netbird network-resource names are account-globally unique.
name = "O11Y_${upper(var.env)}_VRACK_CIDR"
address = var.private_network_cidr
groups = [netbird_group.o11y_resource.id]
enabled = true
}
# NetBird default-denies; allow yucca operators to reach THIS env's routed subnet on
# the Talos management ports only (apid + kube-apiserver). Per-env policy so access
# can later differ by env — tighter than yucca's all-protocol yucca->yucca_resource.
resource "netbird_policy" "yucca_to_o11y_resource" {
name = "O11Y_${upper(var.env)}_YUCCA_TO_RESOURCE"
enabled = true
rule {
name = "YUCCA_TO_O11Y_RESOURCE"
action = "accept"
protocol = "tcp"
enabled = true
bidirectional = false
sources = [data.netbird_group.yucca.id]
destinations = [netbird_group.o11y_resource.id]
ports = ["50000", "6443"]
}
}
@@ -0,0 +1,6 @@
# Plaintext setup key fed to the Talos netbird extension (NB_SETUP_KEY) by the
# talos/cluster module, which consumes this via a terragrunt dependency.
output "talos_setup_key" {
sensitive = true
value = netbird_setup_key.talos.key
}
@@ -0,0 +1,5 @@
provider "netbird" {
# PAT from the shared_tf vault (Netbird Cloud). management_url defaults to
# https://api.netbird.io, so it is left unset.
token = var.netbird_tf_pat
}
@@ -6,7 +6,26 @@ terraform {
}
}
# Tailnet-global; state key intentionally not parameterised by env.
locals {
env = get_env("TF_VAR_env")
stage = get_env("TF_VAR_stage")
}
# Per-env: the route advertises this env's vRack CIDR, sourced from the ovh module.
dependency "ovh" {
config_path = "../../ovh/account"
mock_outputs = {
private_network_cidr = "10.150.200.0/24"
}
mock_outputs_allowed_terraform_commands = ["init", "validate", "plan"]
mock_outputs_merge_strategy_with_state = "shallow"
}
inputs = {
private_network_cidr = dependency.ovh.outputs.private_network_cidr
}
generate "backend" {
path = "backend.tf"
if_exists = "overwrite_terragrunt"
@@ -14,7 +33,7 @@ generate "backend" {
terraform {
backend "s3" {
bucket = "${get_env("TF_VAR_tf_state_s3_bucket")}"
key = "yucca/o11y/v3/tailscale/account/global"
key = "yucca/o11y/v3/netbird/cluster/${local.env}${local.stage != "" ? "/${local.stage}" : ""}"
region = "${get_env("TF_VAR_tf_state_s3_region")}"
access_key = "${get_env("TF_VAR_tf_state_s3_access_key")}"
secret_key = "${get_env("TF_VAR_tf_state_s3_secret_key")}"
@@ -0,0 +1,11 @@
variable "netbird_tf_pat" {
sensitive = true
}
variable "env" {}
# Matches the env's private_network_cidr from the ovh module — the vRack subnet
# the Talos nodes advertise as a NetBird network route for operator access.
variable "private_network_cidr" {
type = string
}
+6 -6
View File
@@ -42,28 +42,28 @@ variable "vrack_name" {
variable "talos_version" {
type = string
default = "v1.13.0"
default = "v1.13.5"
}
# Control-plane (Public Cloud / KVM) schematic: tailscale + qemu-guest-agent.
# Control-plane (Public Cloud / KVM) schematic: qemu-guest-agent + netbird.
variable "talos_schematic_id" {
type = string
default = "7d4c31cbd96db9f90c874990697c523482b2bae27fb4631d5583dcd9c281b1ff"
default = "bbfcb7053b1609712a977830952455432825890922cb6bac23cea34b980970f1"
}
# Worker (bare-metal) schematic: tailscale only. qemu-guest-agent must NOT be
# Worker (bare-metal) schematic: netbird only. qemu-guest-agent must NOT be
# present on bare metal — it blocks on a virtio port that never appears, which
# wedges the Talos boot sequence and reboots the node in a loop.
variable "talos_worker_schematic_id" {
type = string
default = "4a0d65c669d46663f377e7161e50cfd570c401f26fd9e7bda34a0216b6f1922b"
default = "7326f0cbca7a0e700ac1efa3f32e88df9ebe5010e6e842a8ed36fdc99ee98ead"
}
# Image must be pre-uploaded out-of-band (talos:dl:cp + talos:ul:cp mise tasks) —
# the OVH provider doesn't upload custom images.
variable "talos_public_cloud_image_name" {
type = string
default = "talos-1.13.0-tailscale-qemu"
default = "talos-1.13.5-qemu-netbird"
}
# IPLB tier and geographic zone (public-IP location). The LB reaches the workers
+1 -1
View File
@@ -61,7 +61,7 @@ resource "ovh_dedicated_server" "worker" {
customizations = {
efi_bootloader_path = "\\EFI\\BOOT\\BOOTX64.EFI"
# OVH fetches this raw straight from the Talos Factory at order time (no
# OVH-side upload). The URL resolves the Tailscale-only worker schematic;
# OVH-side upload). The URL resolves the netbird-only worker schematic;
# qemu-guest-agent here would reboot-loop the bare-metal node.
image_url = replace(data.talos_image_factory_urls.metal.urls.iso, ".iso", ".raw")
image_type = "raw"
-35
View File
@@ -1,35 +0,0 @@
# This file is maintained automatically by "tofu init".
# Manual edits may be lost in future updates.
provider "registry.opentofu.org/tailscale/tailscale" {
version = "0.29.2"
constraints = "0.29.2"
hashes = [
"h1:7D3VzQoUKr4NYJ7ZMMRdV9iQUK8d6yJBrnBhi0DtBfA=",
"h1:910l+uQ0y8nSWPo/CsGI0Ni+lM6oc/h6yckucLmivAY=",
"h1:IPFMdH5vsXeNRj8H/Y6Z6iq3Fko8zkmgcROSXeRJ7MM=",
"h1:KG/OMAOors/W0AZlHkcUD9aeGBIWDCloKa652+JlATM=",
"h1:SSZ93MdSAaJ1Xi/VIvZDz5z1sve3BIS+WqDKACvJut0=",
"h1:Vzj5bDkG9nOQbRMKHPhsol0+BrvzZ22lwkbLSIs4ycs=",
"h1:Xjuo1Cwe065i1qJfhKm3dti4eeEwIT9rNNqU3R6id0A=",
"h1:eADNOR3ZnirZXCP+3k0hy9CKQK8sgeVC0lpo08UntZY=",
"h1:gzBWWbJc4JOwQEIINtnYkbwErRkA2oLvhtRW4lAQ7UU=",
"h1:ipdqf/NJpSaP7em0/+n3hu0KVzwai3cIn2VgP4Rxdhs=",
"h1:lNitoP/DTekzHnRjZ3RBLthPRZ3aHy6lR7jo2rtoH+0=",
"h1:laqGsqlHY2/R1JQnhniKwcT0U8gb+tF0K9vr+Tovskc=",
"h1:wUSb/6AFeFi905uIodIMA11SyyboCTtDqqNhJYdHweM=",
"zh:32b453302a684584198a03c2e09d99ef1f6deac2fe26f8dc134d973b07fd0c23",
"zh:3920bb891a476f30e29248533f667a507c97e93e0a2af010b242e980d6411ca6",
"zh:630cceb40d8806945ad36e04517f313520ea5cee17bd36e26a8e999567835e1a",
"zh:6816de7f6bd3cfe341451af3be95b0af17c5539733b165f7505e88de6b730fc5",
"zh:729d75ab50efb675716ffa609358f6d9e80c66f7f64e01e11c8926c5293385be",
"zh:7d1024f621fe02b3731dc65ab20105150762b29df46e398da2c73d007bb62fb2",
"zh:81cf9cd23a70e5196b9d774482259c17e08471556932c348d3e9df0f7a482af1",
"zh:8de75b9c4b89ed1affcdb3ce32e00afaf213cfd76e93181ea4758ffd10b6eedc",
"zh:8f0b9b175ceb124c1cf01e6bbb660bf1bce0ff77416e026dc0679ed7156bc737",
"zh:8fb4cde10eb346ef9e2233591d75c39bcc2cafee578b669196192e8acdb39f12",
"zh:96798e58b8fcddc2add2da0fe8f1df9f913153d4cbccd68a7557b64633f5ea00",
"zh:a598bd615f4465b828f3b6ed2db694d8148c9469d5497922c7bc17c08b655cf4",
"zh:b2604d5067f3f259ff9be9bb0c4f9ec801d27c905cbf6ceafee50fc2a23f8331",
]
}
@@ -1,10 +0,0 @@
terraform {
required_version = "~> 1.10"
required_providers {
tailscale = {
source = "tailscale/tailscale"
version = "0.29.2"
}
}
}
@@ -1,6 +0,0 @@
provider "tailscale" {
# OAuth client from the o11y_tf vault. Must hold write on the tailnet policy file.
oauth_client_id = var.tailscale_oauth_client_id
oauth_client_secret = var.tailscale_oauth_client_secret
tailnet = var.tailscale_tailnet_id
}
@@ -1,59 +0,0 @@
# Tailnet-global. Declares all envs in one document so applying for one env
# doesn't wipe another env's tags. autoApprovers pre-approves each env's CP
# subnet route so kubectl/talosctl over Tailscale works without manual
# approval.
resource "tailscale_acl" "this" {
overwrite_existing_content = true
acl = jsonencode({
tagOwners = merge(
{
"tag:management" = []
"tag:project-yucca" = ["autogroup:admin"]
},
{
for env in keys(var.subnet_routes_by_env) :
"tag:env-${env}" => ["autogroup:admin"]
},
)
autoApprovers = {
routes = {
for env, cidr in var.subnet_routes_by_env :
cidr => ["tag:env-${env}"]
}
}
grants = [
{
src = ["*"]
dst = ["*"]
ip = ["*"]
}
]
ssh = [
{
action = "check"
src = ["autogroup:member"]
dst = ["autogroup:self"]
users = ["autogroup:nonroot", "root"]
},
{
action = "accept"
src = ["autogroup:admin"]
dst = ["tag:management"]
users = ["autogroup:nonroot"]
}
]
})
}
resource "tailscale_tailnet_settings" "org" {
devices_approval_on = true
devices_auto_updates_on = true
devices_key_duration_days = 5
users_approval_on = true
users_role_allowed_to_join_external_tailnet = "member"
https_enabled = true
}
@@ -1,20 +0,0 @@
variable "tailscale_oauth_client_id" {
sensitive = true
}
variable "tailscale_oauth_client_secret" {
sensitive = true
}
variable "tailscale_tailnet_id" {
sensitive = true
}
# CIDRs must match each env's `private_network_cidr` in
# deployment/modules/ovh/account/terragrunt.hcl.
variable "subnet_routes_by_env" {
type = map(string)
default = {
development = "10.150.50.0/24"
staging = "10.150.200.0/24"
production = "10.150.100.0/24"
}
}
-33
View File
@@ -22,36 +22,3 @@ provider "registry.opentofu.org/siderolabs/talos" {
"zh:d218bab0f67a2a8b15add9b51df3d30f514b57e9a7c1d733ebe97966ea132acb",
]
}
provider "registry.opentofu.org/tailscale/tailscale" {
version = "0.29.2"
constraints = "0.29.2"
hashes = [
"h1:7D3VzQoUKr4NYJ7ZMMRdV9iQUK8d6yJBrnBhi0DtBfA=",
"h1:910l+uQ0y8nSWPo/CsGI0Ni+lM6oc/h6yckucLmivAY=",
"h1:IPFMdH5vsXeNRj8H/Y6Z6iq3Fko8zkmgcROSXeRJ7MM=",
"h1:KG/OMAOors/W0AZlHkcUD9aeGBIWDCloKa652+JlATM=",
"h1:SSZ93MdSAaJ1Xi/VIvZDz5z1sve3BIS+WqDKACvJut0=",
"h1:Vzj5bDkG9nOQbRMKHPhsol0+BrvzZ22lwkbLSIs4ycs=",
"h1:Xjuo1Cwe065i1qJfhKm3dti4eeEwIT9rNNqU3R6id0A=",
"h1:eADNOR3ZnirZXCP+3k0hy9CKQK8sgeVC0lpo08UntZY=",
"h1:gzBWWbJc4JOwQEIINtnYkbwErRkA2oLvhtRW4lAQ7UU=",
"h1:ipdqf/NJpSaP7em0/+n3hu0KVzwai3cIn2VgP4Rxdhs=",
"h1:lNitoP/DTekzHnRjZ3RBLthPRZ3aHy6lR7jo2rtoH+0=",
"h1:laqGsqlHY2/R1JQnhniKwcT0U8gb+tF0K9vr+Tovskc=",
"h1:wUSb/6AFeFi905uIodIMA11SyyboCTtDqqNhJYdHweM=",
"zh:32b453302a684584198a03c2e09d99ef1f6deac2fe26f8dc134d973b07fd0c23",
"zh:3920bb891a476f30e29248533f667a507c97e93e0a2af010b242e980d6411ca6",
"zh:630cceb40d8806945ad36e04517f313520ea5cee17bd36e26a8e999567835e1a",
"zh:6816de7f6bd3cfe341451af3be95b0af17c5539733b165f7505e88de6b730fc5",
"zh:729d75ab50efb675716ffa609358f6d9e80c66f7f64e01e11c8926c5293385be",
"zh:7d1024f621fe02b3731dc65ab20105150762b29df46e398da2c73d007bb62fb2",
"zh:81cf9cd23a70e5196b9d774482259c17e08471556932c348d3e9df0f7a482af1",
"zh:8de75b9c4b89ed1affcdb3ce32e00afaf213cfd76e93181ea4758ffd10b6eedc",
"zh:8f0b9b175ceb124c1cf01e6bbb660bf1bce0ff77416e026dc0679ed7156bc737",
"zh:8fb4cde10eb346ef9e2233591d75c39bcc2cafee578b669196192e8acdb39f12",
"zh:96798e58b8fcddc2add2da0fe8f1df9f913153d4cbccd68a7557b64633f5ea00",
"zh:a598bd615f4465b828f3b6ed2db694d8148c9469d5497922c7bc17c08b655cf4",
"zh:b2604d5067f3f259ff9be9bb0c4f9ec801d27c905cbf6ceafee50fc2a23f8331",
]
}
@@ -6,9 +6,5 @@ terraform {
source = "siderolabs/talos"
version = "0.11.0"
}
tailscale = {
source = "tailscale/tailscale"
version = "0.29.2"
}
}
}
@@ -1,27 +1,3 @@
resource "tailscale_tailnet_key" "controlplane" {
for_each = var.controlplane_nodes
reusable = true
ephemeral = true
preauthorized = true
recreate_if_invalid = "always"
expiry = 7776000
description = "Talos key ${each.value.name}"
tags = [
"tag:project-yucca",
"tag:env-${var.env}",
]
}
data "tailscale_device" "controlplane" {
for_each = var.controlplane_nodes
hostname = each.value.name
wait_for = "300s"
depends_on = [talos_machine_bootstrap.this]
}
data "talos_machine_configuration" "controlplane" {
cluster_name = local.cluster_name
cluster_endpoint = local.cluster_endpoint
@@ -63,8 +39,8 @@ resource "talos_machine_configuration_apply" "controlplane" {
}
]
# Pin the kubelet's node IP to the vRack subnet so the Kubernetes
# InternalIP is always the private IP and never falls back to the
# Tailscale CGNAT address (which happens if eth1 has no IP at kubelet
# InternalIP is always the private IP and never falls back to a mesh
# overlay address (which happens if eth1 has no IP at kubelet
# start — see the static address below).
kubelet = {
nodeIP = {
@@ -138,7 +114,7 @@ resource "talos_machine_configuration_apply" "controlplane" {
}
apiServer = {
# VIP for in-cluster traffic; CP private IPs for operators reaching the
# apiserver over Tailscale (the floating VIP doesn't ARP reliably across DCs).
# apiserver over the mesh (the floating VIP doesn't ARP reliably across DCs).
certSANs = concat(
[local.controlplane_vip],
[for k in local.controlplane_keys : var.controlplane_nodes[k].private_ip],
@@ -155,19 +131,21 @@ resource "talos_machine_configuration_apply" "controlplane" {
hostname: ${each.value.name}
EOT
,
# Netbird overlay — the node management mesh. Operators reach the vRack IPs
# via the Netbird route (server-side, masqueraded to a routing-peer's vRack
# IP). NB_MANAGEMENT_URL is set explicitly to mirror yucca (Cloud default).
<<-EOT
name: tailscale
name: netbird
apiVersion: v1alpha1
kind: ExtensionServiceConfig
environment:
- TS_AUTHKEY=${tailscale_tailnet_key.controlplane[each.key].key}
- TS_HOSTNAME=${each.value.name}
- TS_ROUTES=${var.private_network_cidr}
- TS_EXTRA_ARGS=--accept-dns=false
- NB_SETUP_KEY=${var.netbird_setup_key}
- NB_MANAGEMENT_URL=https://api.netbird.io
EOT
,
# Talos ingress firewall — default-deny on host-bound services. Allow
# rules use Tailscale CGNAT (operators) and the vRack CIDR (intra-cluster).
# Talos ingress firewall — default-deny on host-bound services. Operators reach
# apid/apiserver via the Netbird route, masqueraded to a routing-peer's vRack IP,
# so every allow rule is the vRack CIDR (+ pod CIDR for in-cluster metrics).
<<-EOT
apiVersion: v1alpha1
kind: NetworkDefaultActionConfig
@@ -183,8 +161,6 @@ resource "talos_machine_configuration_apply" "controlplane" {
- 50000
protocol: tcp
ingress:
- subnet: 100.64.0.0/10
- subnet: fd7a:115c:a1e0::/48
- subnet: ${var.private_network_cidr}
EOT
,
@@ -209,8 +185,6 @@ resource "talos_machine_configuration_apply" "controlplane" {
- 6443
protocol: tcp
ingress:
- subnet: 100.64.0.0/10
- subnet: fd7a:115c:a1e0::/48
- subnet: ${var.private_network_cidr}
EOT
,
@@ -11,13 +11,6 @@ output "cluster" {
}
}
output "controlplane_tailscale_ips" {
value = {
for k, _ in var.controlplane_nodes :
k => data.tailscale_device.controlplane[k].addresses[0]
}
}
output "talos_client_configuration" {
sensitive = true
value = data.talos_client_configuration.this.talos_config
@@ -1,7 +0,0 @@
provider "tailscale" {
# OAuth client from the o11y_tf vault. Must hold write on auth keys + read on
# devices, and own the tag:project-yucca / tag:env-* tags it issues keys with.
oauth_client_id = var.tailscale_oauth_client_id
oauth_client_secret = var.tailscale_oauth_client_secret
tailnet = var.tailscale_tailnet_id
}
@@ -63,11 +63,12 @@ dependency "ovh" {
mock_outputs_merge_strategy_with_state = "shallow"
}
dependency "tailscale" {
config_path = "../../tailscale/account"
dependency "netbird_cluster" {
config_path = "../../netbird/cluster"
mock_outputs = {
tailscale_output = "mock-tailscale-output"
talos_setup_key = "mock-netbird-setup-key"
}
mock_outputs_allowed_terraform_commands = ["init", "validate", "plan"]
}
inputs = {
@@ -78,6 +79,7 @@ inputs = {
worker_data_disk_match = local.worker_data_disk_match
worker_data_disk2_match = local.worker_data_disk2_match
worker_nics = local.worker_nics
netbird_setup_key = dependency.netbird_cluster.outputs.talos_setup_key
}
generate "backend" {
+7 -10
View File
@@ -1,13 +1,10 @@
variable "env" {}
variable "stage" {}
variable "tailscale_oauth_client_id" {
sensitive = true
}
variable "tailscale_oauth_client_secret" {
sensitive = true
}
variable "tailscale_tailnet_id" {
# Netbird setup key from the netbird/cluster module (terragrunt dependency). Fed to
# every node's netbird ExtensionServiceConfig (NB_SETUP_KEY). Takes effect once the
# node runs a schematic that includes siderolabs/netbird.
variable "netbird_setup_key" {
sensitive = true
}
@@ -43,7 +40,7 @@ variable "talos_installer_images" {
variable "talos_version" {
type = string
default = "v1.13.0"
default = "v1.13.5"
}
variable "controlplane_vip_offset" {
@@ -87,9 +84,9 @@ variable "worker_nics" {
}))
}
# True only during initial bring-up of a brand-new env, before the Tailscale
# True only during initial bring-up of a brand-new env, before the Netbird
# extension has registered any node. Drop back to false once each node is on
# the tailnet, so future applies go via the vRack and the ingress firewall
# the netbird mesh, so future applies go via the vRack and the ingress firewall
# can drop public-NIC traffic without locking terraform out.
variable "use_public_endpoints" {
type = bool
+6 -23
View File
@@ -1,18 +1,3 @@
resource "tailscale_tailnet_key" "worker" {
for_each = var.worker_nodes
reusable = true
ephemeral = true
preauthorized = true
recreate_if_invalid = "always"
expiry = 7776000
description = "Talos key ${each.value.name}"
tags = [
"tag:project-yucca",
"tag:env-${var.env}",
]
}
data "talos_machine_configuration" "worker" {
cluster_name = local.cluster_name
cluster_endpoint = local.cluster_endpoint
@@ -80,16 +65,16 @@ resource "talos_machine_configuration_apply" "worker" {
hostname: ${each.value.name}
EOT
,
# Tailscale extension is baked into the worker image's schematic; without
# this config block the service starts unauthenticated and hangs.
# Netbird overlay — the node management mesh. Operators reach the vRack IPs
# via the Netbird route (server-side, masqueraded to a routing-peer's vRack
# IP). NB_MANAGEMENT_URL is set explicitly to mirror yucca (Cloud default).
<<-EOT
name: tailscale
name: netbird
apiVersion: v1alpha1
kind: ExtensionServiceConfig
environment:
- TS_AUTHKEY=${tailscale_tailnet_key.worker[each.key].key}
- TS_HOSTNAME=${each.value.name}
- TS_EXTRA_ARGS=--accept-dns=false
- NB_SETUP_KEY=${var.netbird_setup_key}
- NB_MANAGEMENT_URL=https://api.netbird.io
EOT
,
<<-EOT
@@ -144,8 +129,6 @@ resource "talos_machine_configuration_apply" "worker" {
- 50000
protocol: tcp
ingress:
- subnet: 100.64.0.0/10
- subnet: fd7a:115c:a1e0::/48
- subnet: ${var.private_network_cidr}
EOT
,
+10 -10
View File
@@ -12,8 +12,8 @@ How to stand up an environment from nothing. The cluster is built by Terragrunt
mise run talos:dl:cp && mise run talos:ul:cp
```
This image carries the `qemu-guest-agent` + `tailscale` schematic.
4. Workers need **no download or upload** — they are OVH BYOI and pull the bare-metal raw straight from the Talos Factory at order time. The worker schematic must stay **tailscale-only**: `qemu-guest-agent` on bare metal blocks on a virtio port that never appears and reboot-loops the node.
This image carries the `qemu-guest-agent` + `netbird` schematic.
4. Workers need **no download or upload** — they are OVH BYOI and pull the bare-metal raw straight from the Talos Factory at order time. The worker schematic must stay **netbird-only**: `qemu-guest-agent` on bare metal blocks on a virtio port that never appears and reboot-loops the node.
5. For production: delete the apex DNS records via the OVH dashboard before applying.
## Apply order
@@ -31,19 +31,19 @@ export TF_VAR_env=staging
mise run tg run --working-dir deployment/modules/ovh/account apply
```
2. **Tailscale** — tailnet-global ACL (only needs one run across all environments).
2. **NetBird** — the per-environment mesh objects: the Talos node group, a reusable setup key, the vRack network route (Talos nodes as routing peers), and the `yucca → resource` access policy. The Talos module consumes the setup key from here, so apply NetBird first.
```bash
mise run tg run --working-dir deployment/modules/tailscale/account apply
mise run tg run --working-dir deployment/modules/netbird/cluster apply
```
3. **Talos (bootstrap)** — initial bring-up over public IPs, because the Tailscale extension isn't running yet.
3. **Talos (bootstrap)** — initial bring-up over public IPs, because the NetBird extension isn't running yet.
```bash
TF_VAR_use_public_endpoints=true mise run tg run --working-dir deployment/modules/talos/cluster apply
```
4. **Verify** the cluster is up and operator-side Tailscale routing works. Pull the configs (see [Cluster access](#cluster-access)) and hit the APIs over the tailnet:
4. **Verify** the cluster is up and operator-side NetBird routing works. Pull the configs (see [Cluster access](#cluster-access)) and hit the APIs over the NetBird network:
```bash
mise run talos:kubeconfig && mise run talos:talosconfig
@@ -51,7 +51,7 @@ export TF_VAR_env=staging
talosctl --talosconfig .private/$ENVIRONMENT/talosconfig -n 10.150.200.10 get members
```
5. **Talos (steady state)** — drop the public-endpoints override now that Tailscale routes work; the host firewall closes the public NIC (everything except `:30443` on workers).
5. **Talos (steady state)** — drop the public-endpoints override now that NetBird routes work; the host firewall closes the public NIC (everything except `:30443` on workers).
```bash
unset TF_VAR_use_public_endpoints
@@ -72,7 +72,7 @@ How to get `kubectl` / `talosctl` access to an **existing** cluster (no bootstra
**Prerequisites:**
* **Tailscale** — the cluster APIs are reachable only over the tailnet, so your host needs Tailscale running with subnet-route consumption enabled: `tailscale set --accept-routes` on Linux, or the "Use Tailscale subnets" toggle in the macOS app.
* **NetBird** — the cluster APIs are reachable only over the NetBird network, so your host must be running the NetBird client (`netbird up`) and joined to the FUTO NetBird account, which places your peer in the `yucca` group. The access policy then distributes the route to the cluster's vRack CIDR, so `kubectl`/`talosctl` can reach the nodes' private IPs.
* **1Password CLI (`op`)** — installed and signed in to the `team-futo.1password.com` account. `mise run talos:config` fetches the configs through `mise run tg`, which wraps `op run` to inject the Terraform state credentials; without an authenticated `op` it can't read state.
**Fetch the configs.** Two tasks pull `kubeconfig` and `talosconfig` for the environment (run whichever you need):
@@ -84,9 +84,9 @@ mise run talos:kubeconfig # for kubectl
mise run talos:talosconfig # for talosctl
```
Each writes to `.private/$ENVIRONMENT/` (mode 600) from the Talos module's Terraform outputs. `talos:kubeconfig` also repoints the kubeconfig `server:` from the floating VIP (`10.150.200.5`) to a control-plane private IP (`10.150.200.10`) — the VIP doesn't ARP reliably across DCs over Tailscale, and every CP IP is in the apiserver cert SANs so TLS still validates.
Each writes to `.private/$ENVIRONMENT/` (mode 600) from the Talos module's Terraform outputs. `talos:kubeconfig` also repoints the kubeconfig `server:` from the floating VIP (`10.150.200.5`) to a control-plane private IP (`10.150.200.10`) — the VIP doesn't ARP reliably across DCs over the NetBird network, and every CP IP is in the apiserver cert SANs so TLS still validates.
> **A highly-available operator API endpoint is TBD.** `kubectl` is pinned to a single control-plane IP, so if that CP is down you currently repoint to another by hand (any CP IP works — they're all cert SANs). The floating VIP is HA *inside* the cluster (kubelet and in-cluster clients use it) but doesn't ARP across DCs over Tailscale, so there's no HA endpoint for operators yet.
> **A highly-available operator API endpoint is TBD.** `kubectl` is pinned to a single control-plane IP, so if that CP is down you currently repoint to another by hand (any CP IP works — they're all cert SANs). The floating VIP is HA *inside* the cluster (kubelet and in-cluster clients use it) but doesn't ARP across DCs over the NetBird network, so there's no HA endpoint for operators yet.
**Point your tools at them:**
+5 -5
View File
@@ -2,7 +2,7 @@
The OVH foundation under the cluster: compute, private network, and public ingress. Each environment is a fully independent build of the same shape; they differ only in worker tier and IPLB zone count.
> **Status:** staging is built and running (`o11y-staging`). Production is planned (`o11y-production`) — same shape, larger workers, multi-zone ingress.
> **Status:** both environments are built and running — `o11y-staging` and `o11y-production`. Same shape; production has larger workers and multi-zone ingress.
## Shape
@@ -42,11 +42,11 @@ The worker host firewall scopes `:30443` to OVH's IPLB NAT range (`10.108.0.0/14
Because the farm targets the workers' public IPs (not the vRack), three things are required and are handled in the cluster config: NodePorts must answer on the public NIC, exactly one Envoy must run per worker, and Envoy must parse PROXY protocol. See the cluster architecture guide for those details.
## Operator access (Tailscale)
## Operator access (NetBird)
Tailscale runs as a Talos system extension on **every** node, so operators reach `talosctl` and `kubectl` over the tailnet without exposing those APIs publicly. Control planes advertise the private CIDR as a subnet route, auto-approved by the tailnet ACL; workers consume the routes. The ACL is environment-scoped (`tag:env-staging` vs `tag:env-production`) so staging operators can't pivot into production.
NetBird runs as a Talos system extension on **every** node, so operators reach `talosctl` and `kubectl` over the NetBird network without exposing those APIs publicly. The vRack subnet is published as a NetBird network route with the Talos nodes as routing peers — any node can route, so it's HA — and operator traffic is masqueraded to the routing peer's vRack IP, which the host firewall already trusts. A per-environment access policy lets the shared `yucca` operator group reach this environment's routed subnet on the management ports only (apid `50000`, kube-apiserver `6443`); the groups and policy are environment-scoped (`O11Y_STAGING_*` vs `O11Y_PRODUCTION_*`), so staging operators can't pivot into production.
Operators point `kubectl`/`talosctl` at a specific control plane's static private IP — not the floating VIP, since cross-DC ARP for the VIP over Tailscale subnet routes is unreliable. The VIP remains the in-cluster apiserver endpoint used by kubelet and other in-cluster components.
Operators point `kubectl`/`talosctl` at a specific control plane's static private IP — not the floating VIP, since cross-DC ARP for the VIP over the NetBird network route is unreliable. The VIP remains the in-cluster apiserver endpoint used by kubelet and other in-cluster components.
## Cost
@@ -69,5 +69,5 @@ Staging + production run-rate ≈ **$955/mo** plus the one-time **$221** product
| Workers | 3× `SYS-2` (`24sys022`) | 3× `Rise-2` (`24rise02-v1`) |
| IPLB | 1 zone (`gra`) | 3 zones (`gra` + `rbx` + `sbg`), anycast |
| Private CIDR | `10.150.200.0/24` | `10.150.100.0/24` |
| Tailscale tag | `tag:env-staging` | `tag:env-production` |
| NetBird objects | `O11Y_STAGING_*` | `O11Y_PRODUCTION_*` |
| Flux source | `staging` overlay | `production` overlay |
+7 -5
View File
@@ -12,8 +12,8 @@ Control planes and workers use **different** Talos Factory schematics, on purpos
| Node type | Platform | Schematic |
|-----------|----------|-----------|
| Control plane | OVH Public Cloud (KVM) | `tailscale` + `qemu-guest-agent` |
| Worker | Bare metal | `tailscale` only |
| Control plane | OVH Public Cloud (KVM) | `netbird` + `qemu-guest-agent` |
| Worker | Bare metal | `netbird` only |
`qemu-guest-agent` on bare metal wedges boot — it waits on a virtio-serial port that isn't present and reboot-loops the node. The control-plane image is an OpenStack image uploaded to OVH glance once; workers are BYOI and fetch their raw image from the Factory at order time.
@@ -36,9 +36,9 @@ Default-deny ingress on every node; anything not listed is dropped at the host.
| Service | Port(s) | Allowed sources |
|---------|---------|-----------------|
| apid | 50000/tcp | Tailscale (`100.64.0.0/10`, `fd7a:115c:a1e0::/48`), vRack |
| trustd | 50001/tcp | Tailscale, vRack |
| kube-apiserver (CPs) | 6443/tcp | Tailscale, vRack |
| apid | 50000/tcp | vRack |
| trustd | 50001/tcp | vRack |
| kube-apiserver (CPs) | 6443/tcp | vRack |
| etcd (CPs) | 2379–2380/tcp | vRack |
| kubelet | 10250/tcp | vRack + pod CIDR `10.244.0.0/16` |
| flannel VXLAN | 4789/udp | vRack |
@@ -49,6 +49,8 @@ Default-deny ingress on every node; anything not listed is dropped at the host.
Pod CIDR is allowed on `kubelet` and the metrics ports because pod-to-own-node-IP traffic skips flannel masquerade (a same-node scrape keeps its pod-IP source), which the vRack-only rule would otherwise drop.
Operator `talosctl`/`kubectl` traffic needs no rule of its own: it arrives over the NetBird network route masqueraded to a routing peer's vRack IP, so the vRack allow on `apid` and `kube-apiserver` already covers it.
## Kubernetes
Kubernetes with flannel CNI and kube-proxy in nftables mode. Spegel runs as a peer-to-peer image registry mirror so each node's containerd pulls layers from its peers before the upstream registry (this requires `discard_unpacked_layers = false` in the worker containerd config).