mirror of
https://github.com/immich-app/yucca-o11y.git
synced 2026-09-30 13:23:23 +08:00
feat: migrate tailscale to netbird (#84)
Signed-off-by: Devin Buhl <devin@buhl.casa>
This commit is contained in:
+7
-7
@@ -25,8 +25,8 @@ TF_VAR_dist_dir="{{config_root}}/dist"
|
||||
ENVIRONMENT_SHORT = "{% set e = get_env(name='ENVIRONMENT', default='development') %}{% if e == 'development' or e == '' %}dev{% elif e == 'production' %}prod{% else %}{{ e }}{% endif %}"
|
||||
|
||||
# Control plane and workers boot from different Talos Factory schematics:
|
||||
# control plane (Public Cloud / KVM): tailscale + qemu-guest-agent
|
||||
# worker (bare metal): tailscale only — qemu-guest-agent wedges
|
||||
# control plane (Public Cloud / KVM): qemu-guest-agent + netbird
|
||||
# worker (bare metal): netbird only — qemu-guest-agent wedges
|
||||
# bare-metal boot and reboot-loops the node
|
||||
# Control planes need the image uploaded to OVH glance (talos:{dl,ul}:cp); workers
|
||||
# are OVH BYOI, so OVH fetches their raw from the Factory at order time (no upload).
|
||||
@@ -34,10 +34,10 @@ ENVIRONMENT_SHORT = "{% set e = get_env(name='ENVIRONMENT', default='development
|
||||
[tasks."talos:dl:cp"]
|
||||
run = """
|
||||
mkdir -p {{config_root}}/.private/dist
|
||||
wget https://factory.talos.dev/image/7d4c31cbd96db9f90c874990697c523482b2bae27fb4631d5583dcd9c281b1ff/v1.13.0/openstack-amd64.raw.xz -O {{config_root}}/.private/dist/talos.1.13.0-tailscale-qemu.raw.xz
|
||||
unxz {{config_root}}/.private/dist/talos.1.13.0-tailscale-qemu.raw.xz --force
|
||||
wget https://factory.talos.dev/image/bbfcb7053b1609712a977830952455432825890922cb6bac23cea34b980970f1/v1.13.5/openstack-amd64.raw.xz -O {{config_root}}/.private/dist/talos.1.13.5-qemu-netbird.raw.xz
|
||||
unxz {{config_root}}/.private/dist/talos.1.13.5-qemu-netbird.raw.xz --force
|
||||
"""
|
||||
description = "Download the control-plane Talos image (OpenStack, tailscale + qemu-guest-agent schematic)"
|
||||
description = "Download the control-plane Talos image (OpenStack, qemu-guest-agent + netbird schematic)"
|
||||
dir = "{{cwd}}"
|
||||
|
||||
[tasks."talos:ul:cp"]
|
||||
@@ -45,9 +45,9 @@ run = """
|
||||
for region in "RBX-A" "GRA9" "EU-WEST-PAR"; do
|
||||
echo "Uploading to region ${region}..."
|
||||
source {{config_root}}/.private/openstack/${ENVIRONMENT}/openrc.sh
|
||||
openstack image create "talos-1.13.0-tailscale-qemu" \
|
||||
openstack image create "talos-1.13.5-qemu-netbird" \
|
||||
--os-region "${region}" \
|
||||
--file {{config_root}}/.private/dist/talos.1.13.0-tailscale-qemu.raw \
|
||||
--file {{config_root}}/.private/dist/talos.1.13.5-qemu-netbird.raw \
|
||||
--disk-format raw \
|
||||
--container-format bare \
|
||||
--property hw_qemu_guest_agent=yes \
|
||||
|
||||
@@ -4,7 +4,7 @@ The centralized observability platform for FUTO services. A single Talos Kuberne
|
||||
|
||||
Built for geographic resilience with single-cluster operational simplicity: three control planes in low-latency DCs hold one etcd quorum, three bare-metal workers carry the observability workload, and everything talks over a private OVH vRack.
|
||||
|
||||
**Status:** staging is built and running (`o11y-staging`); production is planned (`o11y-production`).
|
||||
**Status:** both environments are built and running — `o11y-staging` and `o11y-production`.
|
||||
|
||||
## Documentation
|
||||
|
||||
@@ -20,7 +20,7 @@ Built for geographic resilience with single-cluster operational simplicity: thre
|
||||
```text
|
||||
deployment/modules/
|
||||
├── ovh/account/ # cloud project, vRack, private network, CPs, workers, IPLB, DNS
|
||||
├── tailscale/account/ # tailnet-global ACL and tailnet settings
|
||||
├── netbird/cluster/ # per-env mesh: node group, setup key, vRack network route, access policy
|
||||
├── talos/cluster/ # machine secrets, CP + worker configs, bootstrap, ingress firewall
|
||||
└── kubernetes/helm/ # Flux Operator + Instance, env-scoped secrets
|
||||
|
||||
@@ -34,4 +34,4 @@ kubernetes/
|
||||
└── cluster-settings.yaml # per-env ConfigMap: APP_DOMAIN, CLUSTER_NAME
|
||||
```
|
||||
|
||||
State lives in S3 under `yucca/o11y/v3/<module>/<env>`. Secrets and OVH/Tailscale tokens come from the environment's 1Password vault via `op run` and `deployment/.env`.
|
||||
State lives in S3 under `yucca/o11y/v3/<module>/<env>`. Secrets and OVH/NetBird tokens come from 1Password via `op run` and `deployment/.env`.
|
||||
|
||||
+3
-4
@@ -10,10 +10,9 @@ export TF_VAR_tf_state_s3_region=op://o11y_tf/TF_STATE_S3_REGION/password
|
||||
export TF_VAR_tf_state_s3_access_key=op://o11y_tf/TF_STATE_S3_ACCESS_KEY/password
|
||||
export TF_VAR_tf_state_s3_secret_key=op://o11y_tf/TF_STATE_S3_SECRET_KEY/password
|
||||
|
||||
export TF_VAR_tailscale_oauth_client_id=op://o11y_tf/TAILSCALE_OAUTH_CLIENT_ID/password
|
||||
export TF_VAR_tailscale_oauth_client_secret=op://o11y_tf/TAILSCALE_OAUTH_CLIENT_SECRET/password
|
||||
export TF_VAR_tailscale_tailnet_id=op://o11y_tf/TAILSCALE_TAILNET_ID/password
|
||||
|
||||
export TF_VAR_op_credentials_file=op://o11y_tf/1PASS_CONNECT_SERVER_CREDENTIALS_FILE/password
|
||||
export TF_VAR_op_connect_token=op://o11y_tf/1PASS_CONNECT_O11Y_SUPERUSER/password
|
||||
export TF_VAR_op_connect_token_env=op://o11y_tf_${ENVIRONMENT_SHORT}/1PASS_CONNECT_O11Y_READ/password
|
||||
|
||||
# Netbird Cloud PAT (shared across envs) — drives the netbird/account TF provider.
|
||||
export TF_VAR_netbird_tf_pat=op://shared_tf/NETBIRD_TF_PAT/password
|
||||
|
||||
@@ -0,0 +1,38 @@
|
||||
# This file is maintained automatically by "tofu init".
|
||||
# Manual edits may be lost in future updates.
|
||||
|
||||
provider "registry.opentofu.org/netbirdio/netbird" {
|
||||
version = "0.0.9"
|
||||
constraints = "0.0.9"
|
||||
hashes = [
|
||||
"h1:0Vi0MLMk+K1CEuBZB+ByU55wXx87eGd5j0fGACekCDQ=",
|
||||
"h1:2ska0C0jDxvbjApKlUmdkfrXQFjHiNf/KKYt355Isgo=",
|
||||
"h1:H8GG0MZqAAXqnrMCw0Q3vcE3gGFEqYmDF9YY2RnD18Y=",
|
||||
"h1:HKmwtSPE6k++umLH99WkpT/HcNzy69OWRTVZMsDFJh8=",
|
||||
"h1:KFbgGf03fTGEgJRIpzAryWB26BHicfLQC8nuE5IiSys=",
|
||||
"h1:VMUHvZOWIU/IWmK+NyEUIafKS88gGTvff+nK8N7NE3E=",
|
||||
"h1:aFK3rhEjnnwiHU7Fl/nBDz5Y1b/2sRYXYomu4qewlwM=",
|
||||
"h1:ba2yqgEM9Ssy+C/0nP3XHh209Y20gpYIErZtYVVb2k8=",
|
||||
"h1:hC4RZOIVv5KTHcF+AuIHQe01hJmfw7yioSRTDb4ImDg=",
|
||||
"h1:jaLhsZX1wOdBWKED+hJ0kyfV5TE3Laim6Kd7Uw+krzQ=",
|
||||
"h1:mWLdkV/wXnK6SSMIKw+BJFJPonrts+7L5IEefRQ0rV8=",
|
||||
"h1:nvVlkGxQSfphiXMNgGmvuTZQOOjhtdHMe/Upx3/HWjA=",
|
||||
"h1:pjY52PZ8FJXUbeFcB0kSV8guz4h/xzlXLarWXelAwes=",
|
||||
"h1:trYvM2c8h5lQ0m8+3Bgvgnbkm04+VOAL2pr6odmFfS8=",
|
||||
"zh:1f34fba3ecfe0efa36d4b4bddf5714c69124e705d1e371e65abd291887d43385",
|
||||
"zh:203f671af4d1f5376f4e20fb8d82cc8dc6c4fd136d9abffe8eb7b7f34e27197b",
|
||||
"zh:21cb344302bdbadbc2779768116c7351d915bb7678a4847e9e2c8623d0032c48",
|
||||
"zh:2368868e0f86b458f83b6598317ff5fde0d4079500d4567e99f46bf80c63f07a",
|
||||
"zh:3237a426818eb3906153b1cf7fb6dfe9d52128972e7cab17c324d88788f86863",
|
||||
"zh:427f8e8ba190d69cfd493b1598b4bf7940fe9a521a333df3540d2e0ea07543cf",
|
||||
"zh:689cb219f3936500f5a3147299961f8740e505612b5182d93840cb8152d89b4f",
|
||||
"zh:6e594e69d9e107c45ed4677a9bff68de2dd4232900b636327b24c819bee72c9e",
|
||||
"zh:715634d6d0b052ac3f34aeb4c2df42960cdf44dd5c45d060867db78716aebb48",
|
||||
"zh:890df766e9b839623b1f0437355032a3c006226a6c200cd911e15ee1a9014e9f",
|
||||
"zh:8cbbdde5a0b59f0bb427b403da8993b215d1a6231e0d6b9b4faa89db76b10705",
|
||||
"zh:af92281c3e14c6af53fb93cee584c5118c5c7bee0acb938b5a367a219d9bc2e2",
|
||||
"zh:bcdeef5fadf092e33228f28f013af7d8d9553b9d2fb2cecd1894351d5ad37a7a",
|
||||
"zh:d6c31cb12f18b25c9663c14e737554e06144d4e270e607337bac7963ca3b9542",
|
||||
"zh:e5873e8e0b29c8cbfd98f04759579da3c0529c8b38608c0f1ea92eb566c8ddc1",
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,10 @@
|
||||
terraform {
|
||||
required_version = "~> 1.10"
|
||||
|
||||
required_providers {
|
||||
netbird = {
|
||||
source = "netbirdio/netbird"
|
||||
version = "0.0.9"
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,82 @@
|
||||
# Per-env o11y NetBird objects (state key .../netbird/cluster/<env>). Everything is
|
||||
# per-env — resource groups and policies included — so access can differ by env
|
||||
# (Zack: some people get dev/staging but not prod). No account-wide layer.
|
||||
#
|
||||
# All object names are UPPER_SNAKE to match yucca's convention (groups, setup keys,
|
||||
# networks, network resources, and policies/rules).
|
||||
|
||||
# Existing account-wide users group (yucca peers — populated via users' auto_groups,
|
||||
# per NetBird's model). Referenced as the policy source so the same operators that
|
||||
# reach yucca also reach o11y — per the team decision.
|
||||
data "netbird_group" "yucca" {
|
||||
name = "yucca"
|
||||
}
|
||||
|
||||
# Group the Talos nodes auto-join via the setup key below.
|
||||
resource "netbird_group" "talos" {
|
||||
name = "O11Y_${upper(var.env)}_TALOS"
|
||||
}
|
||||
|
||||
# Per-env tag for this cluster's routed resources (the vRack subnet below tags into
|
||||
# it; the yucca->resource policy grants it). Created with its final name — the
|
||||
# NetBird provider can't rename a group once network resources are tagged into it.
|
||||
resource "netbird_group" "o11y_resource" {
|
||||
name = "O11Y_${upper(var.env)}_RESOURCE"
|
||||
}
|
||||
|
||||
# Reusable, non-ephemeral enrollment key for the Talos nodes — fed to the netbird
|
||||
# Talos extension as NB_SETUP_KEY. NOT ephemeral: ephemeral peers are reaped after
|
||||
# 10m idle, which would delete live nodes.
|
||||
resource "netbird_setup_key" "talos" {
|
||||
name = "O11Y_${upper(var.env)}_TALOS"
|
||||
type = "reusable"
|
||||
ephemeral = false
|
||||
expiry_seconds = 0 # unlimited — nodes re-enroll with the same key on reprovision
|
||||
usage_limit = 0 # unlimited
|
||||
auto_groups = [netbird_group.talos.id]
|
||||
}
|
||||
|
||||
# vRack subnet advertised to operators with the Talos nodes as routing peers — the
|
||||
# Netbird "Networks" model. Every node sits on the vRack, so any can route (HA);
|
||||
# masquerade NATs operator traffic to the routing peer's vRack IP, which the node
|
||||
# firewall already trusts.
|
||||
resource "netbird_network" "vrack" {
|
||||
name = "O11Y_${upper(var.env)}_VRACK"
|
||||
description = "o11y ${var.env} vRack private subnet"
|
||||
}
|
||||
|
||||
resource "netbird_network_router" "vrack" {
|
||||
network_id = netbird_network.vrack.id
|
||||
peer_groups = [netbird_group.talos.id]
|
||||
masquerade = true
|
||||
metric = 9999
|
||||
enabled = true
|
||||
}
|
||||
|
||||
resource "netbird_network_resource" "vrack" {
|
||||
network_id = netbird_network.vrack.id
|
||||
# Per-env: Netbird network-resource names are account-globally unique.
|
||||
name = "O11Y_${upper(var.env)}_VRACK_CIDR"
|
||||
address = var.private_network_cidr
|
||||
groups = [netbird_group.o11y_resource.id]
|
||||
enabled = true
|
||||
}
|
||||
|
||||
# NetBird default-denies; allow yucca operators to reach THIS env's routed subnet on
|
||||
# the Talos management ports only (apid + kube-apiserver). Per-env policy so access
|
||||
# can later differ by env — tighter than yucca's all-protocol yucca->yucca_resource.
|
||||
resource "netbird_policy" "yucca_to_o11y_resource" {
|
||||
name = "O11Y_${upper(var.env)}_YUCCA_TO_RESOURCE"
|
||||
enabled = true
|
||||
|
||||
rule {
|
||||
name = "YUCCA_TO_O11Y_RESOURCE"
|
||||
action = "accept"
|
||||
protocol = "tcp"
|
||||
enabled = true
|
||||
bidirectional = false
|
||||
sources = [data.netbird_group.yucca.id]
|
||||
destinations = [netbird_group.o11y_resource.id]
|
||||
ports = ["50000", "6443"]
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,6 @@
|
||||
# Plaintext setup key fed to the Talos netbird extension (NB_SETUP_KEY) by the
|
||||
# talos/cluster module, which consumes this via a terragrunt dependency.
|
||||
output "talos_setup_key" {
|
||||
sensitive = true
|
||||
value = netbird_setup_key.talos.key
|
||||
}
|
||||
@@ -0,0 +1,5 @@
|
||||
provider "netbird" {
|
||||
# PAT from the shared_tf vault (Netbird Cloud). management_url defaults to
|
||||
# https://api.netbird.io, so it is left unset.
|
||||
token = var.netbird_tf_pat
|
||||
}
|
||||
+21
-2
@@ -6,7 +6,26 @@ terraform {
|
||||
}
|
||||
}
|
||||
|
||||
# Tailnet-global; state key intentionally not parameterised by env.
|
||||
locals {
|
||||
env = get_env("TF_VAR_env")
|
||||
stage = get_env("TF_VAR_stage")
|
||||
}
|
||||
|
||||
# Per-env: the route advertises this env's vRack CIDR, sourced from the ovh module.
|
||||
dependency "ovh" {
|
||||
config_path = "../../ovh/account"
|
||||
|
||||
mock_outputs = {
|
||||
private_network_cidr = "10.150.200.0/24"
|
||||
}
|
||||
mock_outputs_allowed_terraform_commands = ["init", "validate", "plan"]
|
||||
mock_outputs_merge_strategy_with_state = "shallow"
|
||||
}
|
||||
|
||||
inputs = {
|
||||
private_network_cidr = dependency.ovh.outputs.private_network_cidr
|
||||
}
|
||||
|
||||
generate "backend" {
|
||||
path = "backend.tf"
|
||||
if_exists = "overwrite_terragrunt"
|
||||
@@ -14,7 +33,7 @@ generate "backend" {
|
||||
terraform {
|
||||
backend "s3" {
|
||||
bucket = "${get_env("TF_VAR_tf_state_s3_bucket")}"
|
||||
key = "yucca/o11y/v3/tailscale/account/global"
|
||||
key = "yucca/o11y/v3/netbird/cluster/${local.env}${local.stage != "" ? "/${local.stage}" : ""}"
|
||||
region = "${get_env("TF_VAR_tf_state_s3_region")}"
|
||||
access_key = "${get_env("TF_VAR_tf_state_s3_access_key")}"
|
||||
secret_key = "${get_env("TF_VAR_tf_state_s3_secret_key")}"
|
||||
@@ -0,0 +1,11 @@
|
||||
variable "netbird_tf_pat" {
|
||||
sensitive = true
|
||||
}
|
||||
|
||||
variable "env" {}
|
||||
|
||||
# Matches the env's private_network_cidr from the ovh module — the vRack subnet
|
||||
# the Talos nodes advertise as a NetBird network route for operator access.
|
||||
variable "private_network_cidr" {
|
||||
type = string
|
||||
}
|
||||
@@ -42,28 +42,28 @@ variable "vrack_name" {
|
||||
|
||||
variable "talos_version" {
|
||||
type = string
|
||||
default = "v1.13.0"
|
||||
default = "v1.13.5"
|
||||
}
|
||||
|
||||
# Control-plane (Public Cloud / KVM) schematic: tailscale + qemu-guest-agent.
|
||||
# Control-plane (Public Cloud / KVM) schematic: qemu-guest-agent + netbird.
|
||||
variable "talos_schematic_id" {
|
||||
type = string
|
||||
default = "7d4c31cbd96db9f90c874990697c523482b2bae27fb4631d5583dcd9c281b1ff"
|
||||
default = "bbfcb7053b1609712a977830952455432825890922cb6bac23cea34b980970f1"
|
||||
}
|
||||
|
||||
# Worker (bare-metal) schematic: tailscale only. qemu-guest-agent must NOT be
|
||||
# Worker (bare-metal) schematic: netbird only. qemu-guest-agent must NOT be
|
||||
# present on bare metal — it blocks on a virtio port that never appears, which
|
||||
# wedges the Talos boot sequence and reboots the node in a loop.
|
||||
variable "talos_worker_schematic_id" {
|
||||
type = string
|
||||
default = "4a0d65c669d46663f377e7161e50cfd570c401f26fd9e7bda34a0216b6f1922b"
|
||||
default = "7326f0cbca7a0e700ac1efa3f32e88df9ebe5010e6e842a8ed36fdc99ee98ead"
|
||||
}
|
||||
|
||||
# Image must be pre-uploaded out-of-band (talos:dl:cp + talos:ul:cp mise tasks) —
|
||||
# the OVH provider doesn't upload custom images.
|
||||
variable "talos_public_cloud_image_name" {
|
||||
type = string
|
||||
default = "talos-1.13.0-tailscale-qemu"
|
||||
default = "talos-1.13.5-qemu-netbird"
|
||||
}
|
||||
|
||||
# IPLB tier and geographic zone (public-IP location). The LB reaches the workers
|
||||
|
||||
@@ -61,7 +61,7 @@ resource "ovh_dedicated_server" "worker" {
|
||||
customizations = {
|
||||
efi_bootloader_path = "\\EFI\\BOOT\\BOOTX64.EFI"
|
||||
# OVH fetches this raw straight from the Talos Factory at order time (no
|
||||
# OVH-side upload). The URL resolves the Tailscale-only worker schematic;
|
||||
# OVH-side upload). The URL resolves the netbird-only worker schematic;
|
||||
# qemu-guest-agent here would reboot-loop the bare-metal node.
|
||||
image_url = replace(data.talos_image_factory_urls.metal.urls.iso, ".iso", ".raw")
|
||||
image_type = "raw"
|
||||
|
||||
@@ -1,35 +0,0 @@
|
||||
# This file is maintained automatically by "tofu init".
|
||||
# Manual edits may be lost in future updates.
|
||||
|
||||
provider "registry.opentofu.org/tailscale/tailscale" {
|
||||
version = "0.29.2"
|
||||
constraints = "0.29.2"
|
||||
hashes = [
|
||||
"h1:7D3VzQoUKr4NYJ7ZMMRdV9iQUK8d6yJBrnBhi0DtBfA=",
|
||||
"h1:910l+uQ0y8nSWPo/CsGI0Ni+lM6oc/h6yckucLmivAY=",
|
||||
"h1:IPFMdH5vsXeNRj8H/Y6Z6iq3Fko8zkmgcROSXeRJ7MM=",
|
||||
"h1:KG/OMAOors/W0AZlHkcUD9aeGBIWDCloKa652+JlATM=",
|
||||
"h1:SSZ93MdSAaJ1Xi/VIvZDz5z1sve3BIS+WqDKACvJut0=",
|
||||
"h1:Vzj5bDkG9nOQbRMKHPhsol0+BrvzZ22lwkbLSIs4ycs=",
|
||||
"h1:Xjuo1Cwe065i1qJfhKm3dti4eeEwIT9rNNqU3R6id0A=",
|
||||
"h1:eADNOR3ZnirZXCP+3k0hy9CKQK8sgeVC0lpo08UntZY=",
|
||||
"h1:gzBWWbJc4JOwQEIINtnYkbwErRkA2oLvhtRW4lAQ7UU=",
|
||||
"h1:ipdqf/NJpSaP7em0/+n3hu0KVzwai3cIn2VgP4Rxdhs=",
|
||||
"h1:lNitoP/DTekzHnRjZ3RBLthPRZ3aHy6lR7jo2rtoH+0=",
|
||||
"h1:laqGsqlHY2/R1JQnhniKwcT0U8gb+tF0K9vr+Tovskc=",
|
||||
"h1:wUSb/6AFeFi905uIodIMA11SyyboCTtDqqNhJYdHweM=",
|
||||
"zh:32b453302a684584198a03c2e09d99ef1f6deac2fe26f8dc134d973b07fd0c23",
|
||||
"zh:3920bb891a476f30e29248533f667a507c97e93e0a2af010b242e980d6411ca6",
|
||||
"zh:630cceb40d8806945ad36e04517f313520ea5cee17bd36e26a8e999567835e1a",
|
||||
"zh:6816de7f6bd3cfe341451af3be95b0af17c5539733b165f7505e88de6b730fc5",
|
||||
"zh:729d75ab50efb675716ffa609358f6d9e80c66f7f64e01e11c8926c5293385be",
|
||||
"zh:7d1024f621fe02b3731dc65ab20105150762b29df46e398da2c73d007bb62fb2",
|
||||
"zh:81cf9cd23a70e5196b9d774482259c17e08471556932c348d3e9df0f7a482af1",
|
||||
"zh:8de75b9c4b89ed1affcdb3ce32e00afaf213cfd76e93181ea4758ffd10b6eedc",
|
||||
"zh:8f0b9b175ceb124c1cf01e6bbb660bf1bce0ff77416e026dc0679ed7156bc737",
|
||||
"zh:8fb4cde10eb346ef9e2233591d75c39bcc2cafee578b669196192e8acdb39f12",
|
||||
"zh:96798e58b8fcddc2add2da0fe8f1df9f913153d4cbccd68a7557b64633f5ea00",
|
||||
"zh:a598bd615f4465b828f3b6ed2db694d8148c9469d5497922c7bc17c08b655cf4",
|
||||
"zh:b2604d5067f3f259ff9be9bb0c4f9ec801d27c905cbf6ceafee50fc2a23f8331",
|
||||
]
|
||||
}
|
||||
@@ -1,10 +0,0 @@
|
||||
terraform {
|
||||
required_version = "~> 1.10"
|
||||
|
||||
required_providers {
|
||||
tailscale = {
|
||||
source = "tailscale/tailscale"
|
||||
version = "0.29.2"
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,6 +0,0 @@
|
||||
provider "tailscale" {
|
||||
# OAuth client from the o11y_tf vault. Must hold write on the tailnet policy file.
|
||||
oauth_client_id = var.tailscale_oauth_client_id
|
||||
oauth_client_secret = var.tailscale_oauth_client_secret
|
||||
tailnet = var.tailscale_tailnet_id
|
||||
}
|
||||
@@ -1,59 +0,0 @@
|
||||
# Tailnet-global. Declares all envs in one document so applying for one env
|
||||
# doesn't wipe another env's tags. autoApprovers pre-approves each env's CP
|
||||
# subnet route so kubectl/talosctl over Tailscale works without manual
|
||||
# approval.
|
||||
resource "tailscale_acl" "this" {
|
||||
overwrite_existing_content = true
|
||||
|
||||
acl = jsonencode({
|
||||
tagOwners = merge(
|
||||
{
|
||||
"tag:management" = []
|
||||
"tag:project-yucca" = ["autogroup:admin"]
|
||||
},
|
||||
{
|
||||
for env in keys(var.subnet_routes_by_env) :
|
||||
"tag:env-${env}" => ["autogroup:admin"]
|
||||
},
|
||||
)
|
||||
|
||||
autoApprovers = {
|
||||
routes = {
|
||||
for env, cidr in var.subnet_routes_by_env :
|
||||
cidr => ["tag:env-${env}"]
|
||||
}
|
||||
}
|
||||
|
||||
grants = [
|
||||
{
|
||||
src = ["*"]
|
||||
dst = ["*"]
|
||||
ip = ["*"]
|
||||
}
|
||||
]
|
||||
|
||||
ssh = [
|
||||
{
|
||||
action = "check"
|
||||
src = ["autogroup:member"]
|
||||
dst = ["autogroup:self"]
|
||||
users = ["autogroup:nonroot", "root"]
|
||||
},
|
||||
{
|
||||
action = "accept"
|
||||
src = ["autogroup:admin"]
|
||||
dst = ["tag:management"]
|
||||
users = ["autogroup:nonroot"]
|
||||
}
|
||||
]
|
||||
})
|
||||
}
|
||||
|
||||
resource "tailscale_tailnet_settings" "org" {
|
||||
devices_approval_on = true
|
||||
devices_auto_updates_on = true
|
||||
devices_key_duration_days = 5
|
||||
users_approval_on = true
|
||||
users_role_allowed_to_join_external_tailnet = "member"
|
||||
https_enabled = true
|
||||
}
|
||||
@@ -1,20 +0,0 @@
|
||||
variable "tailscale_oauth_client_id" {
|
||||
sensitive = true
|
||||
}
|
||||
variable "tailscale_oauth_client_secret" {
|
||||
sensitive = true
|
||||
}
|
||||
variable "tailscale_tailnet_id" {
|
||||
sensitive = true
|
||||
}
|
||||
|
||||
# CIDRs must match each env's `private_network_cidr` in
|
||||
# deployment/modules/ovh/account/terragrunt.hcl.
|
||||
variable "subnet_routes_by_env" {
|
||||
type = map(string)
|
||||
default = {
|
||||
development = "10.150.50.0/24"
|
||||
staging = "10.150.200.0/24"
|
||||
production = "10.150.100.0/24"
|
||||
}
|
||||
}
|
||||
@@ -22,36 +22,3 @@ provider "registry.opentofu.org/siderolabs/talos" {
|
||||
"zh:d218bab0f67a2a8b15add9b51df3d30f514b57e9a7c1d733ebe97966ea132acb",
|
||||
]
|
||||
}
|
||||
|
||||
provider "registry.opentofu.org/tailscale/tailscale" {
|
||||
version = "0.29.2"
|
||||
constraints = "0.29.2"
|
||||
hashes = [
|
||||
"h1:7D3VzQoUKr4NYJ7ZMMRdV9iQUK8d6yJBrnBhi0DtBfA=",
|
||||
"h1:910l+uQ0y8nSWPo/CsGI0Ni+lM6oc/h6yckucLmivAY=",
|
||||
"h1:IPFMdH5vsXeNRj8H/Y6Z6iq3Fko8zkmgcROSXeRJ7MM=",
|
||||
"h1:KG/OMAOors/W0AZlHkcUD9aeGBIWDCloKa652+JlATM=",
|
||||
"h1:SSZ93MdSAaJ1Xi/VIvZDz5z1sve3BIS+WqDKACvJut0=",
|
||||
"h1:Vzj5bDkG9nOQbRMKHPhsol0+BrvzZ22lwkbLSIs4ycs=",
|
||||
"h1:Xjuo1Cwe065i1qJfhKm3dti4eeEwIT9rNNqU3R6id0A=",
|
||||
"h1:eADNOR3ZnirZXCP+3k0hy9CKQK8sgeVC0lpo08UntZY=",
|
||||
"h1:gzBWWbJc4JOwQEIINtnYkbwErRkA2oLvhtRW4lAQ7UU=",
|
||||
"h1:ipdqf/NJpSaP7em0/+n3hu0KVzwai3cIn2VgP4Rxdhs=",
|
||||
"h1:lNitoP/DTekzHnRjZ3RBLthPRZ3aHy6lR7jo2rtoH+0=",
|
||||
"h1:laqGsqlHY2/R1JQnhniKwcT0U8gb+tF0K9vr+Tovskc=",
|
||||
"h1:wUSb/6AFeFi905uIodIMA11SyyboCTtDqqNhJYdHweM=",
|
||||
"zh:32b453302a684584198a03c2e09d99ef1f6deac2fe26f8dc134d973b07fd0c23",
|
||||
"zh:3920bb891a476f30e29248533f667a507c97e93e0a2af010b242e980d6411ca6",
|
||||
"zh:630cceb40d8806945ad36e04517f313520ea5cee17bd36e26a8e999567835e1a",
|
||||
"zh:6816de7f6bd3cfe341451af3be95b0af17c5539733b165f7505e88de6b730fc5",
|
||||
"zh:729d75ab50efb675716ffa609358f6d9e80c66f7f64e01e11c8926c5293385be",
|
||||
"zh:7d1024f621fe02b3731dc65ab20105150762b29df46e398da2c73d007bb62fb2",
|
||||
"zh:81cf9cd23a70e5196b9d774482259c17e08471556932c348d3e9df0f7a482af1",
|
||||
"zh:8de75b9c4b89ed1affcdb3ce32e00afaf213cfd76e93181ea4758ffd10b6eedc",
|
||||
"zh:8f0b9b175ceb124c1cf01e6bbb660bf1bce0ff77416e026dc0679ed7156bc737",
|
||||
"zh:8fb4cde10eb346ef9e2233591d75c39bcc2cafee578b669196192e8acdb39f12",
|
||||
"zh:96798e58b8fcddc2add2da0fe8f1df9f913153d4cbccd68a7557b64633f5ea00",
|
||||
"zh:a598bd615f4465b828f3b6ed2db694d8148c9469d5497922c7bc17c08b655cf4",
|
||||
"zh:b2604d5067f3f259ff9be9bb0c4f9ec801d27c905cbf6ceafee50fc2a23f8331",
|
||||
]
|
||||
}
|
||||
|
||||
@@ -6,9 +6,5 @@ terraform {
|
||||
source = "siderolabs/talos"
|
||||
version = "0.11.0"
|
||||
}
|
||||
tailscale = {
|
||||
source = "tailscale/tailscale"
|
||||
version = "0.29.2"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,27 +1,3 @@
|
||||
resource "tailscale_tailnet_key" "controlplane" {
|
||||
for_each = var.controlplane_nodes
|
||||
|
||||
reusable = true
|
||||
ephemeral = true
|
||||
preauthorized = true
|
||||
recreate_if_invalid = "always"
|
||||
expiry = 7776000
|
||||
description = "Talos key ${each.value.name}"
|
||||
tags = [
|
||||
"tag:project-yucca",
|
||||
"tag:env-${var.env}",
|
||||
]
|
||||
}
|
||||
|
||||
data "tailscale_device" "controlplane" {
|
||||
for_each = var.controlplane_nodes
|
||||
|
||||
hostname = each.value.name
|
||||
wait_for = "300s"
|
||||
|
||||
depends_on = [talos_machine_bootstrap.this]
|
||||
}
|
||||
|
||||
data "talos_machine_configuration" "controlplane" {
|
||||
cluster_name = local.cluster_name
|
||||
cluster_endpoint = local.cluster_endpoint
|
||||
@@ -63,8 +39,8 @@ resource "talos_machine_configuration_apply" "controlplane" {
|
||||
}
|
||||
]
|
||||
# Pin the kubelet's node IP to the vRack subnet so the Kubernetes
|
||||
# InternalIP is always the private IP and never falls back to the
|
||||
# Tailscale CGNAT address (which happens if eth1 has no IP at kubelet
|
||||
# InternalIP is always the private IP and never falls back to a mesh
|
||||
# overlay address (which happens if eth1 has no IP at kubelet
|
||||
# start — see the static address below).
|
||||
kubelet = {
|
||||
nodeIP = {
|
||||
@@ -138,7 +114,7 @@ resource "talos_machine_configuration_apply" "controlplane" {
|
||||
}
|
||||
apiServer = {
|
||||
# VIP for in-cluster traffic; CP private IPs for operators reaching the
|
||||
# apiserver over Tailscale (the floating VIP doesn't ARP reliably across DCs).
|
||||
# apiserver over the mesh (the floating VIP doesn't ARP reliably across DCs).
|
||||
certSANs = concat(
|
||||
[local.controlplane_vip],
|
||||
[for k in local.controlplane_keys : var.controlplane_nodes[k].private_ip],
|
||||
@@ -155,19 +131,21 @@ resource "talos_machine_configuration_apply" "controlplane" {
|
||||
hostname: ${each.value.name}
|
||||
EOT
|
||||
,
|
||||
# Netbird overlay — the node management mesh. Operators reach the vRack IPs
|
||||
# via the Netbird route (server-side, masqueraded to a routing-peer's vRack
|
||||
# IP). NB_MANAGEMENT_URL is set explicitly to mirror yucca (Cloud default).
|
||||
<<-EOT
|
||||
name: tailscale
|
||||
name: netbird
|
||||
apiVersion: v1alpha1
|
||||
kind: ExtensionServiceConfig
|
||||
environment:
|
||||
- TS_AUTHKEY=${tailscale_tailnet_key.controlplane[each.key].key}
|
||||
- TS_HOSTNAME=${each.value.name}
|
||||
- TS_ROUTES=${var.private_network_cidr}
|
||||
- TS_EXTRA_ARGS=--accept-dns=false
|
||||
- NB_SETUP_KEY=${var.netbird_setup_key}
|
||||
- NB_MANAGEMENT_URL=https://api.netbird.io
|
||||
EOT
|
||||
,
|
||||
# Talos ingress firewall — default-deny on host-bound services. Allow
|
||||
# rules use Tailscale CGNAT (operators) and the vRack CIDR (intra-cluster).
|
||||
# Talos ingress firewall — default-deny on host-bound services. Operators reach
|
||||
# apid/apiserver via the Netbird route, masqueraded to a routing-peer's vRack IP,
|
||||
# so every allow rule is the vRack CIDR (+ pod CIDR for in-cluster metrics).
|
||||
<<-EOT
|
||||
apiVersion: v1alpha1
|
||||
kind: NetworkDefaultActionConfig
|
||||
@@ -183,8 +161,6 @@ resource "talos_machine_configuration_apply" "controlplane" {
|
||||
- 50000
|
||||
protocol: tcp
|
||||
ingress:
|
||||
- subnet: 100.64.0.0/10
|
||||
- subnet: fd7a:115c:a1e0::/48
|
||||
- subnet: ${var.private_network_cidr}
|
||||
EOT
|
||||
,
|
||||
@@ -209,8 +185,6 @@ resource "talos_machine_configuration_apply" "controlplane" {
|
||||
- 6443
|
||||
protocol: tcp
|
||||
ingress:
|
||||
- subnet: 100.64.0.0/10
|
||||
- subnet: fd7a:115c:a1e0::/48
|
||||
- subnet: ${var.private_network_cidr}
|
||||
EOT
|
||||
,
|
||||
|
||||
@@ -11,13 +11,6 @@ output "cluster" {
|
||||
}
|
||||
}
|
||||
|
||||
output "controlplane_tailscale_ips" {
|
||||
value = {
|
||||
for k, _ in var.controlplane_nodes :
|
||||
k => data.tailscale_device.controlplane[k].addresses[0]
|
||||
}
|
||||
}
|
||||
|
||||
output "talos_client_configuration" {
|
||||
sensitive = true
|
||||
value = data.talos_client_configuration.this.talos_config
|
||||
|
||||
@@ -1,7 +0,0 @@
|
||||
provider "tailscale" {
|
||||
# OAuth client from the o11y_tf vault. Must hold write on auth keys + read on
|
||||
# devices, and own the tag:project-yucca / tag:env-* tags it issues keys with.
|
||||
oauth_client_id = var.tailscale_oauth_client_id
|
||||
oauth_client_secret = var.tailscale_oauth_client_secret
|
||||
tailnet = var.tailscale_tailnet_id
|
||||
}
|
||||
@@ -63,11 +63,12 @@ dependency "ovh" {
|
||||
mock_outputs_merge_strategy_with_state = "shallow"
|
||||
}
|
||||
|
||||
dependency "tailscale" {
|
||||
config_path = "../../tailscale/account"
|
||||
dependency "netbird_cluster" {
|
||||
config_path = "../../netbird/cluster"
|
||||
mock_outputs = {
|
||||
tailscale_output = "mock-tailscale-output"
|
||||
talos_setup_key = "mock-netbird-setup-key"
|
||||
}
|
||||
mock_outputs_allowed_terraform_commands = ["init", "validate", "plan"]
|
||||
}
|
||||
|
||||
inputs = {
|
||||
@@ -78,6 +79,7 @@ inputs = {
|
||||
worker_data_disk_match = local.worker_data_disk_match
|
||||
worker_data_disk2_match = local.worker_data_disk2_match
|
||||
worker_nics = local.worker_nics
|
||||
netbird_setup_key = dependency.netbird_cluster.outputs.talos_setup_key
|
||||
}
|
||||
|
||||
generate "backend" {
|
||||
|
||||
@@ -1,13 +1,10 @@
|
||||
variable "env" {}
|
||||
variable "stage" {}
|
||||
|
||||
variable "tailscale_oauth_client_id" {
|
||||
sensitive = true
|
||||
}
|
||||
variable "tailscale_oauth_client_secret" {
|
||||
sensitive = true
|
||||
}
|
||||
variable "tailscale_tailnet_id" {
|
||||
# Netbird setup key from the netbird/cluster module (terragrunt dependency). Fed to
|
||||
# every node's netbird ExtensionServiceConfig (NB_SETUP_KEY). Takes effect once the
|
||||
# node runs a schematic that includes siderolabs/netbird.
|
||||
variable "netbird_setup_key" {
|
||||
sensitive = true
|
||||
}
|
||||
|
||||
@@ -43,7 +40,7 @@ variable "talos_installer_images" {
|
||||
|
||||
variable "talos_version" {
|
||||
type = string
|
||||
default = "v1.13.0"
|
||||
default = "v1.13.5"
|
||||
}
|
||||
|
||||
variable "controlplane_vip_offset" {
|
||||
@@ -87,9 +84,9 @@ variable "worker_nics" {
|
||||
}))
|
||||
}
|
||||
|
||||
# True only during initial bring-up of a brand-new env, before the Tailscale
|
||||
# True only during initial bring-up of a brand-new env, before the Netbird
|
||||
# extension has registered any node. Drop back to false once each node is on
|
||||
# the tailnet, so future applies go via the vRack and the ingress firewall
|
||||
# the netbird mesh, so future applies go via the vRack and the ingress firewall
|
||||
# can drop public-NIC traffic without locking terraform out.
|
||||
variable "use_public_endpoints" {
|
||||
type = bool
|
||||
|
||||
@@ -1,18 +1,3 @@
|
||||
resource "tailscale_tailnet_key" "worker" {
|
||||
for_each = var.worker_nodes
|
||||
|
||||
reusable = true
|
||||
ephemeral = true
|
||||
preauthorized = true
|
||||
recreate_if_invalid = "always"
|
||||
expiry = 7776000
|
||||
description = "Talos key ${each.value.name}"
|
||||
tags = [
|
||||
"tag:project-yucca",
|
||||
"tag:env-${var.env}",
|
||||
]
|
||||
}
|
||||
|
||||
data "talos_machine_configuration" "worker" {
|
||||
cluster_name = local.cluster_name
|
||||
cluster_endpoint = local.cluster_endpoint
|
||||
@@ -80,16 +65,16 @@ resource "talos_machine_configuration_apply" "worker" {
|
||||
hostname: ${each.value.name}
|
||||
EOT
|
||||
,
|
||||
# Tailscale extension is baked into the worker image's schematic; without
|
||||
# this config block the service starts unauthenticated and hangs.
|
||||
# Netbird overlay — the node management mesh. Operators reach the vRack IPs
|
||||
# via the Netbird route (server-side, masqueraded to a routing-peer's vRack
|
||||
# IP). NB_MANAGEMENT_URL is set explicitly to mirror yucca (Cloud default).
|
||||
<<-EOT
|
||||
name: tailscale
|
||||
name: netbird
|
||||
apiVersion: v1alpha1
|
||||
kind: ExtensionServiceConfig
|
||||
environment:
|
||||
- TS_AUTHKEY=${tailscale_tailnet_key.worker[each.key].key}
|
||||
- TS_HOSTNAME=${each.value.name}
|
||||
- TS_EXTRA_ARGS=--accept-dns=false
|
||||
- NB_SETUP_KEY=${var.netbird_setup_key}
|
||||
- NB_MANAGEMENT_URL=https://api.netbird.io
|
||||
EOT
|
||||
,
|
||||
<<-EOT
|
||||
@@ -144,8 +129,6 @@ resource "talos_machine_configuration_apply" "worker" {
|
||||
- 50000
|
||||
protocol: tcp
|
||||
ingress:
|
||||
- subnet: 100.64.0.0/10
|
||||
- subnet: fd7a:115c:a1e0::/48
|
||||
- subnet: ${var.private_network_cidr}
|
||||
EOT
|
||||
,
|
||||
|
||||
+10
-10
@@ -12,8 +12,8 @@ How to stand up an environment from nothing. The cluster is built by Terragrunt
|
||||
mise run talos:dl:cp && mise run talos:ul:cp
|
||||
```
|
||||
|
||||
This image carries the `qemu-guest-agent` + `tailscale` schematic.
|
||||
4. Workers need **no download or upload** — they are OVH BYOI and pull the bare-metal raw straight from the Talos Factory at order time. The worker schematic must stay **tailscale-only**: `qemu-guest-agent` on bare metal blocks on a virtio port that never appears and reboot-loops the node.
|
||||
This image carries the `qemu-guest-agent` + `netbird` schematic.
|
||||
4. Workers need **no download or upload** — they are OVH BYOI and pull the bare-metal raw straight from the Talos Factory at order time. The worker schematic must stay **netbird-only**: `qemu-guest-agent` on bare metal blocks on a virtio port that never appears and reboot-loops the node.
|
||||
5. For production: delete the apex DNS records via the OVH dashboard before applying.
|
||||
|
||||
## Apply order
|
||||
@@ -31,19 +31,19 @@ export TF_VAR_env=staging
|
||||
mise run tg run --working-dir deployment/modules/ovh/account apply
|
||||
```
|
||||
|
||||
2. **Tailscale** — tailnet-global ACL (only needs one run across all environments).
|
||||
2. **NetBird** — the per-environment mesh objects: the Talos node group, a reusable setup key, the vRack network route (Talos nodes as routing peers), and the `yucca → resource` access policy. The Talos module consumes the setup key from here, so apply NetBird first.
|
||||
|
||||
```bash
|
||||
mise run tg run --working-dir deployment/modules/tailscale/account apply
|
||||
mise run tg run --working-dir deployment/modules/netbird/cluster apply
|
||||
```
|
||||
|
||||
3. **Talos (bootstrap)** — initial bring-up over public IPs, because the Tailscale extension isn't running yet.
|
||||
3. **Talos (bootstrap)** — initial bring-up over public IPs, because the NetBird extension isn't running yet.
|
||||
|
||||
```bash
|
||||
TF_VAR_use_public_endpoints=true mise run tg run --working-dir deployment/modules/talos/cluster apply
|
||||
```
|
||||
|
||||
4. **Verify** the cluster is up and operator-side Tailscale routing works. Pull the configs (see [Cluster access](#cluster-access)) and hit the APIs over the tailnet:
|
||||
4. **Verify** the cluster is up and operator-side NetBird routing works. Pull the configs (see [Cluster access](#cluster-access)) and hit the APIs over the NetBird network:
|
||||
|
||||
```bash
|
||||
mise run talos:kubeconfig && mise run talos:talosconfig
|
||||
@@ -51,7 +51,7 @@ export TF_VAR_env=staging
|
||||
talosctl --talosconfig .private/$ENVIRONMENT/talosconfig -n 10.150.200.10 get members
|
||||
```
|
||||
|
||||
5. **Talos (steady state)** — drop the public-endpoints override now that Tailscale routes work; the host firewall closes the public NIC (everything except `:30443` on workers).
|
||||
5. **Talos (steady state)** — drop the public-endpoints override now that NetBird routes work; the host firewall closes the public NIC (everything except `:30443` on workers).
|
||||
|
||||
```bash
|
||||
unset TF_VAR_use_public_endpoints
|
||||
@@ -72,7 +72,7 @@ How to get `kubectl` / `talosctl` access to an **existing** cluster (no bootstra
|
||||
|
||||
**Prerequisites:**
|
||||
|
||||
* **Tailscale** — the cluster APIs are reachable only over the tailnet, so your host needs Tailscale running with subnet-route consumption enabled: `tailscale set --accept-routes` on Linux, or the "Use Tailscale subnets" toggle in the macOS app.
|
||||
* **NetBird** — the cluster APIs are reachable only over the NetBird network, so your host must be running the NetBird client (`netbird up`) and joined to the FUTO NetBird account, which places your peer in the `yucca` group. The access policy then distributes the route to the cluster's vRack CIDR, so `kubectl`/`talosctl` can reach the nodes' private IPs.
|
||||
* **1Password CLI (`op`)** — installed and signed in to the `team-futo.1password.com` account. `mise run talos:config` fetches the configs through `mise run tg`, which wraps `op run` to inject the Terraform state credentials; without an authenticated `op` it can't read state.
|
||||
|
||||
**Fetch the configs.** Two tasks pull `kubeconfig` and `talosconfig` for the environment (run whichever you need):
|
||||
@@ -84,9 +84,9 @@ mise run talos:kubeconfig # for kubectl
|
||||
mise run talos:talosconfig # for talosctl
|
||||
```
|
||||
|
||||
Each writes to `.private/$ENVIRONMENT/` (mode 600) from the Talos module's Terraform outputs. `talos:kubeconfig` also repoints the kubeconfig `server:` from the floating VIP (`10.150.200.5`) to a control-plane private IP (`10.150.200.10`) — the VIP doesn't ARP reliably across DCs over Tailscale, and every CP IP is in the apiserver cert SANs so TLS still validates.
|
||||
Each writes to `.private/$ENVIRONMENT/` (mode 600) from the Talos module's Terraform outputs. `talos:kubeconfig` also repoints the kubeconfig `server:` from the floating VIP (`10.150.200.5`) to a control-plane private IP (`10.150.200.10`) — the VIP doesn't ARP reliably across DCs over the NetBird network, and every CP IP is in the apiserver cert SANs so TLS still validates.
|
||||
|
||||
> **A highly-available operator API endpoint is TBD.** `kubectl` is pinned to a single control-plane IP, so if that CP is down you currently repoint to another by hand (any CP IP works — they're all cert SANs). The floating VIP is HA *inside* the cluster (kubelet and in-cluster clients use it) but doesn't ARP across DCs over Tailscale, so there's no HA endpoint for operators yet.
|
||||
> **A highly-available operator API endpoint is TBD.** `kubectl` is pinned to a single control-plane IP, so if that CP is down you currently repoint to another by hand (any CP IP works — they're all cert SANs). The floating VIP is HA *inside* the cluster (kubelet and in-cluster clients use it) but doesn't ARP across DCs over the NetBird network, so there's no HA endpoint for operators yet.
|
||||
|
||||
**Point your tools at them:**
|
||||
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
|
||||
The OVH foundation under the cluster: compute, private network, and public ingress. Each environment is a fully independent build of the same shape; they differ only in worker tier and IPLB zone count.
|
||||
|
||||
> **Status:** staging is built and running (`o11y-staging`). Production is planned (`o11y-production`) — same shape, larger workers, multi-zone ingress.
|
||||
> **Status:** both environments are built and running — `o11y-staging` and `o11y-production`. Same shape; production has larger workers and multi-zone ingress.
|
||||
|
||||
## Shape
|
||||
|
||||
@@ -42,11 +42,11 @@ The worker host firewall scopes `:30443` to OVH's IPLB NAT range (`10.108.0.0/14
|
||||
|
||||
Because the farm targets the workers' public IPs (not the vRack), three things are required and are handled in the cluster config: NodePorts must answer on the public NIC, exactly one Envoy must run per worker, and Envoy must parse PROXY protocol. See the cluster architecture guide for those details.
|
||||
|
||||
## Operator access (Tailscale)
|
||||
## Operator access (NetBird)
|
||||
|
||||
Tailscale runs as a Talos system extension on **every** node, so operators reach `talosctl` and `kubectl` over the tailnet without exposing those APIs publicly. Control planes advertise the private CIDR as a subnet route, auto-approved by the tailnet ACL; workers consume the routes. The ACL is environment-scoped (`tag:env-staging` vs `tag:env-production`) so staging operators can't pivot into production.
|
||||
NetBird runs as a Talos system extension on **every** node, so operators reach `talosctl` and `kubectl` over the NetBird network without exposing those APIs publicly. The vRack subnet is published as a NetBird network route with the Talos nodes as routing peers — any node can route, so it's HA — and operator traffic is masqueraded to the routing peer's vRack IP, which the host firewall already trusts. A per-environment access policy lets the shared `yucca` operator group reach this environment's routed subnet on the management ports only (apid `50000`, kube-apiserver `6443`); the groups and policy are environment-scoped (`O11Y_STAGING_*` vs `O11Y_PRODUCTION_*`), so staging operators can't pivot into production.
|
||||
|
||||
Operators point `kubectl`/`talosctl` at a specific control plane's static private IP — not the floating VIP, since cross-DC ARP for the VIP over Tailscale subnet routes is unreliable. The VIP remains the in-cluster apiserver endpoint used by kubelet and other in-cluster components.
|
||||
Operators point `kubectl`/`talosctl` at a specific control plane's static private IP — not the floating VIP, since cross-DC ARP for the VIP over the NetBird network route is unreliable. The VIP remains the in-cluster apiserver endpoint used by kubelet and other in-cluster components.
|
||||
|
||||
## Cost
|
||||
|
||||
@@ -69,5 +69,5 @@ Staging + production run-rate ≈ **$955/mo** plus the one-time **$221** product
|
||||
| Workers | 3× `SYS-2` (`24sys022`) | 3× `Rise-2` (`24rise02-v1`) |
|
||||
| IPLB | 1 zone (`gra`) | 3 zones (`gra` + `rbx` + `sbg`), anycast |
|
||||
| Private CIDR | `10.150.200.0/24` | `10.150.100.0/24` |
|
||||
| Tailscale tag | `tag:env-staging` | `tag:env-production` |
|
||||
| NetBird objects | `O11Y_STAGING_*` | `O11Y_PRODUCTION_*` |
|
||||
| Flux source | `staging` overlay | `production` overlay |
|
||||
|
||||
@@ -12,8 +12,8 @@ Control planes and workers use **different** Talos Factory schematics, on purpos
|
||||
|
||||
| Node type | Platform | Schematic |
|
||||
|-----------|----------|-----------|
|
||||
| Control plane | OVH Public Cloud (KVM) | `tailscale` + `qemu-guest-agent` |
|
||||
| Worker | Bare metal | `tailscale` only |
|
||||
| Control plane | OVH Public Cloud (KVM) | `netbird` + `qemu-guest-agent` |
|
||||
| Worker | Bare metal | `netbird` only |
|
||||
|
||||
`qemu-guest-agent` on bare metal wedges boot — it waits on a virtio-serial port that isn't present and reboot-loops the node. The control-plane image is an OpenStack image uploaded to OVH glance once; workers are BYOI and fetch their raw image from the Factory at order time.
|
||||
|
||||
@@ -36,9 +36,9 @@ Default-deny ingress on every node; anything not listed is dropped at the host.
|
||||
|
||||
| Service | Port(s) | Allowed sources |
|
||||
|---------|---------|-----------------|
|
||||
| apid | 50000/tcp | Tailscale (`100.64.0.0/10`, `fd7a:115c:a1e0::/48`), vRack |
|
||||
| trustd | 50001/tcp | Tailscale, vRack |
|
||||
| kube-apiserver (CPs) | 6443/tcp | Tailscale, vRack |
|
||||
| apid | 50000/tcp | vRack |
|
||||
| trustd | 50001/tcp | vRack |
|
||||
| kube-apiserver (CPs) | 6443/tcp | vRack |
|
||||
| etcd (CPs) | 2379–2380/tcp | vRack |
|
||||
| kubelet | 10250/tcp | vRack + pod CIDR `10.244.0.0/16` |
|
||||
| flannel VXLAN | 4789/udp | vRack |
|
||||
@@ -49,6 +49,8 @@ Default-deny ingress on every node; anything not listed is dropped at the host.
|
||||
|
||||
Pod CIDR is allowed on `kubelet` and the metrics ports because pod-to-own-node-IP traffic skips flannel masquerade (a same-node scrape keeps its pod-IP source), which the vRack-only rule would otherwise drop.
|
||||
|
||||
Operator `talosctl`/`kubectl` traffic needs no rule of its own: it arrives over the NetBird network route masqueraded to a routing peer's vRack IP, so the vRack allow on `apid` and `kube-apiserver` already covers it.
|
||||
|
||||
## Kubernetes
|
||||
|
||||
Kubernetes with flannel CNI and kube-proxy in nftables mode. Spegel runs as a peer-to-peer image registry mirror so each node's containerd pulls layers from its peers before the upstream registry (this requires `discard_unpacked_layers = false` in the worker containerd config).
|
||||
|
||||
Reference in New Issue
Block a user