feat(yucca): add full e2e mgmt provisioning maybe (#182)

* feat(yucca): add full e2e mgmt provisioning maybe

* moar !

* prefer tailscale over public ip if availbale

* ignore files

* fix

* more progress
This commit is contained in:
Antoine Lecompte
2026-06-26 14:45:20 -04:00
committed by GitHub
parent 847409819f
commit 481c5e920a
63 changed files with 1585 additions and 238 deletions
-111
View File
@@ -1,111 +0,0 @@
name: Fabric (Junos/NetBox)
# Manages the prod switch fabric with the vendored JTAF junos-qfx provider (built
# in CI) + the netbox provider. Fans out over every site under tf/deployment/prod/*:
# Plan on PRs; apply on merge to main, each site behind its own
# `prod-fabric-<site>` Environment gate (required reviewers).
# The runner joins the tailnet to reach the switch vme IPs and renders the NETCONF
# key from 1Password (the fabric:* mise tasks; FABRIC_SITE selects the stack).
#
# Prerequisites (out-of-band):
# - Repo secrets: OP_TF_YUCCA_PROD_ENV (+ _WRITE); TS_OAUTH_CLIENT_ID/SECRET.
# - 1P items: NET_SWITCHES_TERRAFORM_SSH_PRIVATE_KEY (yucca_tf_prod), NETBOX_API_TOKEN (yucca_tf).
# - One GitHub Environment per site: `prod-fabric-<site>` with required reviewers.
on:
push:
branches: [main]
paths: &fabric_paths
- "tf/providers/**"
- "tf/shared/modules/fabric-**"
- "tf/shared/modules/core-fabric/**"
- "tf/shared/modules/cluster-fabric/**"
- "tf/deployment/prod/**"
- ".github/workflows/fabric.yml"
pull_request:
paths: *fabric_paths
workflow_dispatch:
# No state locking on the OVH backend — never apply concurrently.
concurrency:
group: ${{ github.workflow }}
cancel-in-progress: false
permissions:
contents: read
jobs:
discover:
name: Discover sites
if: github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository
runs-on: ubuntu-latest
outputs:
matrix: ${{ steps.sites.outputs.matrix }}
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
persist-credentials: false
- id: sites
name: List tf/deployment/prod/* sites
run: |
matrix=$(ls -d tf/deployment/prod/*/ | xargs -n1 basename | jq -R . | jq -cs '{site: .}')
echo "matrix=$matrix" >> "$GITHUB_OUTPUT"
echo "$matrix"
plan:
name: Plan ${{ matrix.site }}
needs: discover
runs-on: ubuntu-latest
strategy:
fail-fast: false
matrix: ${{ fromJSON(needs.discover.outputs.matrix) }}
env:
OP_SERVICE_ACCOUNT_TOKEN: ${{ secrets.OP_TF_YUCCA_PROD_ENV }}
FABRIC_SITE: ${{ matrix.site }}
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
persist-credentials: false
- name: Set up mise (go + opentofu + terragrunt)
uses: jdx/mise-action@e6a8b3978addb5a52f2b4cd9d91eafa7f0ab959d # v4.2.0
- name: Install 1Password CLI
uses: 1password/install-cli-action@a5215d3a7f75c1629216c465ea9ab3ab399c4b71 # v4.0.0
- name: Connect to Tailscale
uses: tailscale/github-action@306e68a486fd2350f2bfc3b19fcd143891a4a2d8 # v4.1.2
with:
oauth-client-id: ${{ secrets.TS_OAUTH_CLIENT_ID }}
oauth-secret: ${{ secrets.TS_OAUTH_SECRET }}
tags: tag:project-yucca
- name: Fabric plan
run: mise run fabric:plan -- --non-interactive
apply:
name: Apply ${{ matrix.site }} (gated)
needs: [discover, plan]
if: github.event_name == 'push' || github.event_name == 'workflow_dispatch'
runs-on: ubuntu-latest
strategy:
fail-fast: false
matrix: ${{ fromJSON(needs.discover.outputs.matrix) }}
# Site-scoped gate: each site's prod fabric has its own required reviewers.
environment:
name: prod-fabric-${{ matrix.site }}
env:
OP_SERVICE_ACCOUNT_TOKEN: ${{ secrets.OP_TF_YUCCA_PROD_ENV_WRITE }}
FABRIC_SITE: ${{ matrix.site }}
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
persist-credentials: false
- name: Set up mise (go + opentofu + terragrunt)
uses: jdx/mise-action@e6a8b3978addb5a52f2b4cd9d91eafa7f0ab959d # v4.2.0
- name: Install 1Password CLI
uses: 1password/install-cli-action@a5215d3a7f75c1629216c465ea9ab3ab399c4b71 # v4.0.0
- name: Connect to Tailscale
uses: tailscale/github-action@306e68a486fd2350f2bfc3b19fcd143891a4a2d8 # v4.1.2
with:
oauth-client-id: ${{ secrets.TS_OAUTH_CLIENT_ID }}
oauth-secret: ${{ secrets.TS_OAUTH_SECRET }}
tags: tag:project-yucca
- name: Fabric apply
run: mise run fabric:apply -- --non-interactive -auto-approve
+243 -59
View File
@@ -1,45 +1,53 @@
name: Infra (Terraform) name: Infra (Terraform)
# Applies the staging Terraform stacks from CI, then converges the bare-metal # Applies the Terraform stacks from CI, path-scoped so each group only runs when
# Ceph cluster with Ansible: # its own files change (a `changes` job emits per-area booleans that gate the
# - tf/deployment/staging/ceph (ceph cluster 1P password items; no node contact) # rest; workflow_dispatch overrides and runs everything). Connectivity to the
# - tf/deployment/staging/talos (Talos config, Cilium, Flux bootstrap, secrets) # bare-metal/private nodes is over the NetBird overlay (Tailscale fully retired):
# - tf/deployment/staging/dns (Cloudflare records for the ingress hosts)
# - tf/deployment/staging/netbird (NetBird Cloud groups/policies/setup keys)
# - tf/deployment/prod/global + prod/htz-fsn1/netbird (prod NetBird, layered:
# a global layer + per-site layers — runs as its own gated jobs at the bottom
# of this file, on the prod 1P SA / tf/.env.prod)
# - ansible/ceph (deploy pipeline) (cephadm convergence: OSDs, RGW realm +
# S3/metrics-worker users, monitoring, tuning, hardening) — the TF stacks
# only MINT the RGW keys into 1P; this step is what actually creates the
# matching RGW users on the cluster, so the metrics worker can authenticate.
# #
# Plan runs on PRs touching tf/** or ansible/ceph/**; apply runs on merge to main # Staging (tf/deployment/staging/*): ceph, talos, dns, netbird. One staging SA,
# behind the `staging-infra` Environment gate (required reviewers). Both the Talos # a single `staging-infra` Environment gate. The netbird stack mints the CI
# stack and the Ansible deploy talk to the nodes on the 10.10.10.0/24 management # setup key; the apply joins the overlay as a `ci` peer to reach the
# VLAN, which GitHub-hosted runners can't reach — so the runner joins the NetBird # 10.10.10.0/24 nodes, then converges the bare-metal Ceph cluster (ansible/ceph).
# overlay as a `ci` peer (via the netbird-connect action + the minted CI setup #
# key) and the existing staging route advertises that VLAN. (The prod fabric # Prod fabric+mgmt (tf/deployment/prod/<site>): the switch fabric + mgmt hosts
# workflow still uses Tailscale; 10.40.5.0/24 isn't on NetBird yet.) # (junos-qfx + hetzner providers, built locally). Per-site `prod-<site>` gate.
# Reaches the switch vme / mgmt hosts over NetBird (the mgmt nodes are the
# route peers for 10.40.5.0/24 et al.); runs the `infra:*` / `mgmt:*` mise tasks.
#
# Prod NetBird (tf/deployment/prod/global + prod/<site>/netbird): account-wide +
# site NetBird groups/keys/policies/routes. Pure api.netbird.io. `prod-infra` gate.
# #
# Prerequisites (provisioned out-of-band): # Prerequisites (provisioned out-of-band):
# - Repo secrets: OP_TF_YUCCA_STAGING_ENV (the staging-scoped 1P service-account # - Repo secrets: OP_TF_YUCCA_STAGING_ENV (+ _WRITE) — staging SAs; and
# token — resolves tf/.env; dev/prod use OP_TF_YUCCA_DEV_ENV / OP_TF_YUCCA_PROD_ENV). # OP_TF_YUCCA_PROD_ENV (read) / OP_TF_YUCCA_PROD_ENV_WRITE (netbird apply) — prod
# - BOOTSTRAP — the staging NetBird stack applied ONCE out-of-band, so the CI # SAs. (The fabric apply escalates to the write SA stored in yucca_tf_prod.)
# setup key (op://yucca_tf_staging/NETBIRD_YUCCA_STAGING_CI_SETUP_KEY) exists # - BOOTSTRAP — the netbird stacks applied ONCE out-of-band so the CI/mgmt setup
# before any talos plan tries to connect. CI can't mint it itself: the apply # keys exist in 1P before anything tries to connect (CI can't mint them itself:
# that mints it is gated behind the plan that needs it. Run once locally: # the apply that mints them is gated behind the plan that needs them). E.g.:
# OP_SERVICE_ACCOUNT_TOKEN=<staging write SA> \ # OP_SERVICE_ACCOUNT_TOKEN=<staging write SA> \
# TF_STACK_DIR=tf/deployment/staging/netbird mise run tf:apply # TF_STACK_DIR=tf/deployment/staging/netbird mise run tf:apply
# Also: a NetBird route advertising 10.10.10.0/24 that the `ci` group may reach. # (staging: NETBIRD_YUCCA_STAGING_CI_SETUP_KEY; prod: NETBIRD_YUCCA_PROD_<SITE>_*).
# - GitHub Environment `staging-infra` with required reviewers (the apply gate). # Also: NetBird routes advertising the node subnets (staging 10.10.10.0/24; prod
# the site subnets via the mgmt peers) that the `ci` group may reach.
# - 1P items: NET_SWITCHES_TERRAFORM_SSH_PRIVATE_KEY (yucca_tf_prod),
# NETBOX_API_TOKEN (yucca_tf), HETZNER_WEBSERVICE_API_USER/PASSWORD (yucca_tf_prod).
# - GitHub Environments with required reviewers: `staging-infra`, `prod-infra`,
# and one `prod-<site>` per prod fabric site (e.g. prod-htz-fsn1).
on: on:
push: push:
branches: [main] branches: [main]
paths: ["tf/**", "ansible/ceph/**"] paths: &paths
- "tf/**"
- "ansible/mgmt/**"
- "ansible/ceph/**"
- ".mise/tasks/**"
- ".mise/config.toml"
- ".github/actions/netbird-connect/**"
- ".github/workflows/infra.yml"
pull_request: pull_request:
paths: ["tf/**", "ansible/ceph/**"] paths: *paths
workflow_dispatch: workflow_dispatch:
# Serialize: the OVH S3 backend has no state locking (single-operator model), # Serialize: the OVH S3 backend has no state locking (single-operator model),
@@ -51,20 +59,75 @@ concurrency:
permissions: permissions:
contents: read contents: read
env:
# read-scoped SA for plan; apply overrides with the write SA
OP_SERVICE_ACCOUNT_TOKEN: ${{ secrets.OP_TF_YUCCA_STAGING_ENV }}
jobs: jobs:
plan: # ── Which areas changed? Outputs gate every downstream job. ──────────────────
name: Plan ${{ matrix.stack }} changes:
# Skip on fork PRs (no access to secrets / the NetBird overlay). name: Detect changes
# Skip on fork PRs (no access to secrets / the overlay anyway).
if: github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository if: github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository
runs-on: ubuntu-latest runs-on: ubuntu-latest
outputs:
staging: ${{ steps.filter.outputs.staging }}
prod_tf: ${{ steps.filter.outputs.prod_tf }}
prod_ansible: ${{ steps.filter.outputs.prod_ansible }}
prod_netbird: ${{ steps.filter.outputs.prod_netbird }}
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
persist-credentials: false
- uses: dorny/paths-filter@de90cc6fb38fc0963ad72b210f1f284cd68cea36 # v3.0.2
id: filter
with:
filters: |
staging:
- 'tf/deployment/staging/**'
- 'tf/shared/**'
- 'tf/op-run.sh'
- 'ansible/ceph/**'
- '.github/actions/netbird-connect/**'
- '.github/workflows/infra.yml'
prod_tf:
# Fabric/mgmt stacks. Netbird-only dirs are covered by prod_netbird;
# an overlap just means a (gated) extra fabric plan — harmless.
- 'tf/deployment/prod/**'
- '!tf/deployment/prod/global/**'
- '!tf/deployment/prod/*/netbird/**'
- 'tf/shared/**'
- 'tf/providers/**'
- '.mise/tasks/infra/**'
- '.mise/tasks/fabric/**'
- '.mise/tasks/mgmt/**'
- '.mise/config.toml'
- '.github/actions/netbird-connect/**'
- '.github/workflows/infra.yml'
prod_ansible:
- 'ansible/mgmt/**'
- 'tf/render/**'
- 'tf/shared/modules/identity/**'
- 'tf/shared/modules/fabric-addressing/**'
- 'tf/deployment/prod/*/mgmt-hosts.yaml'
- '.mise/tasks/mgmt/**'
- '.mise/config.toml'
- '.github/actions/netbird-connect/**'
- '.github/workflows/infra.yml'
prod_netbird:
- 'tf/deployment/prod/global/**'
- 'tf/deployment/prod/*/netbird/**'
- 'tf/shared/modules/netbird-env/**'
- '.github/workflows/infra.yml'
# ── Staging stacks ──────────────────────────────────────────────────────────
staging-plan:
name: Staging plan ${{ matrix.stack }}
needs: changes
if: needs.changes.outputs.staging == 'true' || github.event_name == 'workflow_dispatch'
runs-on: ubuntu-latest
strategy: strategy:
fail-fast: false fail-fast: false
matrix: matrix:
stack: [talos, dns, ceph, netbird] stack: [talos, dns, ceph, netbird]
env:
OP_SERVICE_ACCOUNT_TOKEN: ${{ secrets.OP_TF_YUCCA_STAGING_ENV }}
steps: steps:
- name: Checkout - name: Checkout
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
@@ -98,11 +161,11 @@ jobs:
--working-dir tf/deployment/staging/${{ matrix.stack }} --working-dir tf/deployment/staging/${{ matrix.stack }}
--non-interactive plan --non-interactive plan
apply: staging-apply:
name: Apply (gated) name: Staging apply (gated)
needs: plan needs: [changes, staging-plan]
# Only on merge to main (or manual dispatch) — never on PRs. # Only on merge to main (or manual dispatch) — never on PRs.
if: github.event_name == 'push' || github.event_name == 'workflow_dispatch' if: (github.event_name == 'push' && needs.changes.outputs.staging == 'true') || github.event_name == 'workflow_dispatch'
runs-on: ubuntu-latest runs-on: ubuntu-latest
# The approval gate: `staging-infra` Environment with required reviewers. # The approval gate: `staging-infra` Environment with required reviewers.
environment: staging-infra environment: staging-infra
@@ -132,7 +195,7 @@ jobs:
# Join the NetBird overlay as a `ci` peer so the node-touching stacks and # Join the NetBird overlay as a `ci` peer so the node-touching stacks and
# the Ansible deploy below reach 10.10.10.0/24 (the staging route advertises # the Ansible deploy below reach 10.10.10.0/24 (the staging route advertises
# the LAN). Replaces the old Tailscale subnet-router path. # the LAN).
- name: Connect to NetBird - name: Connect to NetBird
uses: ./.github/actions/netbird-connect uses: ./.github/actions/netbird-connect
with: with:
@@ -165,7 +228,6 @@ jobs:
# cluster. Runs after the TF apply so the inventory (rendered from the # cluster. Runs after the TF apply so the inventory (rendered from the
# ceph stack's `render` output) and the keys exist. Reuses the NetBird # ceph stack's `render` output) and the keys exist. Reuses the NetBird
# overlay + 1Password session already established in this job. # overlay + 1Password session already established in this job.
- name: Render Ansible inventory from the ceph TF state - name: Render Ansible inventory from the ceph TF state
run: ansible/ceph/scripts/render-inventories.sh staging run: ansible/ceph/scripts/render-inventories.sh staging
@@ -192,22 +254,16 @@ jobs:
CEPH_ENV: inventories/sietch-ceph.staging.austin.int/inventory.ini CEPH_ENV: inventories/sietch-ceph.staging.austin.int/inventory.ini
run: mise run deploy run: mise run deploy
# ── prod NetBird (global + site layers) ────────────────────────────────────── # ── Prod NetBird (global + site layers) — mints the CI/mgmt setup keys ────────
# Prod is its own env (separate 1P SA + env file), so it can't ride the # Pure api.netbird.io (no nodes/overlay). Layered: prod/global (account-wide
# staging matrix above (that job's OP_SERVICE_ACCOUNT_TOKEN is the staging SA). # groups + the yucca→yucca_resource policy) then prod/<site>/netbird (site
# NetBird is pure api.netbird.io — no nodes, no tailnet — so these are the only # groups/keys/policies + the routed network). Runs before the fabric/mgmt jobs
# prod TF stacks CI touches today. Layered: prod/global (account-wide groups + # conceptually (it mints the keys they consume), but they're only loosely coupled
# operator setup keys) then prod/<site>/netbird (site groups/keys/policies that # — the keys persist in 1P across runs, so a fresh bootstrap applies this first.
# reference the global groups). Both select tf/.env.prod via OP_ENV_FILE.
#
# Additional prerequisites (out-of-band) beyond the staging ones above:
# - Repo secrets OP_TF_YUCCA_PROD_ENV (read) / OP_TF_YUCCA_PROD_ENV_WRITE
# (apply) — a prod-scoped 1P service account granted shared_tf (read, for
# NETBIRD_TF_PAT) + yucca_tf_prod (read/write, for the minted setup keys).
# - GitHub Environment `prod-infra` with required reviewers (the apply gate).
netbird-prod-plan: netbird-prod-plan:
name: Plan prod/netbird name: Plan prod/netbird
if: github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository needs: changes
if: needs.changes.outputs.prod_netbird == 'true' || github.event_name == 'workflow_dispatch'
runs-on: ubuntu-latest runs-on: ubuntu-latest
env: env:
OP_ENV_FILE: tf/.env.prod OP_ENV_FILE: tf/.env.prod
@@ -241,8 +297,8 @@ jobs:
netbird-prod-apply: netbird-prod-apply:
name: Apply prod/netbird (gated) name: Apply prod/netbird (gated)
needs: netbird-prod-plan needs: [changes, netbird-prod-plan]
if: github.event_name == 'push' || github.event_name == 'workflow_dispatch' if: (github.event_name == 'push' && needs.changes.outputs.prod_netbird == 'true') || github.event_name == 'workflow_dispatch'
runs-on: ubuntu-latest runs-on: ubuntu-latest
environment: prod-infra environment: prod-infra
env: env:
@@ -273,3 +329,131 @@ jobs:
tf/op-run.sh terragrunt tf/op-run.sh terragrunt
--working-dir tf/deployment/prod/htz-fsn1/netbird --working-dir tf/deployment/prod/htz-fsn1/netbird
--non-interactive apply -auto-approve --non-interactive apply -auto-approve
# ── Prod fabric + mgmt stacks (one per site) ─────────────────────────────────
prod-discover:
name: Discover prod fabric sites
needs: changes
# Needed by both the TF and ansible prod jobs, so run if either area changed.
if: needs.changes.outputs.prod_tf == 'true' || needs.changes.outputs.prod_ansible == 'true' || github.event_name == 'workflow_dispatch'
runs-on: ubuntu-latest
outputs:
matrix: ${{ steps.sites.outputs.matrix }}
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
persist-credentials: false
- id: sites
name: List fabric sites (dirs with a fabric.tf; excludes netbird-only dirs)
run: |
matrix=$(for d in tf/deployment/prod/*/; do [ -f "${d}fabric.tf" ] && basename "$d"; done | jq -R . | jq -cs '{site: .}')
echo "matrix=$matrix" >> "$GITHUB_OUTPUT"
echo "$matrix"
prod-plan:
name: Prod plan ${{ matrix.site }}
needs: [changes, prod-discover]
if: needs.changes.outputs.prod_tf == 'true' || github.event_name == 'workflow_dispatch'
runs-on: ubuntu-latest
strategy:
fail-fast: false
matrix: ${{ fromJSON(needs.prod-discover.outputs.matrix) }}
env:
OP_SERVICE_ACCOUNT_TOKEN: ${{ secrets.OP_TF_YUCCA_PROD_ENV }}
SITE: ${{ matrix.site }}
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
persist-credentials: false
- name: Set up mise (go + opentofu + terragrunt)
uses: jdx/mise-action@e6a8b3978addb5a52f2b4cd9d91eafa7f0ab959d # v4.2.0
- name: Install 1Password CLI
uses: 1password/install-cli-action@a5215d3a7f75c1629216c465ea9ab3ab399c4b71 # v4.0.0
# Reach the switch vme (routed via the mgmt NetBird peers) over the overlay.
- name: Resolve NetBird CI setup-key ref
run: echo "NB_CI_KEY_REF=op://yucca_tf_prod/NETBIRD_YUCCA_PROD_$(echo "$SITE" | tr 'a-z-' 'A-Z_')_CI_SETUP_KEY/password" >> "$GITHUB_ENV"
- name: Connect to NetBird
uses: ./.github/actions/netbird-connect
with:
setup-key-ref: ${{ env.NB_CI_KEY_REF }}
hostname: gha-prod-plan-${{ matrix.site }}-${{ github.run_id }}
- name: Deploy plan
run: mise run infra:plan -- --non-interactive
prod-apply:
name: Prod apply ${{ matrix.site }} (gated)
needs: [changes, prod-discover, prod-plan]
if: (github.event_name == 'push' && needs.changes.outputs.prod_tf == 'true') || github.event_name == 'workflow_dispatch'
runs-on: ubuntu-latest
strategy:
fail-fast: false
matrix: ${{ fromJSON(needs.prod-discover.outputs.matrix) }}
# Site-scoped gate: each site's prod stack has its own required reviewers.
environment:
name: prod-${{ matrix.site }}
env:
# Read-scoped token; infra:apply escalates to the write SA stored in the vault.
OP_SERVICE_ACCOUNT_TOKEN: ${{ secrets.OP_TF_YUCCA_PROD_ENV }}
SITE: ${{ matrix.site }}
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
persist-credentials: false
- name: Set up mise (go + opentofu + terragrunt)
uses: jdx/mise-action@e6a8b3978addb5a52f2b4cd9d91eafa7f0ab959d # v4.2.0
- name: Install 1Password CLI
uses: 1password/install-cli-action@a5215d3a7f75c1629216c465ea9ab3ab399c4b71 # v4.0.0
- name: Resolve NetBird CI setup-key ref
run: echo "NB_CI_KEY_REF=op://yucca_tf_prod/NETBIRD_YUCCA_PROD_$(echo "$SITE" | tr 'a-z-' 'A-Z_')_CI_SETUP_KEY/password" >> "$GITHUB_ENV"
- name: Connect to NetBird
uses: ./.github/actions/netbird-connect
with:
setup-key-ref: ${{ env.NB_CI_KEY_REF }}
hostname: gha-prod-apply-${{ matrix.site }}-${{ github.run_id }}
- name: Deploy apply
run: mise run infra:apply -- --non-interactive -auto-approve
prod-ansible:
name: Prod ansible ${{ matrix.site }}
needs: [changes, prod-discover, prod-apply]
# Runs after the TF apply, but also on ansible-only changes (apply skipped).
# always() so a skipped prod-apply (TF unchanged) doesn't skip this; still
# bails if discover failed or the apply actually failed.
if: >-
always()
&& needs.prod-discover.result == 'success'
&& needs.prod-apply.result != 'failure'
&& needs.prod-apply.result != 'cancelled'
&& ((github.event_name == 'push' && needs.changes.outputs.prod_ansible == 'true') || github.event_name == 'workflow_dispatch')
runs-on: ubuntu-latest
strategy:
fail-fast: false
matrix: ${{ fromJSON(needs.prod-discover.outputs.matrix) }}
# No environment gate: the prod-apply gate already approved this deploy, and
# ansible/mgmt only reads from 1Password (read-scoped repo secret).
env:
OP_SERVICE_ACCOUNT_TOKEN: ${{ secrets.OP_TF_YUCCA_PROD_ENV }}
SITE: ${{ matrix.site }}
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
persist-credentials: false
- name: Set up mise (ansible + opentofu)
uses: jdx/mise-action@e6a8b3978addb5a52f2b4cd9d91eafa7f0ab959d # v4.2.0
- name: Install 1Password CLI
uses: 1password/install-cli-action@a5215d3a7f75c1629216c465ea9ab3ab399c4b71 # v4.0.0
# On the overlay so the playbook can reach mgmt hosts that have joined NetBird
# (it prefers their NetBird IP, falling back to the public IP otherwise).
- name: Resolve NetBird CI setup-key ref
run: echo "NB_CI_KEY_REF=op://yucca_tf_prod/NETBIRD_YUCCA_PROD_$(echo "$SITE" | tr 'a-z-' 'A-Z_')_CI_SETUP_KEY/password" >> "$GITHUB_ENV"
- name: Connect to NetBird
uses: ./.github/actions/netbird-connect
with:
setup-key-ref: ${{ env.NB_CI_KEY_REF }}
hostname: gha-prod-ansible-${{ matrix.site }}-${{ github.run_id }}
# Renders the inventory from TF (tf/render/ansible-mgmt) then converges the
# mgmt hosts — over NetBird if they've joined, else their public IP (the
# TF-generated provisioning key authorizes root). No-op for sites without an
# ansible/mgmt inventory; requires the hosts to have been reprovisioned.
- name: Ansible converge (mgmt hosts)
run: mise run mgmt:ansible
+15 -4
View File
@@ -55,9 +55,20 @@ ansible/*/.venv/
# on operator workstation, but not tracked). # on operator workstation, but not tracked).
analysis/ analysis/
# fabric (Junos/JTAF) terraform — generated artifacts # prod deployment-stack terraform — locally-built providers + generated artifacts
tf/.terraformrc.fabric tf/.terraformrc.local
.mise/.fabric-provider-bin/ .mise/.provider-bin/
.mise/.fabric-provider-mirror/ .mise/.provider-mirror/
.mise/.hetzner-provider-src/
# self-built junos-qfx (mirror) makes this lock platform-specific; regenerated by init # self-built junos-qfx (mirror) makes this lock platform-specific; regenerated by init
tf/deployment/prod/htz-fsn1/.terraform.lock.hcl tf/deployment/prod/htz-fsn1/.terraform.lock.hcl
# ansible/mgmt inventory is TF-generated at run time (tf/render/ansible-mgmt) —
# Terraform is the source of truth, not these files.
ansible/mgmt/inventories/*/hosts.yml
ansible/mgmt/inventories/*/host_vars/
ansible/mgmt/inventories/*/group_vars/all/users.generated.yml
# ephemeral local state for the render roots
tf/render/*/.terraform/
tf/render/*/.terraform.lock.hcl
tf/render/*/terraform.tfstate*
+2
View File
@@ -24,6 +24,8 @@ uv = "0.9.18"
# Infrastructure tooling (added for tf/ and ansible/ subtrees) # Infrastructure tooling (added for tf/ and ansible/ subtrees)
opentofu = "1.11.5" opentofu = "1.11.5"
terragrunt = "0.99.4" terragrunt = "0.99.4"
# ansible/mgmt convergence (mgmt:ansible task, run from CI on prod apply).
"pipx:ansible-core" = "2.18.1"
[tasks.dev] [tasks.dev]
description = "Start all services in development mode" description = "Start all services in development mode"
-17
View File
@@ -1,17 +0,0 @@
#!/usr/bin/env bash
#MISE description="terragrunt apply for prod/htz-fsn1 fabric (builds provider, renders creds from 1Password)"
set -euo pipefail
ROOT=$(git rev-parse --show-toplevel)
mise run fabric:provider-build
KEYF=$(mktemp); chmod 600 "$KEYF"; trap 'rm -f "$KEYF"' EXIT
ACCT=(); [ -z "${OP_SERVICE_ACCOUNT_TOKEN:-}" ] && ACCT=(--account "${OP_ACCOUNT:-team-futo}")
op read "${ACCT[@]}" "op://yucca_tf_prod/NET_SWITCHES_TERRAFORM_SSH_PRIVATE_KEY/password" > "$KEYF"
export TF_VAR_netconf_ssh_key_path="$KEYF"
export TF_CLI_CONFIG_FILE="$ROOT/tf/.terraformrc.fabric"
SITE="${FABRIC_SITE:-htz-fsn1}"
# -parallelism=1: the JTAF junos-qfx provider isn't concurrency-safe — parallel
# ApplyResourceChange calls across the VCs crash the plugin ("Plugin did not respond").
OP_ENV_FILE=tf/.env.prod "$ROOT/tf/op-run.sh" terragrunt --working-dir "tf/deployment/prod/$SITE" apply -parallelism=1 "$@"
-18
View File
@@ -1,18 +0,0 @@
#!/usr/bin/env bash
#MISE description="terragrunt plan for prod/htz-fsn1 fabric (builds provider, renders creds from 1Password)"
set -euo pipefail
ROOT=$(git rev-parse --show-toplevel)
mise run fabric:provider-build
# Render the NETCONF SSH key from 1Password to a 0600 temp file (op run can't
# write files). Use --account only for interactive logins; CI uses an SA token.
KEYF=$(mktemp); chmod 600 "$KEYF"; trap 'rm -f "$KEYF"' EXIT
ACCT=(); [ -z "${OP_SERVICE_ACCOUNT_TOKEN:-}" ] && ACCT=(--account "${OP_ACCOUNT:-team-futo}")
op read "${ACCT[@]}" "op://yucca_tf_prod/NET_SWITCHES_TERRAFORM_SSH_PRIVATE_KEY/password" > "$KEYF"
export TF_VAR_netconf_ssh_key_path="$KEYF"
export TF_CLI_CONFIG_FILE="$ROOT/tf/.terraformrc.fabric"
SITE="${FABRIC_SITE:-htz-fsn1}"
# -parallelism=1: the JTAF junos-qfx provider isn't concurrency-safe (see apply).
OP_ENV_FILE=tf/.env.prod "$ROOT/tf/op-run.sh" terragrunt --working-dir "tf/deployment/prod/$SITE" plan -parallelism=1 "$@"
+4 -17
View File
@@ -1,33 +1,20 @@
#!/usr/bin/env bash #!/usr/bin/env bash
#MISE description="Build the vendored JTAF junos-qfx provider into a filesystem mirror + write its terraformrc" #MISE description="Build the vendored JTAF junos-qfx provider binary into the shared provider mirror"
set -euo pipefail set -euo pipefail
ROOT=$(git rev-parse --show-toplevel) ROOT=$(git rev-parse --show-toplevel)
SRC="$ROOT/tf/providers/terraform-provider-junos-qfx" SRC="$ROOT/tf/providers/terraform-provider-junos-qfx"
MIRROR="${FABRIC_PROVIDER_MIRROR:-$ROOT/.mise/.fabric-provider-mirror}" MIRROR="${PROVIDER_MIRROR:-$ROOT/.mise/.provider-mirror}"
VER="${FABRIC_PROVIDER_VERSION:-0.0.1}" VER="${JUNOS_PROVIDER_VERSION:-0.0.1}"
GOOS=$(go env GOOS); GOARCH=$(go env GOARCH) GOOS=$(go env GOOS); GOARCH=$(go env GOARCH)
BIN="terraform-provider-junos-qfx_v${VER}" BIN="terraform-provider-junos-qfx_v${VER}"
# Build into the unpacked filesystem-mirror layout for both registry hosts # Build into the unpacked filesystem-mirror layout for both registry hosts
# (OpenTofu defaults to registry.opentofu.org; Terraform to registry.terraform.io). # (OpenTofu defaults to registry.opentofu.org; Terraform to registry.terraform.io).
# The shared terraformrc that wires this mirror up is written by `infra:providers`.
for HOST in registry.opentofu.org registry.terraform.io; do for HOST in registry.opentofu.org registry.terraform.io; do
DEST="$MIRROR/$HOST/hashicorp/junos-qfx/$VER/${GOOS}_${GOARCH}" DEST="$MIRROR/$HOST/hashicorp/junos-qfx/$VER/${GOOS}_${GOARCH}"
mkdir -p "$DEST" mkdir -p "$DEST"
( cd "$SRC" && go build -o "$DEST/$BIN" . ) ( cd "$SRC" && go build -o "$DEST/$BIN" . )
done done
cat > "$ROOT/tf/.terraformrc.fabric" <<EOF
# Generated by 'mise run fabric:provider-build' — do not edit.
provider_installation {
filesystem_mirror {
path = "$MIRROR"
include = ["registry.opentofu.org/hashicorp/junos-qfx", "registry.terraform.io/hashicorp/junos-qfx"]
}
direct {
exclude = ["registry.opentofu.org/hashicorp/junos-qfx", "registry.terraform.io/hashicorp/junos-qfx"]
}
}
EOF
echo "built junos-qfx v$VER (${GOOS}_${GOARCH}) -> $MIRROR" echo "built junos-qfx v$VER (${GOOS}_${GOARCH}) -> $MIRROR"
echo "wrote filesystem_mirror -> tf/.terraformrc.fabric"
+38
View File
@@ -0,0 +1,38 @@
#!/usr/bin/env bash
#MISE description="terragrunt apply for a prod deployment stack (builds providers, escalates to the write SA, renders creds from 1Password). SITE selects the stack."
set -euo pipefail
ROOT=$(git rev-parse --show-toplevel)
# OpenTofu's Go binary doesn't read the macOS keychain, so the OVH S3 backend's
# cert fails to verify locally ("unknown authority"). Point it at the system CA
# bundle on macOS; Linux/CI reads its trust store fine and is left alone.
if [ -z "${SSL_CERT_FILE:-}" ] && [ "$(uname -s)" = "Darwin" ] && [ -f /etc/ssl/cert.pem ]; then
export SSL_CERT_FILE=/etc/ssl/cert.pem AWS_CA_BUNDLE=/etc/ssl/cert.pem
fi
mise run infra:providers
# Escalate to the write-capable service-account token, which is itself stored in
# the vault: CI hands us a read-scoped token in OP_SERVICE_ACCOUNT_TOKEN (locally
# we sign in to team-futo), and we use it to read the write SA, then run the apply
# as that SA so the onepassword provider can create/update items. One GitHub
# secret (the read token) is enough; the write privilege lives in 1Password.
ACCT=(); [ -z "${OP_SERVICE_ACCOUNT_TOKEN:-}" ] && ACCT=(--account "${OP_ACCOUNT:-team-futo}")
OP_SERVICE_ACCOUNT_TOKEN=$(op read "${ACCT[@]}" \
"op://yucca_tf_prod/yucca_futo_1pass_service_account_write/password")
export OP_SERVICE_ACCOUNT_TOKEN
# Render the NETCONF SSH key from 1Password to a 0600 temp file (op run can't
# write files) — now via the escalated SA.
KEYF=$(mktemp); chmod 600 "$KEYF"; trap 'rm -f "$KEYF"' EXIT
op read "op://yucca_tf_prod/NET_SWITCHES_TERRAFORM_SSH_PRIVATE_KEY/password" > "$KEYF"
export TF_VAR_netconf_ssh_key_path="$KEYF"
export TF_CLI_CONFIG_FILE="$ROOT/tf/.terraformrc.local"
SITE="${SITE:-htz-fsn1}"
# With a service-account token in the env, drop OP_ACCOUNT — the onepassword
# provider rejects having both set ("service_account_token and account are set").
unset OP_ACCOUNT
# -parallelism=1: the JTAF junos-qfx provider isn't concurrency-safe — parallel
# ApplyResourceChange calls across the VCs crash the plugin ("Plugin did not respond").
OP_ENV_FILE=tf/.env.prod "$ROOT/tf/op-run.sh" terragrunt --working-dir "tf/deployment/prod/$SITE" apply -parallelism=1 "$@"
+37
View File
@@ -0,0 +1,37 @@
#!/usr/bin/env bash
#MISE description="terragrunt plan for a prod deployment stack (builds providers, renders creds from 1Password). SITE selects the stack."
set -euo pipefail
ROOT=$(git rev-parse --show-toplevel)
# OpenTofu's Go binary doesn't read the macOS keychain, so the OVH S3 backend's
# cert fails to verify locally ("unknown authority"). Point it at the system CA
# bundle on macOS; Linux/CI reads its trust store fine and is left alone.
if [ -z "${SSL_CERT_FILE:-}" ] && [ "$(uname -s)" = "Darwin" ] && [ -f /etc/ssl/cert.pem ]; then
export SSL_CERT_FILE=/etc/ssl/cert.pem AWS_CA_BUNDLE=/etc/ssl/cert.pem
fi
mise run infra:providers
# Resolve a (read-scoped) service-account token. CI provides one in
# OP_SERVICE_ACCOUNT_TOKEN; locally we read it from the vault via an interactive
# team-futo sign-in. Exporting it makes every downstream `op`/`op run` use that
# SA (and the right account) — plan stays read-only, so no write escalation.
if [ -z "${OP_SERVICE_ACCOUNT_TOKEN:-}" ]; then
OP_SERVICE_ACCOUNT_TOKEN=$(op read --account "${OP_ACCOUNT:-team-futo}" \
"op://yucca_tf_prod/yucca_futo_1pass_service_account/password")
export OP_SERVICE_ACCOUNT_TOKEN
fi
# Render the NETCONF SSH key from 1Password to a 0600 temp file (op run can't
# write files).
KEYF=$(mktemp); chmod 600 "$KEYF"; trap 'rm -f "$KEYF"' EXIT
op read "op://yucca_tf_prod/NET_SWITCHES_TERRAFORM_SSH_PRIVATE_KEY/password" > "$KEYF"
export TF_VAR_netconf_ssh_key_path="$KEYF"
export TF_CLI_CONFIG_FILE="$ROOT/tf/.terraformrc.local"
SITE="${SITE:-htz-fsn1}"
# With a service-account token in the env, drop OP_ACCOUNT — the onepassword
# provider rejects having both set ("service_account_token and account are set").
unset OP_ACCOUNT
# -parallelism=1: the JTAF junos-qfx provider isn't concurrency-safe (see apply).
OP_ENV_FILE=tf/.env.prod "$ROOT/tf/op-run.sh" terragrunt --working-dir "tf/deployment/prod/$SITE" plan -parallelism=1 "$@"
+34
View File
@@ -0,0 +1,34 @@
#!/usr/bin/env bash
#MISE description="Build all of the deployment stack's local providers (junos-qfx + hetzner) into a filesystem mirror + write its terraformrc"
set -euo pipefail
ROOT=$(git rev-parse --show-toplevel)
MIRROR="${PROVIDER_MIRROR:-$ROOT/.mise/.provider-mirror}"
# Per-provider component builders (each builds one binary into the mirror).
mise run fabric:provider-build # junos-qfx (switch fabric)
mise run mgmt:provider-build # hetzner (mgmt-host reprovisioning)
# A single terraformrc points TF/tofu at the mirror for every locally-built
# provider, for both registry hosts (OpenTofu -> registry.opentofu.org,
# Terraform -> registry.terraform.io). Consumed via TF_CLI_CONFIG_FILE.
cat > "$ROOT/tf/.terraformrc.local" <<'EOF'
# Generated by 'mise run infra:providers' — do not edit.
provider_installation {
filesystem_mirror {
path = "__MIRROR__"
include = [
"registry.opentofu.org/hashicorp/junos-qfx", "registry.terraform.io/hashicorp/junos-qfx",
"registry.opentofu.org/zack/hetzner", "registry.terraform.io/zack/hetzner",
]
}
direct {
exclude = [
"registry.opentofu.org/hashicorp/junos-qfx", "registry.terraform.io/hashicorp/junos-qfx",
"registry.opentofu.org/zack/hetzner", "registry.terraform.io/zack/hetzner",
]
}
}
EOF
# shellcheck disable=SC2016
sed -i.bak "s#__MIRROR__#${MIRROR}#" "$ROOT/tf/.terraformrc.local" && rm -f "$ROOT/tf/.terraformrc.local.bak"
echo "wrote filesystem_mirror -> tf/.terraformrc.local"
+36
View File
@@ -0,0 +1,36 @@
#!/usr/bin/env bash
#MISE description="Converge a site's management hosts with ansible/mgmt (root via the TF provisioning key from 1Password). SITE selects the inventory."
set -euo pipefail
ROOT=$(git rev-parse --show-toplevel)
SITE="${SITE:-htz-fsn1}"
INV="$ROOT/ansible/mgmt/inventories/$SITE"
# Per-site: only run where an inventory exists (keeps the prod matrix happy).
if [ ! -d "$INV" ]; then
echo "mgmt:ansible: no ansible/mgmt inventory for site '$SITE' — skipping."
exit 0
fi
# Generate the inventory (hosts, host_vars, users) from Terraform's SoT.
mise run mgmt:render-inventory
# op creds: CI provides a (read-scoped) SA token in OP_SERVICE_ACCOUNT_TOKEN;
# locally we sign in to team-futo. Reads only — no write escalation needed.
ACCT=(); [ -z "${OP_SERVICE_ACCOUNT_TOKEN:-}" ] && ACCT=(--account "${OP_ACCOUNT:-team-futo}")
# Provisioning private key (root login) -> 0600 temp file. Item name is
# site-derived: htz-fsn1 -> HTZ_FSN1_PROVISIONING_SSH_PRIVATE_KEY (set by mgmt.tf).
KEY_ITEM="$(printf '%s' "$SITE" | tr 'a-z-' 'A-Z_')_PROVISIONING_SSH_PRIVATE_KEY"
KEYF=$(mktemp); chmod 600 "$KEYF"; trap 'rm -f "$KEYF"' EXIT
op read "${ACCT[@]}" "op://yucca_tf_prod/$KEY_ITEM/password" > "$KEYF"
# NetBird "mgmt" setup key (auto_groups=["mgmt"]) — joins the node to the overlay
# as a route peer. Site-derived item: htz-fsn1 -> NETBIRD_YUCCA_PROD_HTZ_FSN1_MGMT_SETUP_KEY.
NB_KEY_ITEM="NETBIRD_YUCCA_PROD_$(printf '%s' "$SITE" | tr 'a-z-' 'A-Z_')_MGMT_SETUP_KEY"
NB_SETUP_KEY=$(op read "${ACCT[@]}" "op://yucca_tf_prod/$NB_KEY_ITEM/password")
cd "$ROOT/ansible/mgmt"
ansible-galaxy collection install -r requirements.yml >/dev/null
ansible-playbook -i "inventories/$SITE" site.yml \
--private-key "$KEYF" \
--extra-vars "mgmt_netbird_setup_key=$NB_SETUP_KEY" "$@"
+28
View File
@@ -0,0 +1,28 @@
#!/usr/bin/env bash
#MISE description="Clone + build the Hetzner robot provider (zack/hetzner) into the shared provider mirror"
set -euo pipefail
ROOT=$(git rev-parse --show-toplevel)
MIRROR="${PROVIDER_MIRROR:-$ROOT/.mise/.provider-mirror}"
SRC="${HETZNER_PROVIDER_SRC:-$ROOT/.mise/.hetzner-provider-src}"
REPO="${HETZNER_PROVIDER_REPO:-https://github.com/zackpollard/terraform-provider-hetzner.git}"
REF="${HETZNER_PROVIDER_REF:-v0.1.0}"
VER="${HETZNER_PROVIDER_VERSION:-0.1.0}"
# Pinned shallow clone of the upstream provider (not vendored — it's stock).
if [ ! -d "$SRC/.git" ]; then
git clone --depth 1 --branch "$REF" "$REPO" "$SRC"
else
( cd "$SRC" && git fetch --depth 1 origin "$REF" && git checkout -q FETCH_HEAD )
fi
# Build into the unpacked filesystem-mirror layout for both registry hosts
# (OpenTofu defaults to registry.opentofu.org; Terraform to registry.terraform.io).
# Provider address is registry.<host>/zack/hetzner (see its main.go).
GOOS=$(go env GOOS); GOARCH=$(go env GOARCH)
BIN="terraform-provider-hetzner_v${VER}"
for HOST in registry.opentofu.org registry.terraform.io; do
DEST="$MIRROR/$HOST/zack/hetzner/$VER/${GOOS}_${GOARCH}"
mkdir -p "$DEST"
( cd "$SRC" && go build -o "$DEST/$BIN" . )
done
echo "built hetzner v$VER (${GOOS}_${GOARCH}) -> $MIRROR"
+10
View File
@@ -0,0 +1,10 @@
#!/usr/bin/env bash
#MISE description="Render a site's ansible/mgmt inventory from Terraform (tf/render/ansible-mgmt: addressing + identity + mgmt-hosts.yaml). SITE selects the site."
set -euo pipefail
ROOT=$(git rev-parse --show-toplevel)
SITE="${SITE:-htz-fsn1}"
cd "$ROOT/tf/render/ansible-mgmt"
tofu init -input=false >/dev/null
tofu apply -input=false -auto-approve -var "site=$SITE" >/dev/null
echo "rendered ansible/mgmt/inventories/$SITE/ (hosts.yml, host_vars/, group_vars/all/users.generated.yml)"
+6
View File
@@ -0,0 +1,6 @@
---
# All roles share the mgmt_ variable prefix for consistency across the
# stack. The var-naming[no-role-prefix] rule expects each role to use its
# own prefix, which doesn't fit our single-product layout.
skip_list:
- var-naming[no-role-prefix]
+29
View File
@@ -0,0 +1,29 @@
root = true
[*]
end_of_line = lf
insert_final_newline = true
trim_trailing_whitespace = true
charset = utf-8
[*.{yml,yaml}]
indent_style = space
indent_size = 2
[*.{j2,jinja2}]
indent_style = space
indent_size = 2
[*.py]
indent_style = space
indent_size = 4
[*.{sh,bash}]
indent_style = space
indent_size = 2
[Makefile]
indent_style = tab
[*.md]
trim_trailing_whitespace = false
+20
View File
@@ -0,0 +1,20 @@
# Ansible
ansible.log
ansible.*.log
*.retry
.ansible/
.ansible_facts_cache/
# Operator-local host_vars overrides
inventories/*/host_vars/*.local.yml
# Python virtualenv and bytecode
.venv/
__pycache__/
*.pyc
# OS / editor
.DS_Store
*.swp
*.swo
*~
+12
View File
@@ -0,0 +1,12 @@
---
extends: relaxed
rules:
line-length:
max: 260
allow-non-breakable-inline-mappings: true
comments:
min-spaces-from-content: 1
octal-values:
forbid-implicit-octal: true
forbid-explicit-octal: true
+143
View File
@@ -0,0 +1,143 @@
# mgmt — Hetzner FSN1 Management Hosts
Ansible automation that configures the two Hetzner management hosts after they
are reprovisioned to Debian 13 ("trixie"). Mirrors the conventions of the
sibling `ansible/ceph/` tree (layout, `ansible.cfg`, role structure,
systemd-networkd templating).
| Host | Public IP | Role |
|------|-----------|------|
| `htz-fsn-mgmt-1` | `178.63.124.40` | NetBird route peer (routes `10.40.5.0/24` et al.) |
| `htz-fsn-mgmt-2` | `178.63.124.41` | NetBird route peer |
Both are AX41-NVMe, both join the NetBird `mgmt` group and route the site
subnets (the routed network is declared in TF, `tf/deployment/prod/htz-fsn1/netbird`).
The inventory is **TF-generated** at run time (see "Generated inventory").
## What it does
`site.yml` applies these roles in order:
1. **baseline** — apt cache + base packages, timezone UTC, hostname from
inventory, `/etc/hosts`.
2. **users** — login users for the members of the identity registry's
`server`-mapped groups (`nutgood`, `andy` — both in `server_admins`,
`sudo = ALL`). The `mgmt_users` list is **TF-generated** from
`tf/shared/modules/identity` (see "Generated inventory" below).
3. **security** — nftables firewall, SSH hardening (no password auth,
`PermitRootLogin prohibit-password`), unattended-upgrades.
4. **networkd** — systemd-networkd VLAN sub-interfaces on the 25G fabric NIC.
**Gated on `mgmt_networkd_enabled` (default false)** — see the 25G caveat
below.
5. **netbird** — install NetBird, `netbird up` with the `mgmt` setup key, enable
IP forwarding (the mgmt nodes are the NetBird route peers for the site subnets;
the routed network itself is declared in TF).
## Generated inventory
The inventory is **generated by Terraform at run time**, not hand-written.
`mise run mgmt:render-inventory` (run automatically by `mgmt:ansible`) invokes
`tf/render/ansible-mgmt`, which derives everything from the single sources of
truth and writes these **gitignored** files:
| file | generated from |
|---|---|
| `inventories/<site>/hosts.yml` | `mgmt-hosts.yaml` (host names + public IPs) |
| `inventories/<site>/host_vars/*.yml` | `mgmt-hosts.yaml` (NIC) + `fabric-addressing` (VLAN ids/addresses, subnet route) |
| `inventories/<site>/group_vars/all/users.generated.yml` | `tf/shared/modules/identity` (`server`-mapped users) |
Only `group_vars/all/main.yml` (static config) and `roles/**` are committed. To
change hosts, addresses, or users, edit the Terraform sources — never the
generated files. The render uses only the `local` provider (no backend, no
secrets), so it runs anywhere.
## Reprovision → Ansible flow
1. Reprovision both hosts to Debian 13 via Hetzner robot auto-install.
2. Post-reprovision, root is reachable over SSH with the TF-generated
provisioning key (stored in 1Password). Render it and run `site.yml`.
```bash
# Render the provisioning private key from 1Password to a temp file
umask 077
op read --account team-futo \
"op://yucca_tf_prod/HTZ_FSN1_PROVISIONING_SSH_PRIVATE_KEY/password" \
> /tmp/htz-fsn1-prov-key
chmod 600 /tmp/htz-fsn1-prov-key
# Run the playbook (root, provisioning key, NetBird mgmt setup key from 1P)
ansible-playbook -i inventories/htz-fsn1 site.yml \
--private-key /tmp/htz-fsn1-prov-key \
--extra-vars "mgmt_netbird_setup_key=$(op read --account team-futo \
'op://yucca_tf_prod/NETBIRD_YUCCA_PROD_HTZ_FSN1_MGMT_SETUP_KEY/password')"
# Clean up
shred -u /tmp/htz-fsn1-prov-key
```
`ansible.cfg` sets `inventory = inventories/htz-fsn1/hosts.yml`, so `-i` is
optional. The inventory connects as `ansible_user: root`.
The whole render-key + run flow above is wrapped by `mise run mgmt:ansible`
(`SITE` selects the inventory; defaults to `htz-fsn1`), which CI also runs on
every **prod** apply — the `Ansible converge (mgmt hosts)` step of
`.github/workflows/infra.yml`, right after the Terraform apply. It's idempotent
and reaches the hosts over their public IP, so it requires them to already be
reprovisioned (provisioning key authorized).
> The `--private-key` flow is preferred over `ansible_ssh_private_key_file` in
> group_vars so the key never has to be persisted to a committed path — it is
> rendered to a temp file, used, and shredded.
## Connection: public IP → NetBird
A freshly-reprovisioned host is only reachable over its **public IP**, so that's
the bootstrap address (`mgmt_public_ip` in host_vars). The first play in
`site.yml` probes the overlay from the control node (`netbird status --json`):
if the host is a **connected peer**, it switches `ansible_host` to the host's
**NetBird IP** for the rest of the run; otherwise it stays on the public IP.
So the first run provisions over the public IP and brings the host onto the
overlay (the `netbird` role); every subsequent run reconnects over NetBird
automatically. This needs the control node on the overlay too — the CI
`prod-ansible` job joins via the `ci` setup key; locally, be connected to NetBird.
For the very first run after a reinstall, pass `-e mgmt_bootstrap=true` to force
the public IP (skips any stale peer entry for the host).
## 25G fabric caveat
The 25G fabric link (Intel E810, "ice" driver) is **currently physically
unreliable**. The `networkd` role that configures its VLAN sub-interfaces is
gated off by default (`mgmt_networkd_enabled: false`), so a normal `site.yml`
run is a no-op for networking. Once the fabric links:
1. Confirm the NIC name on each host with `ip link` (prior name:
`enp33s0f0np0`) and correct `mgmt_fabric_nic` in the host_vars if it
differs.
2. Set `mgmt_networkd_enabled: true` (e.g. `--extra-vars` or group_vars).
VLAN layout (gateways are `.1` on the leaf IRB):
| VLAN | Network | mgmt-1 | mgmt-2 |
|------|---------|--------|--------|
| 20 (cluster public) | `10.40.20.0/23` | `10.40.20.2` | `10.40.20.3` |
| 22 (cluster private) | `10.40.22.0/23` | `10.40.22.2` | `10.40.22.3` |
The primary public NIC keeps Hetzner's DHCP default — this tree does not touch
it.
## Setup
```bash
ansible-galaxy collection install -r requirements.yml
ansible-playbook -i inventories/htz-fsn1 --syntax-check site.yml
```
## Secrets
Nothing secret is committed. SSH public keys are public data and are TF-generated
into `group_vars/all/users.generated.yml`. Runtime secrets are passed via `op read`:
- Provisioning private key: `op://yucca_tf_prod/HTZ_FSN1_PROVISIONING_SSH_PRIVATE_KEY/password`
- NetBird mgmt setup key: `op://yucca_tf_prod/NETBIRD_YUCCA_PROD_<SITE>_MGMT_SETUP_KEY/password`
(the site's reusable `mgmt` key, `auto_groups=["mgmt"]`; minted by the prod netbird stack)
+30
View File
@@ -0,0 +1,30 @@
[defaults]
# Connects as root with the TF-provisioning key (see README). Inventory is
# committed (these are two static hosts, not TF-rendered like ceph).
inventory = inventories/htz-fsn1/hosts.yml
log_path = ansible.log
# Performance
forks = 20
gathering = smart
fact_caching = jsonfile
fact_caching_connection = .ansible_facts_cache
fact_caching_timeout = 3600
# Output
stdout_callback = default
result_format = yaml
callbacks_enabled = ansible.posix.timer, ansible.posix.profile_tasks
force_color = True
diff_always = True
deprecation_warnings = False
retry_files_enabled = False
display_skipped_hosts = False
# Security
host_key_checking = False
timeout = 30
[ssh_connection]
ssh_args = -o ControlMaster=auto -o ControlPersist=60s -o StrictHostKeyChecking=no
pipelining = True
@@ -0,0 +1,14 @@
---
# Static shared vars for the htz-fsn1 mgmt hosts.
#
# Host/user/addressing data is TF-GENERATED at run time (hosts.yml,
# host_vars/*.yml, users.generated.yml) by tf/render/ansible-mgmt from the
# Terraform sources (mgmt-hosts.yaml, fabric-addressing, identity). Edit those,
# not the generated files. Only truly-static config lives here.
timezone: UTC
mgmt_domain: fsn.htz.futo.cloud
# NetBird "mgmt" setup key — NOT hardcoded; passed at run time by
# `mise run mgmt:ansible` from op://yucca_tf_prod/NETBIRD_YUCCA_PROD_<SITE>_MGMT_SETUP_KEY.
mgmt_netbird_setup_key: ""
+6
View File
@@ -0,0 +1,6 @@
---
collections:
- name: ansible.posix
version: ">=2.0.0"
- name: community.general
version: ">=9.0.0"
@@ -0,0 +1,19 @@
---
# Post-reprovision OS baseline defaults.
# --- Packages ---
mgmt_base_packages:
- curl
- gnupg
- ca-certificates
- apt-transport-https
- ethtool
- jq
- htop
- tmux
- ncdu
- sysstat
- bsdmainutils
# --- /etc/hosts ---
mgmt_manage_hosts: true
+10
View File
@@ -0,0 +1,10 @@
---
# First role in site.yml — runs on the freshly installed OS.
dependencies: []
galaxy_info:
author: FUTO
license: AGPL-3.0-only
role_name: baseline
description: Post-reprovision OS baseline — packages, hostname, timezone, hosts
min_ansible_version: "2.18"
@@ -0,0 +1,12 @@
---
# Post-reprovision OS baseline.
# Runs as root over SSH on the freshly installed Debian 13 host.
# Convergeable: re-running corrects drift in packages, hostname, hosts file.
- name: Configure system packages
ansible.builtin.import_tasks: packages.yml
tags: [packages]
- name: Configure hostname, timezone, /etc/hosts
ansible.builtin.import_tasks: system.yml
tags: [system]
@@ -0,0 +1,12 @@
---
# Base package set. Convergeable: re-running installs anything missing.
- name: Update apt cache
ansible.builtin.apt:
update_cache: true
cache_valid_time: 3600
- name: Install base packages
ansible.builtin.apt:
name: "{{ mgmt_base_packages }}"
state: present
@@ -0,0 +1,20 @@
---
# Hostname, timezone, /etc/hosts.
# Convergeable: re-running corrects drift.
- name: Set hostname from inventory
ansible.builtin.hostname:
name: "{{ inventory_hostname }}"
- name: Set timezone
community.general.timezone:
name: "{{ timezone }}"
- name: Render /etc/hosts with mgmt node entries
ansible.builtin.template:
src: hosts.j2
dest: /etc/hosts
owner: root
group: root
mode: '0644'
when: mgmt_manage_hosts | bool
@@ -0,0 +1,13 @@
127.0.0.1 localhost
127.0.1.1 {{ inventory_hostname }}.{{ mgmt_domain }} {{ inventory_hostname }}
# Management nodes (25G fabric VLAN 20 — cluster public)
{% for host in groups['mgmt'] %}
{% for vlan in hostvars[host]['mgmt_fabric_vlans'] | default([]) if vlan.id == 20 %}
{{ vlan.address | regex_replace('/.*$', '') }} {{ host }}.{{ mgmt_domain }} {{ host }}
{% endfor %}
{% endfor %}
::1 localhost ip6-localhost ip6-loopback
ff02::1 ip6-allnodes
ff02::2 ip6-allrouters
@@ -0,0 +1,9 @@
---
# NetBird defaults.
mgmt_netbird_management_url: "https://api.netbird.io"
# Setup key — NEVER hardcoded. Passed at run time (op:// ref, see README). The
# site's reusable "mgmt" key (auto_groups=["mgmt"]) so the node joins the mgmt
# group and becomes a route peer for the site subnets (10.40.5.0/24, api, cluster
# nets — the routed network is defined in TF: tf/deployment/prod/<site>/netbird).
mgmt_netbird_setup_key: ""
+10
View File
@@ -0,0 +1,10 @@
---
# Install NetBird and join the overlay (mgmt nodes are the site route peers).
dependencies: []
galaxy_info:
author: FUTO
license: AGPL-3.0-only
role_name: netbird
description: Install NetBird and bring the node up on the overlay
min_ansible_version: "2.18"
+56
View File
@@ -0,0 +1,56 @@
---
# Install NetBird and join the overlay. The mgmt nodes are the NetBird route
# peers for the site subnets (the routed network is declared in TF, not here), so
# enable IP forwarding. Convergeable: `netbird up` is skipped when already
# connected.
- name: Install NetBird (official script — adds the apt repo + installs)
ansible.builtin.shell:
cmd: curl -fsSL https://pkgs.netbird.io/install.sh | sh
creates: /usr/bin/netbird
environment:
NETBIRD_SKIP_UP: "true" # don't auto-join during install; we run `up` below
- name: Enable and start netbird
ansible.builtin.systemd:
name: netbird
enabled: true
state: started
- name: Assert NetBird setup key is provided
ansible.builtin.assert:
that:
- mgmt_netbird_setup_key | length > 0
fail_msg: >-
mgmt_netbird_setup_key is empty. Pass it at run time (mgmt:ansible does this
from op://yucca_tf_prod/NETBIRD_YUCCA_PROD_<SITE>_MGMT_SETUP_KEY).
# mgmt nodes route the site subnets to the overlay.
- name: Enable IP forwarding (NetBird route peer)
ansible.posix.sysctl:
name: "{{ item }}"
value: '1'
sysctl_set: true
state: present
reload: true
loop:
- net.ipv4.ip_forward
- net.ipv6.conf.all.forwarding
- name: Check NetBird connection status
ansible.builtin.command: netbird status
register: mgmt_netbird_status
changed_when: false
failed_when: false
- name: Bring the node up on the overlay
ansible.builtin.command:
cmd: >-
netbird up
--setup-key {{ mgmt_netbird_setup_key }}
--management-url {{ mgmt_netbird_management_url }}
--hostname {{ inventory_hostname }}
register: mgmt_netbird_up
changed_when: mgmt_netbird_up.rc == 0
no_log: true
when: "'Management: Connected' not in mgmt_netbird_status.stdout"
@@ -0,0 +1,14 @@
---
# systemd-networkd VLAN sub-interfaces on the 25G fabric NIC.
#
# Prerequisites:
# - systemd-networkd present (Debian 13 base)
# - mgmt_fabric_nic, mgmt_fabric_vlans defined in host_vars
#
# Master toggle. False by default because the 25G fabric link is currently
# unreliable — flip to true (and confirm mgmt_fabric_nic via `ip link`) once
# the fabric is up.
mgmt_networkd_enabled: false
# --- Paths ---
mgmt_networkd_config_dir: /etc/systemd/network
@@ -0,0 +1,5 @@
---
- name: Reload systemd for networkd
ansible.builtin.systemd:
name: systemd-networkd
state: reloaded
+11
View File
@@ -0,0 +1,11 @@
---
# 25G fabric VLAN sub-interfaces. Gated on mgmt_networkd_enabled (default
# false) — applies once the 25G fabric links.
dependencies: []
galaxy_info:
author: FUTO
license: AGPL-3.0-only
role_name: networkd
description: systemd-networkd VLAN sub-interfaces on the 25G fabric NIC
min_ansible_version: "2.18"
@@ -0,0 +1,38 @@
---
# Write systemd-networkd config files for the 25G fabric VLANs.
# Safe to re-run.
- name: Deploy fabric parent .network (declares VLANs, no L3 of its own)
ansible.builtin.template:
src: fabric.network.j2
dest: "{{ mgmt_networkd_config_dir }}/20-{{ mgmt_fabric_nic }}.network"
owner: root
group: root
mode: '0644'
notify: Reload systemd for networkd
- name: Deploy VLAN .netdev files
ansible.builtin.template:
src: vlan.netdev.j2
dest: "{{ mgmt_networkd_config_dir }}/30-{{ mgmt_fabric_nic }}.{{ vlan.id }}.netdev"
owner: root
group: root
mode: '0644'
loop: "{{ mgmt_fabric_vlans }}"
loop_control:
loop_var: vlan
label: "vlan {{ vlan.id }}"
notify: Reload systemd for networkd
- name: Deploy VLAN .network files
ansible.builtin.template:
src: vlan.network.j2
dest: "{{ mgmt_networkd_config_dir }}/30-{{ mgmt_fabric_nic }}.{{ vlan.id }}.network"
owner: root
group: root
mode: '0644'
loop: "{{ mgmt_fabric_vlans }}"
loop_control:
loop_var: vlan
label: "vlan {{ vlan.id }}"
notify: Reload systemd for networkd
@@ -0,0 +1,17 @@
---
# systemd-networkd VLAN sub-interfaces on the 25G fabric NIC.
#
# NOTE: applies once the 25G fabric links — the 25G link is currently
# physically unreliable (see notes). Gated on mgmt_networkd_enabled
# (default: false) so a normal site.yml run is a no-op until the fabric
# is up and the operator opts in.
#
# 1. write parent .network (DHCP off on fabric, declares VLANs)
# 2. write per-VLAN .netdev + .network (static addresses)
# 3. reload networkd
#
# The primary public NIC keeps Hetzner's DHCP default — untouched here.
- name: Deploy networkd configs
ansible.builtin.import_tasks: deploy.yml
when: mgmt_networkd_enabled | bool
@@ -0,0 +1,15 @@
# {{ ansible_managed }}
# 25G fabric parent NIC: {{ mgmt_fabric_nic }} (Intel E810, "ice").
# No L3 on the parent itself — addresses live on the VLAN sub-interfaces.
# The `VLAN=` directives below tell networkd to instantiate the sub-interfaces
# declared by the .netdev files (Kind=vlan alone only registers the type).
[Match]
Name={{ mgmt_fabric_nic }}
[Network]
{% for vlan in mgmt_fabric_vlans %}
VLAN={{ mgmt_fabric_nic }}.{{ vlan.id }}
{% endfor %}
LinkLocalAddressing=no
IPv6AcceptRA=no
@@ -0,0 +1,9 @@
# {{ ansible_managed }}
# Tagged VLAN {{ vlan.id }} sub-interface on {{ mgmt_fabric_nic }}.
[NetDev]
Name={{ mgmt_fabric_nic }}.{{ vlan.id }}
Kind=vlan
[VLAN]
Id={{ vlan.id }}
@@ -0,0 +1,11 @@
# {{ ansible_managed }}
# L3 on VLAN {{ vlan.id }} sub-interface: {{ vlan.address }} (gateway .1 on the leaf IRB).
[Match]
Name={{ mgmt_fabric_nic }}.{{ vlan.id }}
[Network]
Address={{ vlan.address }}
[Link]
RequiredForOnline=no
@@ -0,0 +1,25 @@
---
# Security hardening defaults for management hosts.
# --- Firewall (nftables) ---
mgmt_firewall_enabled: true
# Trusted source networks allowed to reach management services beyond SSH.
# RFC1918 + NetBird/CGNAT (100.64.0.0/10). Loopback is always allowed via "lo".
mgmt_firewall_trusted_networks:
- 10.0.0.0/8
- 172.16.0.0/12
- 192.168.0.0/16
- 100.64.0.0/10
# Allow SSH from any source (true) or restrict to trusted networks (false).
# Default true: these hosts are reached over the public IP for bootstrapping
# and NetBird admin. Tighten once NetBird access is confirmed.
mgmt_firewall_ssh_any_source: true
# --- SSH hardening ---
mgmt_ssh_max_auth_tries: 3
# Key-based root is required for the initial Ansible run (TF provisioning key).
# prohibit-password keeps password root login off while allowing key auth.
mgmt_ssh_permit_root_login: prohibit-password
mgmt_ssh_allowed_users: "root nutgood andy"
@@ -0,0 +1,3 @@
// Managed by Ansible (security role)
APT::Periodic::Update-Package-Lists "1";
APT::Periodic::Unattended-Upgrade "1";
@@ -0,0 +1,5 @@
---
- name: Restart sshd
ansible.builtin.systemd:
name: ssh
state: restarted
+10
View File
@@ -0,0 +1,10 @@
---
# nftables firewall, SSH hardening, unattended-upgrades.
dependencies: []
galaxy_info:
author: FUTO
license: AGPL-3.0-only
role_name: security
description: nftables firewall, SSH hardening, unattended-upgrades for mgmt hosts
min_ansible_version: "2.18"
@@ -0,0 +1,70 @@
---
# Security hardening for management hosts.
# nftables firewall + SSH hardening + unattended-upgrades.
# --- nftables firewall ---
- name: Install nftables
ansible.builtin.apt:
name: nftables
state: present
- name: Deploy nftables ruleset
ansible.builtin.template:
src: nftables.conf.j2
dest: /etc/nftables.conf
owner: root
group: root
mode: '0644'
validate: "nft -c -f %s"
register: nftables_config
when: mgmt_firewall_enabled | bool
- name: Enable and start nftables
ansible.builtin.systemd:
name: nftables
enabled: true
state: "{{ 'restarted' if nftables_config is changed else 'started' }}"
when: mgmt_firewall_enabled | bool
# --- SSH hardening ---
- name: Deploy SSH hardening config
ansible.builtin.template:
src: sshd-hardening.conf.j2
dest: /etc/ssh/sshd_config.d/50-hardening.conf
owner: root
group: root
mode: '0644'
notify: Restart sshd
- name: Verify full SSH config is valid after drop-in
ansible.builtin.command: sshd -t
changed_when: false
# --- unattended-upgrades ---
- name: Install unattended-upgrades
ansible.builtin.apt:
name:
- unattended-upgrades
- apt-listchanges
state: present
- name: Enable unattended-upgrades
ansible.builtin.copy:
src: 20auto-upgrades
dest: /etc/apt/apt.conf.d/20auto-upgrades
owner: root
group: root
mode: '0644'
# --- Summary ---
- name: Report security state
ansible.builtin.debug:
msg:
- "Firewall: {{ 'enabled' if mgmt_firewall_enabled else 'disabled' }}"
- "Trusted networks: {{ mgmt_firewall_trusted_networks | join(', ') }}"
- "SSH AllowUsers: {{ mgmt_ssh_allowed_users }}"
- "SSH PermitRootLogin: {{ mgmt_ssh_permit_root_login }}"
@@ -0,0 +1,47 @@
#!/usr/sbin/nft -f
# Management host firewall — managed by Ansible (security role)
# Allows SSH + established/related + ICMP. Drops everything else.
flush ruleset
table inet filter {
chain input {
type filter hook input priority 0; policy drop;
# Loopback
iifname "lo" accept
# NetBird overlay interface (mesh + subnet routing)
iifname "wt0" accept
# Established/related
ct state established,related accept
# ICMP (ping, MTU discovery)
ip protocol icmp accept
ip6 nexthdr icmpv6 accept
# SSH
{% if mgmt_firewall_ssh_any_source | bool %}
tcp dport 22 accept
{% else %}
{% for net in mgmt_firewall_trusted_networks %}
ip saddr {{ net }} tcp dport 22 accept
{% endfor %}
{% endif %}
# Log + drop everything else
limit rate 5/minute log prefix "nftables-drop: " level warn
drop
}
chain forward {
# NetBird route peer forwards between the overlay and the site subnets
# (10.40.5.0/24 et al.).
type filter hook forward priority 0; policy accept;
}
chain output {
type filter hook output priority 0; policy accept;
}
}
@@ -0,0 +1,10 @@
# SSH hardening — managed by Ansible (security role)
# Dropped into /etc/ssh/sshd_config.d/ to override defaults without
# touching the main sshd_config.
MaxAuthTries {{ mgmt_ssh_max_auth_tries }}
PermitRootLogin {{ mgmt_ssh_permit_root_login }}
AllowUsers {{ mgmt_ssh_allowed_users }}
PasswordAuthentication no
PermitEmptyPasswords no
X11Forwarding no
@@ -0,0 +1,4 @@
---
# Login users. The real list lives in group_vars/all.yml (mgmt_users),
# mirrored from tf/shared/modules/identity. Default empty here.
mgmt_users: []
+10
View File
@@ -0,0 +1,10 @@
---
# Creates login users from the identity registry mirror (mgmt_users).
dependencies: []
galaxy_info:
author: FUTO
license: AGPL-3.0-only
role_name: users
description: Login users + SSH keys + sudo from the identity registry
min_ansible_version: "2.18"
+37
View File
@@ -0,0 +1,37 @@
---
# Login users for members of the identity registry's `server`-mapped groups.
# Sourced from mgmt_users (group_vars/all.yml). Convergeable: re-running fixes
# drift in group membership, sudo, and authorized_keys.
- name: Ensure login users exist
ansible.builtin.user:
name: "{{ item.name }}"
shell: /bin/bash
groups: sudo
append: true
create_home: true
state: present
loop: "{{ mgmt_users }}"
loop_control:
label: "{{ item.name }}"
- name: Configure per-user sudo
ansible.builtin.copy:
content: "{{ item.name }} ALL=(ALL) {{ item.sudo }}\n"
dest: "/etc/sudoers.d/{{ item.name }}"
mode: '0440'
validate: "visudo -cf %s"
loop: "{{ mgmt_users }}"
loop_control:
label: "{{ item.name }}"
# exclusive: true makes the registry the single source of truth — keys not
# listed for the user are removed.
- name: Authorize user SSH keys
ansible.posix.authorized_key:
user: "{{ item.name }}"
key: "{{ item.ssh_keys | join('\n') }}"
exclusive: true
loop: "{{ mgmt_users }}"
loop_control:
label: "{{ item.name }}"
+60
View File
@@ -0,0 +1,60 @@
---
# Configure the Hetzner FSN1 management hosts after they are reprovisioned
# to Debian 13 ("trixie") via Hetzner robot auto-install.
#
# Connection: freshly-reprovisioned hosts are reached over their PUBLIC IP
# (bootstrap); once a host has joined the NetBird overlay (the netbird role),
# later runs reconnect over its NetBird IP automatically — see the first play.
#
# Canonical entrypoint is `mise run mgmt:ansible` (renders the inventory from
# Terraform, then runs this). It passes the provisioning key + the NetBird
# "mgmt" setup key (op://yucca_tf_prod/NETBIRD_YUCCA_PROD_<SITE>_MGMT_SETUP_KEY).
#
# NOTE: the networkd role (25G VLAN sub-interfaces) is gated on
# mgmt_networkd_enabled (default false) because the 25G fabric link is
# currently unreliable. Enable it once the fabric is up.
# ── Prefer NetBird, fall back to the public IP ───────────────────────────────
# Runs entirely on the control node (no target connection), so it's safe before
# a host is reachable at all. If the host is a connected NetBird peer, switch the
# connection to its NetBird IP; otherwise keep the public IP (bootstrap).
- name: Select connection address
hosts: mgmt
gather_facts: false
become: false
tasks:
- name: Read the NetBird overlay status from the control node
ansible.builtin.command: netbird status --json
delegate_to: localhost
register: _nb_status
changed_when: false
failed_when: false
- name: Reconnect over NetBird when the host is a connected peer
vars:
_nb_ips: >-
{{ ((_nb_status.stdout | from_json).peers.details | default([]))
| selectattr('fqdn', 'defined')
| selectattr('fqdn', 'match', '^' ~ inventory_hostname ~ '([.]|$)')
| selectattr('status', 'equalto', 'Connected')
| map(attribute='netbirdIp') | select | list }}
ansible.builtin.set_fact:
ansible_host: "{{ _nb_ips[0] if (_nb_ips | length > 0) else mgmt_public_ip }}"
when:
# `-e mgmt_bootstrap=true` forces the public IP — use it for the first run
# after a reinstall, when a stale OLD peer for this host may still show up.
- not (mgmt_bootstrap | default(false) | bool)
- _nb_status.rc == 0
- (_nb_status.stdout | trim | length) > 0
- name: Configure management hosts
hosts: mgmt
become: true
gather_facts: true
roles:
- baseline
- users
- security
- networkd
- netbird
+5
View File
@@ -20,3 +20,8 @@ export TF_VAR_netbox_token=op://yucca_tf/NETBOX_API_TOKEN/password
# Stored at op://yucca_tf_prod/NET_SWITCHES_TERRAFORM_SSH_PRIVATE_KEY/password. # Stored at op://yucca_tf_prod/NET_SWITCHES_TERRAFORM_SSH_PRIVATE_KEY/password.
# NOT exported here as content — the fabric mise task renders it to a 0600 temp # NOT exported here as content — the fabric mise task renders it to a 0600 temp
# file and exports TF_VAR_netconf_ssh_key_path=<that path>. # file and exports TF_VAR_netconf_ssh_key_path=<that path>.
# ── Hetzner Robot API (mgmt-host reprovisioning, zack/hetzner provider) ───────
# The provider reads these env vars directly (no provider config block needed).
export HETZNER_ROBOT_USERNAME=op://yucca_tf_prod/HETZNER_WEBSERVICE_API_USER/password
export HETZNER_ROBOT_PASSWORD=op://yucca_tf_prod/HETZNER_WEBSERVICE_API_PASSWORD/password
+10
View File
@@ -0,0 +1,10 @@
# Generated by 'mise run fabric:provider-build' — do not edit.
provider_installation {
filesystem_mirror {
path = "/Users/leca/src/yucca/.mise/.fabric-provider-mirror"
include = ["registry.opentofu.org/hashicorp/junos-qfx", "registry.terraform.io/hashicorp/junos-qfx"]
}
direct {
exclude = ["registry.opentofu.org/hashicorp/junos-qfx", "registry.terraform.io/hashicorp/junos-qfx"]
}
}
+17 -11
View File
@@ -35,25 +35,31 @@ VLAN id == the network's third octet; gateway = `.1` (IRB on the leaf).
keys committed; passwords via vars from 1Password). Fed by `modules/identity`. keys committed; passwords via vars from 1Password). Fed by `modules/identity`.
- `modules/fabric-netbox` — mirrors the IP plan into NetBox (prefixes + VLANs). - `modules/fabric-netbox` — mirrors the IP plan into NetBox (prefixes + VLANs).
## The provider ## The providers
The `junos-qfx` provider is **JTAF-generated and vendored** in `tf/providers/` (not on This stack builds two providers locally (neither is on a registry) into a shared
any registry). It's built into a local filesystem mirror by `mise run fabric:provider-build`. filesystem mirror via `mise run infra:providers`:
- `mise run fabric:provider-gen` — regenerate from device YANG + live config (only - `junos-qfx` — **JTAF-generated and vendored** in `tf/providers/` (the switch fabric).
when adding new config hierarchies), then commit `tf/providers/`. - `hetzner` (`zack/hetzner`) — the Hetzner Robot API, for mgmt-host reprovisioning
- `mise run fabric:provider-build` — `go build` the vendored source into the mirror (`mgmt.tf`); cloned + built (pinned tag) by `mise run mgmt:provider-build`.
and write `tf/.terraformrc.fabric` (consumed via `TF_CLI_CONFIG_FILE`).
- `mise run fabric:provider-gen` — regenerate the junos-qfx provider from device YANG
+ live config (only when adding new config hierarchies), then commit `tf/providers/`.
- `mise run infra:providers` — build both providers into the mirror and write
`tf/.terraformrc.local` (consumed via `TF_CLI_CONFIG_FILE`).
## Running ## Running
```sh ```sh
mise run fabric:plan # builds provider, renders the NETCONF key from 1Password, terragrunt plan mise run infra:plan # builds providers, renders creds from 1Password, terragrunt plan
mise run fabric:apply # ... apply mise run infra:apply # ... apply
``` ```
CI: `.github/workflows/fabric.yml` — plan on PR, gated apply on merge behind the (`SITE` selects the stack; defaults to `htz-fsn1`.)
site-scoped `prod-fabric-htz-fsn1` GitHub Environment (required reviewers).
CI: `.github/workflows/infra.yml` — plan on PR, gated apply on merge behind the
site-scoped `prod-htz-fsn1` GitHub Environment (required reviewers).
## Adoption caveat (first run) ## Adoption caveat (first run)
@@ -0,0 +1,27 @@
# Canonical mgmt-host roster for htz-fsn1 — the single source of truth consumed by
# BOTH the reprovision stack (mgmt.tf, uses server_number) and the ansible
# inventory render (tf/render/ansible-mgmt, uses everything else).
#
# site_id — drives addressing (must match the stack's var.site_id).
# cluster_id — the cluster whose public/private VLANs these hosts sit on.
# host_index — host's offset within each VLAN /23 (gateway is .1 on the leaf);
# .2/.3 here -> 10.40.20.2/.3 (public), 10.40.22.2/.3 (private).
# fabric_nic — 25G NIC carrying the tagged VLAN sub-interfaces (verify after
# reprovision; predictable name may differ on fresh Debian 13).
# subnet_router — advertises the mgmt /24 over Tailscale (exactly one host).
site_id: 40
cluster_id: 1
hosts:
htz-fsn-mgmt-1:
server_number: 3008208
public_ip: 178.63.124.40
host_index: 2
fabric_nic: enp33s0f0np0
subnet_router: true
htz-fsn-mgmt-2:
server_number: 3008209
public_ip: 178.63.124.41
host_index: 3
fabric_nic: enp33s0f0np0
subnet_router: false
+78
View File
@@ -0,0 +1,78 @@
# mgmt hosts — Hetzner dedicated-server reprovisioning (zack/hetzner robot API).
#
# Two-step, operator-gated flow:
# 1. Terraform arms a fresh Debian auto-install on the robot (hetzner_boot_linux),
# authorizing a TF-owned automation SSH key for root. This is NON-destructive
# — the flag only takes effect on the next boot; the running host is untouched.
# 2. The operator reboots the host (manually, for now) -> the robot wipes the
# disks + installs Debian -> ansible/mgmt configures it (networkd, tailscale,
# users from modules/identity, baseline + hardening).
#
# Only hosts listed in var.mgmt_reprovision_targets are armed, so a normal apply
# does nothing to the mgmt hosts. Re-target deliberately for each reprovision.
#
# Host roster (server numbers = Hetzner robot IDs, GET /server) comes from the
# shared mgmt-hosts.yaml — the same SoT the ansible inventory render reads.
locals {
mgmt_roster = yamldecode(file("${path.module}/mgmt-hosts.yaml"))
mgmt_hosts = local.mgmt_roster.hosts
site_prefix = upper(replace(var.netbox_site_slug, "-", "_")) # HTZ_FSN1
provisioning_key_item = "${local.site_prefix}_PROVISIONING_SSH_PRIVATE_KEY"
}
# Guard: the roster's site_id must match the stack's, or addressing diverges.
resource "terraform_data" "mgmt_site_id_check" {
lifecycle {
precondition {
condition = local.mgmt_roster.site_id == var.site_id
error_message = "mgmt-hosts.yaml site_id (${local.mgmt_roster.site_id}) != var.site_id (${var.site_id})."
}
}
}
# ── Provisioning keypair (TF-owned, recorded in 1Password) ───────────────────
# Generated here, stored in the env-appropriate vault as the source-of-truth
# record, and registered in the Hetzner robot. Authorized for root on freshly-
# reprovisioned hosts; ansible/mgmt reads the private half from 1Password
# (op://<vault>/<title>/password). Survives state loss and is operator-visible.
resource "tls_private_key" "mgmt_provisioning" {
algorithm = "ED25519"
}
data "onepassword_vault" "env" {
name = var.op_vault
}
resource "onepassword_item" "mgmt_provisioning_key" {
vault = data.onepassword_vault.env.uuid
title = local.provisioning_key_item
category = "password"
password = tls_private_key.mgmt_provisioning.private_key_openssh
section {
label = "keypair"
field {
label = "public_key"
type = "STRING"
value = trimspace(tls_private_key.mgmt_provisioning.public_key_openssh)
}
}
}
resource "hetzner_ssh_key" "mgmt_automation" {
name = "${local.site_prefix}-provisioning"
data = trimspace(tls_private_key.mgmt_provisioning.public_key_openssh)
}
# ── Arm a fresh OS install for each targeted host ────────────────────────────
# server_number RequiresReplace, so re-targeting recreates cleanly.
resource "hetzner_boot_linux" "mgmt" {
for_each = toset(var.mgmt_reprovision_targets)
server_number = local.mgmt_hosts[each.key].server_number
dist = var.mgmt_dist
lang = "en"
arch = 64
authorized_key = hetzner_ssh_key.mgmt_automation.fingerprint
}
@@ -27,6 +27,22 @@ policies = {
destinations = ["mgmt", "talos", "k8s_operator"] destinations = ["mgmt", "talos", "k8s_operator"]
}] }]
} }
# CI also reaches the routed site subnets (switch vme 10.40.5.0/24 + api +
# cluster nets), so the fabric jobs can NETCONF the switches over the overlay
# (the switches are routed resources behind the mgmt peers, not peers
# themselves). yucca_resource is the shared tag on every routed resource,
# resolved from the global layer via external_groups.
ci-to-resources = {
description = "CI → routed site subnets (yucca_resource)."
rules = [{
name = "ci-to-resources"
protocol = "all"
bidirectional = false
sources = ["ci"]
destinations = ["yucca_resource"]
}]
}
} }
# Site identifier (mirrors prod/htz-fsn1's site_id). Feeds the fabric-addressing # Site identifier (mirrors prod/htz-fsn1's site_id). Feeds the fabric-addressing
+10
View File
@@ -23,3 +23,13 @@ provider "netbox" {
server_url = var.netbox_url server_url = var.netbox_url
api_token = var.netbox_token api_token = var.netbox_token
} }
# Hetzner Robot API — mgmt-host reprovisioning (mgmt.tf). Credentials come from
# the env (HETZNER_ROBOT_USERNAME/PASSWORD), injected by op-run from tf/.env.prod;
# no secrets in config.
provider "hetzner" {}
# 1Password — stores the TF-generated provisioning key (mgmt.tf) in the env vault.
# Authenticates with OP_SERVICE_ACCOUNT_TOKEN (the `infra:apply` task escalates to
# the write-capable SA pulled from the vault); no Connect host needed.
provider "onepassword" {}
@@ -11,3 +11,13 @@ cls1_leaf_serials = ["XH4925470753", "XH4925460012"]
# spine (corenetsw) VC member serials. # spine (corenetsw) VC member serials.
spine_vc_serials = ["WH3622440738", "WH0220510012"] spine_vc_serials = ["WH3622440738", "WH0220510012"]
# ── mgmt-host reprovisioning (mgmt.tf) ───────────────────────────────────────
# The provisioning keypair is GENERATED by Terraform and stored in 1Password
# (op_vault, default yucca_tf_prod) as HTZ_FSN1_PROVISIONING_SSH_PRIVATE_KEY;
# ansible/mgmt reads the private half from there. Nothing to set here.
# Hosts to ARM for a fresh OS install on next boot (DESTRUCTIVE on reboot).
# Testing the flow on mgmt-2 first (mgmt-1 is the tailscale subnet router).
# Set back to [] once reprovisioning is complete.
mgmt_reprovision_targets = ["htz-fsn-mgmt-2"]
+23
View File
@@ -53,3 +53,26 @@ variable "spine_vc_serials" {
type = list(string) type = list(string)
description = "Spine (corenetsw) VC member chassis serials (member 0, member 1)." description = "Spine (corenetsw) VC member chassis serials (member 0, member 1)."
} }
# ── mgmt-host reprovisioning (mgmt.tf) ───────────────────────────────────────
variable "op_vault" {
type = string
default = "yucca_tf_prod"
description = "1Password vault (env-appropriate) the TF-generated provisioning key is written to."
}
variable "mgmt_dist" {
type = string
default = "Debian 13 base"
description = "Hetzner robot Linux auto-install image (must match an available `dist` exactly; see GET /boot/<n>/linux)."
}
variable "mgmt_reprovision_targets" {
type = list(string)
default = []
description = <<-EOT
mgmt host keys (see local.mgmt_hosts in mgmt.tf) to ARM for a fresh OS install
on next boot. DESTRUCTIVE once the host is rebooted. Keep empty except during a
planned reprovision; set to the host(s) being reprovisioned, apply, then reboot.
EOT
}
+19 -1
View File
@@ -2,7 +2,7 @@ terraform {
required_version = "~> 1.11" required_version = "~> 1.11"
required_providers { required_providers {
# JTAF-generated, vendored in tf/providers/terraform-provider-junos-qfx and # JTAF-generated, vendored in tf/providers/terraform-provider-junos-qfx and
# supplied via dev_overrides (see mise `fabric:provider-build` + the GH workflow). # supplied via dev_overrides (built by `infra:providers`; supplied via TF_CLI_CONFIG_FILE).
junos-qfx = { junos-qfx = {
source = "hashicorp/junos-qfx" source = "hashicorp/junos-qfx"
} }
@@ -10,5 +10,23 @@ terraform {
source = "e-breuninger/netbox" source = "e-breuninger/netbox"
version = "~> 4.0" version = "~> 4.0"
} }
# Hetzner Robot (dedicated-server) API — mgmt-host reprovisioning (mgmt.tf).
# Built locally + supplied via the same filesystem_mirror as junos-qfx
# (mise `mgmt:provider-build`, invoked by `infra:providers`).
hetzner = {
source = "zack/hetzner"
}
# Generates the mgmt-host provisioning keypair (mgmt.tf).
tls = {
source = "hashicorp/tls"
version = "~> 4.0"
}
# Stores the generated provisioning key in 1Password (the env vault). Auth via
# the service-account token in OP_SERVICE_ACCOUNT_TOKEN (apply escalates to the
# write-capable SA stored in the vault; see the `infra:*` mise tasks).
onepassword = {
source = "1Password/onepassword"
version = "~> 2.1"
}
} }
} }
+67
View File
@@ -0,0 +1,67 @@
# Render the ansible/mgmt inventory for a site from Terraform's sources of truth:
# • fabric-addressing — VLAN ids + per-host VLAN addresses
# • identity — server login users (+ keys + sudo)
# • mgmt-hosts.yaml — the host roster (IPs, NIC, host index, subnet router)
#
# Lightweight by design (local provider only, no backend/secrets), so it can run
# just-in-time in the ansible CI job (and locally) via `mise run mgmt:ansible`.
# Outputs are gitignored — TF is the single source of truth, not the rendered files.
locals {
cfg = yamldecode(file("${path.module}/../../deployment/prod/${var.site}/mgmt-hosts.yaml"))
hosts = local.cfg.hosts
inv_dir = "${path.module}/../../../ansible/mgmt/inventories/${var.site}"
pub_mask = split("/", module.addressing.public_cidr)[1]
priv_mask = split("/", module.addressing.private_cidr)[1]
header = "# GENERATED by `mise run mgmt:render-inventory` (tf/render/ansible-mgmt).\n# Do not edit — edit the Terraform sources (mgmt-hosts.yaml, fabric-addressing, identity).\n"
}
module "addressing" {
source = "../../shared/modules/fabric-addressing"
site_id = local.cfg.site_id
cluster_id = local.cfg.cluster_id
}
module "identity" {
source = "../../shared/modules/identity"
}
# Inventory: hosts + public IPs (ansible connects as root over the public IP).
resource "local_file" "hosts" {
filename = "${local.inv_dir}/hosts.yml"
content = "${local.header}${yamlencode({
mgmt = {
hosts = { for name, h in local.hosts : name => { ansible_host = h.public_ip } }
vars = {
ansible_user = "root"
ansible_python_interpreter = "/usr/bin/python3"
}
}
})}"
}
# Per-host: 25G NIC + the tagged VLAN sub-interfaces (ids + addresses from the
# addressing module) + the subnet-router's advertised route.
resource "local_file" "host_vars" {
for_each = local.hosts
filename = "${local.inv_dir}/host_vars/${each.key}.yml"
content = "${local.header}${yamlencode({
# Bootstrap address — site.yml reconnects over NetBird once the host joins.
# (NetBird routes the site subnets via the mgmt peer group; that's declared in
# tf/deployment/prod/<site>/netbird, so there's no per-host advertise flag.)
mgmt_public_ip = each.value.public_ip
mgmt_fabric_nic = each.value.fabric_nic
mgmt_fabric_vlans = [
{ id = module.addressing.public_vlan_id, address = "${cidrhost(module.addressing.public_cidr, each.value.host_index)}/${local.pub_mask}" },
{ id = module.addressing.private_vlan_id, address = "${cidrhost(module.addressing.private_cidr, each.value.host_index)}/${local.priv_mask}" },
]
})}"
}
# group_vars: login users from the identity registry (server-mapped groups).
resource "local_file" "users" {
filename = "${local.inv_dir}/group_vars/all/users.generated.yml"
content = "${local.header}${yamlencode({ mgmt_users = module.identity.server_login })}"
}
+8
View File
@@ -0,0 +1,8 @@
variable "site" {
type = string
default = "htz-fsn1"
description = <<-EOT
Site slug. Selects the roster at tf/deployment/prod/<site>/mgmt-hosts.yaml and
the inventory rendered under ansible/mgmt/inventories/<site>/.
EOT
}
+11
View File
@@ -0,0 +1,11 @@
terraform {
required_version = "~> 1.11"
required_providers {
# Only the local provider — this render writes files from static, public data
# (addressing + identity). No backend, no secrets, no remote state.
local = {
source = "hashicorp/local"
version = "~> 2.5"
}
}
}
+15
View File
@@ -16,6 +16,21 @@ output "server_authorized_keys" {
]))) ])))
} }
output "server_login" {
description = "Server login users: members of any `server`-mapped group, with their SSH keys + effective sudo. Consumed by server provisioning (ansible/mgmt)."
value = [
for uname, u in var.users : {
name = uname
ssh_keys = sort(distinct(concat(u.ssh_ed25519_keys, u.ssh_rsa_keys)))
sudo = try([
for g in u.groups : var.groups[g].server.sudo
if try(var.groups[g].server, null) != null
][0], "ALL")
}
if length([for g in u.groups : g if try(var.groups[g].server, null) != null]) > 0
]
}
output "members_of" { output "members_of" {
description = "Group name -> sorted member usernames. Lets a consumer (e.g. servers) provision a group's people." description = "Group name -> sorted member usernames. Lets a consumer (e.g. servers) provision a group's people."
value = { value = {