mirror of
https://github.com/immich-app/yucca.git
synced 2026-09-30 13:33:00 +08:00
feat(yucca): add full e2e mgmt provisioning maybe (#182)
* feat(yucca): add full e2e mgmt provisioning maybe * moar ! * prefer tailscale over public ip if availbale * ignore files * fix * more progress
This commit is contained in:
@@ -1,111 +0,0 @@
|
||||
name: Fabric (Junos/NetBox)
|
||||
|
||||
# Manages the prod switch fabric with the vendored JTAF junos-qfx provider (built
|
||||
# in CI) + the netbox provider. Fans out over every site under tf/deployment/prod/*:
|
||||
# Plan on PRs; apply on merge to main, each site behind its own
|
||||
# `prod-fabric-<site>` Environment gate (required reviewers).
|
||||
# The runner joins the tailnet to reach the switch vme IPs and renders the NETCONF
|
||||
# key from 1Password (the fabric:* mise tasks; FABRIC_SITE selects the stack).
|
||||
#
|
||||
# Prerequisites (out-of-band):
|
||||
# - Repo secrets: OP_TF_YUCCA_PROD_ENV (+ _WRITE); TS_OAUTH_CLIENT_ID/SECRET.
|
||||
# - 1P items: NET_SWITCHES_TERRAFORM_SSH_PRIVATE_KEY (yucca_tf_prod), NETBOX_API_TOKEN (yucca_tf).
|
||||
# - One GitHub Environment per site: `prod-fabric-<site>` with required reviewers.
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
paths: &fabric_paths
|
||||
- "tf/providers/**"
|
||||
- "tf/shared/modules/fabric-**"
|
||||
- "tf/shared/modules/core-fabric/**"
|
||||
- "tf/shared/modules/cluster-fabric/**"
|
||||
- "tf/deployment/prod/**"
|
||||
- ".github/workflows/fabric.yml"
|
||||
pull_request:
|
||||
paths: *fabric_paths
|
||||
workflow_dispatch:
|
||||
|
||||
# No state locking on the OVH backend — never apply concurrently.
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}
|
||||
cancel-in-progress: false
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
discover:
|
||||
name: Discover sites
|
||||
if: github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
matrix: ${{ steps.sites.outputs.matrix }}
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
with:
|
||||
persist-credentials: false
|
||||
- id: sites
|
||||
name: List tf/deployment/prod/* sites
|
||||
run: |
|
||||
matrix=$(ls -d tf/deployment/prod/*/ | xargs -n1 basename | jq -R . | jq -cs '{site: .}')
|
||||
echo "matrix=$matrix" >> "$GITHUB_OUTPUT"
|
||||
echo "$matrix"
|
||||
|
||||
plan:
|
||||
name: Plan ${{ matrix.site }}
|
||||
needs: discover
|
||||
runs-on: ubuntu-latest
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix: ${{ fromJSON(needs.discover.outputs.matrix) }}
|
||||
env:
|
||||
OP_SERVICE_ACCOUNT_TOKEN: ${{ secrets.OP_TF_YUCCA_PROD_ENV }}
|
||||
FABRIC_SITE: ${{ matrix.site }}
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
with:
|
||||
persist-credentials: false
|
||||
- name: Set up mise (go + opentofu + terragrunt)
|
||||
uses: jdx/mise-action@e6a8b3978addb5a52f2b4cd9d91eafa7f0ab959d # v4.2.0
|
||||
- name: Install 1Password CLI
|
||||
uses: 1password/install-cli-action@a5215d3a7f75c1629216c465ea9ab3ab399c4b71 # v4.0.0
|
||||
- name: Connect to Tailscale
|
||||
uses: tailscale/github-action@306e68a486fd2350f2bfc3b19fcd143891a4a2d8 # v4.1.2
|
||||
with:
|
||||
oauth-client-id: ${{ secrets.TS_OAUTH_CLIENT_ID }}
|
||||
oauth-secret: ${{ secrets.TS_OAUTH_SECRET }}
|
||||
tags: tag:project-yucca
|
||||
- name: Fabric plan
|
||||
run: mise run fabric:plan -- --non-interactive
|
||||
|
||||
apply:
|
||||
name: Apply ${{ matrix.site }} (gated)
|
||||
needs: [discover, plan]
|
||||
if: github.event_name == 'push' || github.event_name == 'workflow_dispatch'
|
||||
runs-on: ubuntu-latest
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix: ${{ fromJSON(needs.discover.outputs.matrix) }}
|
||||
# Site-scoped gate: each site's prod fabric has its own required reviewers.
|
||||
environment:
|
||||
name: prod-fabric-${{ matrix.site }}
|
||||
env:
|
||||
OP_SERVICE_ACCOUNT_TOKEN: ${{ secrets.OP_TF_YUCCA_PROD_ENV_WRITE }}
|
||||
FABRIC_SITE: ${{ matrix.site }}
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
with:
|
||||
persist-credentials: false
|
||||
- name: Set up mise (go + opentofu + terragrunt)
|
||||
uses: jdx/mise-action@e6a8b3978addb5a52f2b4cd9d91eafa7f0ab959d # v4.2.0
|
||||
- name: Install 1Password CLI
|
||||
uses: 1password/install-cli-action@a5215d3a7f75c1629216c465ea9ab3ab399c4b71 # v4.0.0
|
||||
- name: Connect to Tailscale
|
||||
uses: tailscale/github-action@306e68a486fd2350f2bfc3b19fcd143891a4a2d8 # v4.1.2
|
||||
with:
|
||||
oauth-client-id: ${{ secrets.TS_OAUTH_CLIENT_ID }}
|
||||
oauth-secret: ${{ secrets.TS_OAUTH_SECRET }}
|
||||
tags: tag:project-yucca
|
||||
- name: Fabric apply
|
||||
run: mise run fabric:apply -- --non-interactive -auto-approve
|
||||
+243
-59
@@ -1,45 +1,53 @@
|
||||
name: Infra (Terraform)
|
||||
|
||||
# Applies the staging Terraform stacks from CI, then converges the bare-metal
|
||||
# Ceph cluster with Ansible:
|
||||
# - tf/deployment/staging/ceph (ceph cluster 1P password items; no node contact)
|
||||
# - tf/deployment/staging/talos (Talos config, Cilium, Flux bootstrap, secrets)
|
||||
# - tf/deployment/staging/dns (Cloudflare records for the ingress hosts)
|
||||
# - tf/deployment/staging/netbird (NetBird Cloud groups/policies/setup keys)
|
||||
# - tf/deployment/prod/global + prod/htz-fsn1/netbird (prod NetBird, layered:
|
||||
# a global layer + per-site layers — runs as its own gated jobs at the bottom
|
||||
# of this file, on the prod 1P SA / tf/.env.prod)
|
||||
# - ansible/ceph (deploy pipeline) (cephadm convergence: OSDs, RGW realm +
|
||||
# S3/metrics-worker users, monitoring, tuning, hardening) — the TF stacks
|
||||
# only MINT the RGW keys into 1P; this step is what actually creates the
|
||||
# matching RGW users on the cluster, so the metrics worker can authenticate.
|
||||
# Applies the Terraform stacks from CI, path-scoped so each group only runs when
|
||||
# its own files change (a `changes` job emits per-area booleans that gate the
|
||||
# rest; workflow_dispatch overrides and runs everything). Connectivity to the
|
||||
# bare-metal/private nodes is over the NetBird overlay (Tailscale fully retired):
|
||||
#
|
||||
# Plan runs on PRs touching tf/** or ansible/ceph/**; apply runs on merge to main
|
||||
# behind the `staging-infra` Environment gate (required reviewers). Both the Talos
|
||||
# stack and the Ansible deploy talk to the nodes on the 10.10.10.0/24 management
|
||||
# VLAN, which GitHub-hosted runners can't reach — so the runner joins the NetBird
|
||||
# overlay as a `ci` peer (via the netbird-connect action + the minted CI setup
|
||||
# key) and the existing staging route advertises that VLAN. (The prod fabric
|
||||
# workflow still uses Tailscale; 10.40.5.0/24 isn't on NetBird yet.)
|
||||
# Staging (tf/deployment/staging/*): ceph, talos, dns, netbird. One staging SA,
|
||||
# a single `staging-infra` Environment gate. The netbird stack mints the CI
|
||||
# setup key; the apply joins the overlay as a `ci` peer to reach the
|
||||
# 10.10.10.0/24 nodes, then converges the bare-metal Ceph cluster (ansible/ceph).
|
||||
#
|
||||
# Prod fabric+mgmt (tf/deployment/prod/<site>): the switch fabric + mgmt hosts
|
||||
# (junos-qfx + hetzner providers, built locally). Per-site `prod-<site>` gate.
|
||||
# Reaches the switch vme / mgmt hosts over NetBird (the mgmt nodes are the
|
||||
# route peers for 10.40.5.0/24 et al.); runs the `infra:*` / `mgmt:*` mise tasks.
|
||||
#
|
||||
# Prod NetBird (tf/deployment/prod/global + prod/<site>/netbird): account-wide +
|
||||
# site NetBird groups/keys/policies/routes. Pure api.netbird.io. `prod-infra` gate.
|
||||
#
|
||||
# Prerequisites (provisioned out-of-band):
|
||||
# - Repo secrets: OP_TF_YUCCA_STAGING_ENV (the staging-scoped 1P service-account
|
||||
# token — resolves tf/.env; dev/prod use OP_TF_YUCCA_DEV_ENV / OP_TF_YUCCA_PROD_ENV).
|
||||
# - BOOTSTRAP — the staging NetBird stack applied ONCE out-of-band, so the CI
|
||||
# setup key (op://yucca_tf_staging/NETBIRD_YUCCA_STAGING_CI_SETUP_KEY) exists
|
||||
# before any talos plan tries to connect. CI can't mint it itself: the apply
|
||||
# that mints it is gated behind the plan that needs it. Run once locally:
|
||||
# - Repo secrets: OP_TF_YUCCA_STAGING_ENV (+ _WRITE) — staging SAs; and
|
||||
# OP_TF_YUCCA_PROD_ENV (read) / OP_TF_YUCCA_PROD_ENV_WRITE (netbird apply) — prod
|
||||
# SAs. (The fabric apply escalates to the write SA stored in yucca_tf_prod.)
|
||||
# - BOOTSTRAP — the netbird stacks applied ONCE out-of-band so the CI/mgmt setup
|
||||
# keys exist in 1P before anything tries to connect (CI can't mint them itself:
|
||||
# the apply that mints them is gated behind the plan that needs them). E.g.:
|
||||
# OP_SERVICE_ACCOUNT_TOKEN=<staging write SA> \
|
||||
# TF_STACK_DIR=tf/deployment/staging/netbird mise run tf:apply
|
||||
# Also: a NetBird route advertising 10.10.10.0/24 that the `ci` group may reach.
|
||||
# - GitHub Environment `staging-infra` with required reviewers (the apply gate).
|
||||
# (staging: NETBIRD_YUCCA_STAGING_CI_SETUP_KEY; prod: NETBIRD_YUCCA_PROD_<SITE>_*).
|
||||
# Also: NetBird routes advertising the node subnets (staging 10.10.10.0/24; prod
|
||||
# the site subnets via the mgmt peers) that the `ci` group may reach.
|
||||
# - 1P items: NET_SWITCHES_TERRAFORM_SSH_PRIVATE_KEY (yucca_tf_prod),
|
||||
# NETBOX_API_TOKEN (yucca_tf), HETZNER_WEBSERVICE_API_USER/PASSWORD (yucca_tf_prod).
|
||||
# - GitHub Environments with required reviewers: `staging-infra`, `prod-infra`,
|
||||
# and one `prod-<site>` per prod fabric site (e.g. prod-htz-fsn1).
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
paths: ["tf/**", "ansible/ceph/**"]
|
||||
paths: &paths
|
||||
- "tf/**"
|
||||
- "ansible/mgmt/**"
|
||||
- "ansible/ceph/**"
|
||||
- ".mise/tasks/**"
|
||||
- ".mise/config.toml"
|
||||
- ".github/actions/netbird-connect/**"
|
||||
- ".github/workflows/infra.yml"
|
||||
pull_request:
|
||||
paths: ["tf/**", "ansible/ceph/**"]
|
||||
paths: *paths
|
||||
workflow_dispatch:
|
||||
|
||||
# Serialize: the OVH S3 backend has no state locking (single-operator model),
|
||||
@@ -51,20 +59,75 @@ concurrency:
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
env:
|
||||
# read-scoped SA for plan; apply overrides with the write SA
|
||||
OP_SERVICE_ACCOUNT_TOKEN: ${{ secrets.OP_TF_YUCCA_STAGING_ENV }}
|
||||
|
||||
jobs:
|
||||
plan:
|
||||
name: Plan ${{ matrix.stack }}
|
||||
# Skip on fork PRs (no access to secrets / the NetBird overlay).
|
||||
# ── Which areas changed? Outputs gate every downstream job. ──────────────────
|
||||
changes:
|
||||
name: Detect changes
|
||||
# Skip on fork PRs (no access to secrets / the overlay anyway).
|
||||
if: github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
staging: ${{ steps.filter.outputs.staging }}
|
||||
prod_tf: ${{ steps.filter.outputs.prod_tf }}
|
||||
prod_ansible: ${{ steps.filter.outputs.prod_ansible }}
|
||||
prod_netbird: ${{ steps.filter.outputs.prod_netbird }}
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
with:
|
||||
persist-credentials: false
|
||||
- uses: dorny/paths-filter@de90cc6fb38fc0963ad72b210f1f284cd68cea36 # v3.0.2
|
||||
id: filter
|
||||
with:
|
||||
filters: |
|
||||
staging:
|
||||
- 'tf/deployment/staging/**'
|
||||
- 'tf/shared/**'
|
||||
- 'tf/op-run.sh'
|
||||
- 'ansible/ceph/**'
|
||||
- '.github/actions/netbird-connect/**'
|
||||
- '.github/workflows/infra.yml'
|
||||
prod_tf:
|
||||
# Fabric/mgmt stacks. Netbird-only dirs are covered by prod_netbird;
|
||||
# an overlap just means a (gated) extra fabric plan — harmless.
|
||||
- 'tf/deployment/prod/**'
|
||||
- '!tf/deployment/prod/global/**'
|
||||
- '!tf/deployment/prod/*/netbird/**'
|
||||
- 'tf/shared/**'
|
||||
- 'tf/providers/**'
|
||||
- '.mise/tasks/infra/**'
|
||||
- '.mise/tasks/fabric/**'
|
||||
- '.mise/tasks/mgmt/**'
|
||||
- '.mise/config.toml'
|
||||
- '.github/actions/netbird-connect/**'
|
||||
- '.github/workflows/infra.yml'
|
||||
prod_ansible:
|
||||
- 'ansible/mgmt/**'
|
||||
- 'tf/render/**'
|
||||
- 'tf/shared/modules/identity/**'
|
||||
- 'tf/shared/modules/fabric-addressing/**'
|
||||
- 'tf/deployment/prod/*/mgmt-hosts.yaml'
|
||||
- '.mise/tasks/mgmt/**'
|
||||
- '.mise/config.toml'
|
||||
- '.github/actions/netbird-connect/**'
|
||||
- '.github/workflows/infra.yml'
|
||||
prod_netbird:
|
||||
- 'tf/deployment/prod/global/**'
|
||||
- 'tf/deployment/prod/*/netbird/**'
|
||||
- 'tf/shared/modules/netbird-env/**'
|
||||
- '.github/workflows/infra.yml'
|
||||
|
||||
# ── Staging stacks ──────────────────────────────────────────────────────────
|
||||
staging-plan:
|
||||
name: Staging plan ${{ matrix.stack }}
|
||||
needs: changes
|
||||
if: needs.changes.outputs.staging == 'true' || github.event_name == 'workflow_dispatch'
|
||||
runs-on: ubuntu-latest
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
stack: [talos, dns, ceph, netbird]
|
||||
env:
|
||||
OP_SERVICE_ACCOUNT_TOKEN: ${{ secrets.OP_TF_YUCCA_STAGING_ENV }}
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
@@ -98,11 +161,11 @@ jobs:
|
||||
--working-dir tf/deployment/staging/${{ matrix.stack }}
|
||||
--non-interactive plan
|
||||
|
||||
apply:
|
||||
name: Apply (gated)
|
||||
needs: plan
|
||||
staging-apply:
|
||||
name: Staging apply (gated)
|
||||
needs: [changes, staging-plan]
|
||||
# Only on merge to main (or manual dispatch) — never on PRs.
|
||||
if: github.event_name == 'push' || github.event_name == 'workflow_dispatch'
|
||||
if: (github.event_name == 'push' && needs.changes.outputs.staging == 'true') || github.event_name == 'workflow_dispatch'
|
||||
runs-on: ubuntu-latest
|
||||
# The approval gate: `staging-infra` Environment with required reviewers.
|
||||
environment: staging-infra
|
||||
@@ -132,7 +195,7 @@ jobs:
|
||||
|
||||
# Join the NetBird overlay as a `ci` peer so the node-touching stacks and
|
||||
# the Ansible deploy below reach 10.10.10.0/24 (the staging route advertises
|
||||
# the LAN). Replaces the old Tailscale subnet-router path.
|
||||
# the LAN).
|
||||
- name: Connect to NetBird
|
||||
uses: ./.github/actions/netbird-connect
|
||||
with:
|
||||
@@ -165,7 +228,6 @@ jobs:
|
||||
# cluster. Runs after the TF apply so the inventory (rendered from the
|
||||
# ceph stack's `render` output) and the keys exist. Reuses the NetBird
|
||||
# overlay + 1Password session already established in this job.
|
||||
|
||||
- name: Render Ansible inventory from the ceph TF state
|
||||
run: ansible/ceph/scripts/render-inventories.sh staging
|
||||
|
||||
@@ -192,22 +254,16 @@ jobs:
|
||||
CEPH_ENV: inventories/sietch-ceph.staging.austin.int/inventory.ini
|
||||
run: mise run deploy
|
||||
|
||||
# ── prod NetBird (global + site layers) ──────────────────────────────────────
|
||||
# Prod is its own env (separate 1P SA + env file), so it can't ride the
|
||||
# staging matrix above (that job's OP_SERVICE_ACCOUNT_TOKEN is the staging SA).
|
||||
# NetBird is pure api.netbird.io — no nodes, no tailnet — so these are the only
|
||||
# prod TF stacks CI touches today. Layered: prod/global (account-wide groups +
|
||||
# operator setup keys) then prod/<site>/netbird (site groups/keys/policies that
|
||||
# reference the global groups). Both select tf/.env.prod via OP_ENV_FILE.
|
||||
#
|
||||
# Additional prerequisites (out-of-band) beyond the staging ones above:
|
||||
# - Repo secrets OP_TF_YUCCA_PROD_ENV (read) / OP_TF_YUCCA_PROD_ENV_WRITE
|
||||
# (apply) — a prod-scoped 1P service account granted shared_tf (read, for
|
||||
# NETBIRD_TF_PAT) + yucca_tf_prod (read/write, for the minted setup keys).
|
||||
# - GitHub Environment `prod-infra` with required reviewers (the apply gate).
|
||||
# ── Prod NetBird (global + site layers) — mints the CI/mgmt setup keys ────────
|
||||
# Pure api.netbird.io (no nodes/overlay). Layered: prod/global (account-wide
|
||||
# groups + the yucca→yucca_resource policy) then prod/<site>/netbird (site
|
||||
# groups/keys/policies + the routed network). Runs before the fabric/mgmt jobs
|
||||
# conceptually (it mints the keys they consume), but they're only loosely coupled
|
||||
# — the keys persist in 1P across runs, so a fresh bootstrap applies this first.
|
||||
netbird-prod-plan:
|
||||
name: Plan prod/netbird
|
||||
if: github.event_name != 'pull_request' || github.event.pull_request.head.repo.full_name == github.repository
|
||||
needs: changes
|
||||
if: needs.changes.outputs.prod_netbird == 'true' || github.event_name == 'workflow_dispatch'
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
OP_ENV_FILE: tf/.env.prod
|
||||
@@ -241,8 +297,8 @@ jobs:
|
||||
|
||||
netbird-prod-apply:
|
||||
name: Apply prod/netbird (gated)
|
||||
needs: netbird-prod-plan
|
||||
if: github.event_name == 'push' || github.event_name == 'workflow_dispatch'
|
||||
needs: [changes, netbird-prod-plan]
|
||||
if: (github.event_name == 'push' && needs.changes.outputs.prod_netbird == 'true') || github.event_name == 'workflow_dispatch'
|
||||
runs-on: ubuntu-latest
|
||||
environment: prod-infra
|
||||
env:
|
||||
@@ -273,3 +329,131 @@ jobs:
|
||||
tf/op-run.sh terragrunt
|
||||
--working-dir tf/deployment/prod/htz-fsn1/netbird
|
||||
--non-interactive apply -auto-approve
|
||||
|
||||
# ── Prod fabric + mgmt stacks (one per site) ─────────────────────────────────
|
||||
prod-discover:
|
||||
name: Discover prod fabric sites
|
||||
needs: changes
|
||||
# Needed by both the TF and ansible prod jobs, so run if either area changed.
|
||||
if: needs.changes.outputs.prod_tf == 'true' || needs.changes.outputs.prod_ansible == 'true' || github.event_name == 'workflow_dispatch'
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
matrix: ${{ steps.sites.outputs.matrix }}
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
with:
|
||||
persist-credentials: false
|
||||
- id: sites
|
||||
name: List fabric sites (dirs with a fabric.tf; excludes netbird-only dirs)
|
||||
run: |
|
||||
matrix=$(for d in tf/deployment/prod/*/; do [ -f "${d}fabric.tf" ] && basename "$d"; done | jq -R . | jq -cs '{site: .}')
|
||||
echo "matrix=$matrix" >> "$GITHUB_OUTPUT"
|
||||
echo "$matrix"
|
||||
|
||||
prod-plan:
|
||||
name: Prod plan ${{ matrix.site }}
|
||||
needs: [changes, prod-discover]
|
||||
if: needs.changes.outputs.prod_tf == 'true' || github.event_name == 'workflow_dispatch'
|
||||
runs-on: ubuntu-latest
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix: ${{ fromJSON(needs.prod-discover.outputs.matrix) }}
|
||||
env:
|
||||
OP_SERVICE_ACCOUNT_TOKEN: ${{ secrets.OP_TF_YUCCA_PROD_ENV }}
|
||||
SITE: ${{ matrix.site }}
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
with:
|
||||
persist-credentials: false
|
||||
- name: Set up mise (go + opentofu + terragrunt)
|
||||
uses: jdx/mise-action@e6a8b3978addb5a52f2b4cd9d91eafa7f0ab959d # v4.2.0
|
||||
- name: Install 1Password CLI
|
||||
uses: 1password/install-cli-action@a5215d3a7f75c1629216c465ea9ab3ab399c4b71 # v4.0.0
|
||||
# Reach the switch vme (routed via the mgmt NetBird peers) over the overlay.
|
||||
- name: Resolve NetBird CI setup-key ref
|
||||
run: echo "NB_CI_KEY_REF=op://yucca_tf_prod/NETBIRD_YUCCA_PROD_$(echo "$SITE" | tr 'a-z-' 'A-Z_')_CI_SETUP_KEY/password" >> "$GITHUB_ENV"
|
||||
- name: Connect to NetBird
|
||||
uses: ./.github/actions/netbird-connect
|
||||
with:
|
||||
setup-key-ref: ${{ env.NB_CI_KEY_REF }}
|
||||
hostname: gha-prod-plan-${{ matrix.site }}-${{ github.run_id }}
|
||||
- name: Deploy plan
|
||||
run: mise run infra:plan -- --non-interactive
|
||||
|
||||
prod-apply:
|
||||
name: Prod apply ${{ matrix.site }} (gated)
|
||||
needs: [changes, prod-discover, prod-plan]
|
||||
if: (github.event_name == 'push' && needs.changes.outputs.prod_tf == 'true') || github.event_name == 'workflow_dispatch'
|
||||
runs-on: ubuntu-latest
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix: ${{ fromJSON(needs.prod-discover.outputs.matrix) }}
|
||||
# Site-scoped gate: each site's prod stack has its own required reviewers.
|
||||
environment:
|
||||
name: prod-${{ matrix.site }}
|
||||
env:
|
||||
# Read-scoped token; infra:apply escalates to the write SA stored in the vault.
|
||||
OP_SERVICE_ACCOUNT_TOKEN: ${{ secrets.OP_TF_YUCCA_PROD_ENV }}
|
||||
SITE: ${{ matrix.site }}
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
with:
|
||||
persist-credentials: false
|
||||
- name: Set up mise (go + opentofu + terragrunt)
|
||||
uses: jdx/mise-action@e6a8b3978addb5a52f2b4cd9d91eafa7f0ab959d # v4.2.0
|
||||
- name: Install 1Password CLI
|
||||
uses: 1password/install-cli-action@a5215d3a7f75c1629216c465ea9ab3ab399c4b71 # v4.0.0
|
||||
- name: Resolve NetBird CI setup-key ref
|
||||
run: echo "NB_CI_KEY_REF=op://yucca_tf_prod/NETBIRD_YUCCA_PROD_$(echo "$SITE" | tr 'a-z-' 'A-Z_')_CI_SETUP_KEY/password" >> "$GITHUB_ENV"
|
||||
- name: Connect to NetBird
|
||||
uses: ./.github/actions/netbird-connect
|
||||
with:
|
||||
setup-key-ref: ${{ env.NB_CI_KEY_REF }}
|
||||
hostname: gha-prod-apply-${{ matrix.site }}-${{ github.run_id }}
|
||||
- name: Deploy apply
|
||||
run: mise run infra:apply -- --non-interactive -auto-approve
|
||||
|
||||
prod-ansible:
|
||||
name: Prod ansible ${{ matrix.site }}
|
||||
needs: [changes, prod-discover, prod-apply]
|
||||
# Runs after the TF apply, but also on ansible-only changes (apply skipped).
|
||||
# always() so a skipped prod-apply (TF unchanged) doesn't skip this; still
|
||||
# bails if discover failed or the apply actually failed.
|
||||
if: >-
|
||||
always()
|
||||
&& needs.prod-discover.result == 'success'
|
||||
&& needs.prod-apply.result != 'failure'
|
||||
&& needs.prod-apply.result != 'cancelled'
|
||||
&& ((github.event_name == 'push' && needs.changes.outputs.prod_ansible == 'true') || github.event_name == 'workflow_dispatch')
|
||||
runs-on: ubuntu-latest
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix: ${{ fromJSON(needs.prod-discover.outputs.matrix) }}
|
||||
# No environment gate: the prod-apply gate already approved this deploy, and
|
||||
# ansible/mgmt only reads from 1Password (read-scoped repo secret).
|
||||
env:
|
||||
OP_SERVICE_ACCOUNT_TOKEN: ${{ secrets.OP_TF_YUCCA_PROD_ENV }}
|
||||
SITE: ${{ matrix.site }}
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
with:
|
||||
persist-credentials: false
|
||||
- name: Set up mise (ansible + opentofu)
|
||||
uses: jdx/mise-action@e6a8b3978addb5a52f2b4cd9d91eafa7f0ab959d # v4.2.0
|
||||
- name: Install 1Password CLI
|
||||
uses: 1password/install-cli-action@a5215d3a7f75c1629216c465ea9ab3ab399c4b71 # v4.0.0
|
||||
# On the overlay so the playbook can reach mgmt hosts that have joined NetBird
|
||||
# (it prefers their NetBird IP, falling back to the public IP otherwise).
|
||||
- name: Resolve NetBird CI setup-key ref
|
||||
run: echo "NB_CI_KEY_REF=op://yucca_tf_prod/NETBIRD_YUCCA_PROD_$(echo "$SITE" | tr 'a-z-' 'A-Z_')_CI_SETUP_KEY/password" >> "$GITHUB_ENV"
|
||||
- name: Connect to NetBird
|
||||
uses: ./.github/actions/netbird-connect
|
||||
with:
|
||||
setup-key-ref: ${{ env.NB_CI_KEY_REF }}
|
||||
hostname: gha-prod-ansible-${{ matrix.site }}-${{ github.run_id }}
|
||||
# Renders the inventory from TF (tf/render/ansible-mgmt) then converges the
|
||||
# mgmt hosts — over NetBird if they've joined, else their public IP (the
|
||||
# TF-generated provisioning key authorizes root). No-op for sites without an
|
||||
# ansible/mgmt inventory; requires the hosts to have been reprovisioned.
|
||||
- name: Ansible converge (mgmt hosts)
|
||||
run: mise run mgmt:ansible
|
||||
|
||||
+15
-4
@@ -55,9 +55,20 @@ ansible/*/.venv/
|
||||
# on operator workstation, but not tracked).
|
||||
analysis/
|
||||
|
||||
# fabric (Junos/JTAF) terraform — generated artifacts
|
||||
tf/.terraformrc.fabric
|
||||
.mise/.fabric-provider-bin/
|
||||
.mise/.fabric-provider-mirror/
|
||||
# prod deployment-stack terraform — locally-built providers + generated artifacts
|
||||
tf/.terraformrc.local
|
||||
.mise/.provider-bin/
|
||||
.mise/.provider-mirror/
|
||||
.mise/.hetzner-provider-src/
|
||||
# self-built junos-qfx (mirror) makes this lock platform-specific; regenerated by init
|
||||
tf/deployment/prod/htz-fsn1/.terraform.lock.hcl
|
||||
|
||||
# ansible/mgmt inventory is TF-generated at run time (tf/render/ansible-mgmt) —
|
||||
# Terraform is the source of truth, not these files.
|
||||
ansible/mgmt/inventories/*/hosts.yml
|
||||
ansible/mgmt/inventories/*/host_vars/
|
||||
ansible/mgmt/inventories/*/group_vars/all/users.generated.yml
|
||||
# ephemeral local state for the render roots
|
||||
tf/render/*/.terraform/
|
||||
tf/render/*/.terraform.lock.hcl
|
||||
tf/render/*/terraform.tfstate*
|
||||
|
||||
@@ -24,6 +24,8 @@ uv = "0.9.18"
|
||||
# Infrastructure tooling (added for tf/ and ansible/ subtrees)
|
||||
opentofu = "1.11.5"
|
||||
terragrunt = "0.99.4"
|
||||
# ansible/mgmt convergence (mgmt:ansible task, run from CI on prod apply).
|
||||
"pipx:ansible-core" = "2.18.1"
|
||||
|
||||
[tasks.dev]
|
||||
description = "Start all services in development mode"
|
||||
|
||||
@@ -1,17 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
#MISE description="terragrunt apply for prod/htz-fsn1 fabric (builds provider, renders creds from 1Password)"
|
||||
set -euo pipefail
|
||||
ROOT=$(git rev-parse --show-toplevel)
|
||||
|
||||
mise run fabric:provider-build
|
||||
|
||||
KEYF=$(mktemp); chmod 600 "$KEYF"; trap 'rm -f "$KEYF"' EXIT
|
||||
ACCT=(); [ -z "${OP_SERVICE_ACCOUNT_TOKEN:-}" ] && ACCT=(--account "${OP_ACCOUNT:-team-futo}")
|
||||
op read "${ACCT[@]}" "op://yucca_tf_prod/NET_SWITCHES_TERRAFORM_SSH_PRIVATE_KEY/password" > "$KEYF"
|
||||
|
||||
export TF_VAR_netconf_ssh_key_path="$KEYF"
|
||||
export TF_CLI_CONFIG_FILE="$ROOT/tf/.terraformrc.fabric"
|
||||
SITE="${FABRIC_SITE:-htz-fsn1}"
|
||||
# -parallelism=1: the JTAF junos-qfx provider isn't concurrency-safe — parallel
|
||||
# ApplyResourceChange calls across the VCs crash the plugin ("Plugin did not respond").
|
||||
OP_ENV_FILE=tf/.env.prod "$ROOT/tf/op-run.sh" terragrunt --working-dir "tf/deployment/prod/$SITE" apply -parallelism=1 "$@"
|
||||
@@ -1,18 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
#MISE description="terragrunt plan for prod/htz-fsn1 fabric (builds provider, renders creds from 1Password)"
|
||||
set -euo pipefail
|
||||
ROOT=$(git rev-parse --show-toplevel)
|
||||
|
||||
mise run fabric:provider-build
|
||||
|
||||
# Render the NETCONF SSH key from 1Password to a 0600 temp file (op run can't
|
||||
# write files). Use --account only for interactive logins; CI uses an SA token.
|
||||
KEYF=$(mktemp); chmod 600 "$KEYF"; trap 'rm -f "$KEYF"' EXIT
|
||||
ACCT=(); [ -z "${OP_SERVICE_ACCOUNT_TOKEN:-}" ] && ACCT=(--account "${OP_ACCOUNT:-team-futo}")
|
||||
op read "${ACCT[@]}" "op://yucca_tf_prod/NET_SWITCHES_TERRAFORM_SSH_PRIVATE_KEY/password" > "$KEYF"
|
||||
|
||||
export TF_VAR_netconf_ssh_key_path="$KEYF"
|
||||
export TF_CLI_CONFIG_FILE="$ROOT/tf/.terraformrc.fabric"
|
||||
SITE="${FABRIC_SITE:-htz-fsn1}"
|
||||
# -parallelism=1: the JTAF junos-qfx provider isn't concurrency-safe (see apply).
|
||||
OP_ENV_FILE=tf/.env.prod "$ROOT/tf/op-run.sh" terragrunt --working-dir "tf/deployment/prod/$SITE" plan -parallelism=1 "$@"
|
||||
@@ -1,33 +1,20 @@
|
||||
#!/usr/bin/env bash
|
||||
#MISE description="Build the vendored JTAF junos-qfx provider into a filesystem mirror + write its terraformrc"
|
||||
#MISE description="Build the vendored JTAF junos-qfx provider binary into the shared provider mirror"
|
||||
set -euo pipefail
|
||||
ROOT=$(git rev-parse --show-toplevel)
|
||||
SRC="$ROOT/tf/providers/terraform-provider-junos-qfx"
|
||||
MIRROR="${FABRIC_PROVIDER_MIRROR:-$ROOT/.mise/.fabric-provider-mirror}"
|
||||
VER="${FABRIC_PROVIDER_VERSION:-0.0.1}"
|
||||
MIRROR="${PROVIDER_MIRROR:-$ROOT/.mise/.provider-mirror}"
|
||||
VER="${JUNOS_PROVIDER_VERSION:-0.0.1}"
|
||||
|
||||
GOOS=$(go env GOOS); GOARCH=$(go env GOARCH)
|
||||
BIN="terraform-provider-junos-qfx_v${VER}"
|
||||
|
||||
# Build into the unpacked filesystem-mirror layout for both registry hosts
|
||||
# (OpenTofu defaults to registry.opentofu.org; Terraform to registry.terraform.io).
|
||||
# The shared terraformrc that wires this mirror up is written by `infra:providers`.
|
||||
for HOST in registry.opentofu.org registry.terraform.io; do
|
||||
DEST="$MIRROR/$HOST/hashicorp/junos-qfx/$VER/${GOOS}_${GOARCH}"
|
||||
mkdir -p "$DEST"
|
||||
( cd "$SRC" && go build -o "$DEST/$BIN" . )
|
||||
done
|
||||
|
||||
cat > "$ROOT/tf/.terraformrc.fabric" <<EOF
|
||||
# Generated by 'mise run fabric:provider-build' — do not edit.
|
||||
provider_installation {
|
||||
filesystem_mirror {
|
||||
path = "$MIRROR"
|
||||
include = ["registry.opentofu.org/hashicorp/junos-qfx", "registry.terraform.io/hashicorp/junos-qfx"]
|
||||
}
|
||||
direct {
|
||||
exclude = ["registry.opentofu.org/hashicorp/junos-qfx", "registry.terraform.io/hashicorp/junos-qfx"]
|
||||
}
|
||||
}
|
||||
EOF
|
||||
echo "built junos-qfx v$VER (${GOOS}_${GOARCH}) -> $MIRROR"
|
||||
echo "wrote filesystem_mirror -> tf/.terraformrc.fabric"
|
||||
|
||||
Executable
+38
@@ -0,0 +1,38 @@
|
||||
#!/usr/bin/env bash
|
||||
#MISE description="terragrunt apply for a prod deployment stack (builds providers, escalates to the write SA, renders creds from 1Password). SITE selects the stack."
|
||||
set -euo pipefail
|
||||
ROOT=$(git rev-parse --show-toplevel)
|
||||
|
||||
# OpenTofu's Go binary doesn't read the macOS keychain, so the OVH S3 backend's
|
||||
# cert fails to verify locally ("unknown authority"). Point it at the system CA
|
||||
# bundle on macOS; Linux/CI reads its trust store fine and is left alone.
|
||||
if [ -z "${SSL_CERT_FILE:-}" ] && [ "$(uname -s)" = "Darwin" ] && [ -f /etc/ssl/cert.pem ]; then
|
||||
export SSL_CERT_FILE=/etc/ssl/cert.pem AWS_CA_BUNDLE=/etc/ssl/cert.pem
|
||||
fi
|
||||
|
||||
mise run infra:providers
|
||||
|
||||
# Escalate to the write-capable service-account token, which is itself stored in
|
||||
# the vault: CI hands us a read-scoped token in OP_SERVICE_ACCOUNT_TOKEN (locally
|
||||
# we sign in to team-futo), and we use it to read the write SA, then run the apply
|
||||
# as that SA so the onepassword provider can create/update items. One GitHub
|
||||
# secret (the read token) is enough; the write privilege lives in 1Password.
|
||||
ACCT=(); [ -z "${OP_SERVICE_ACCOUNT_TOKEN:-}" ] && ACCT=(--account "${OP_ACCOUNT:-team-futo}")
|
||||
OP_SERVICE_ACCOUNT_TOKEN=$(op read "${ACCT[@]}" \
|
||||
"op://yucca_tf_prod/yucca_futo_1pass_service_account_write/password")
|
||||
export OP_SERVICE_ACCOUNT_TOKEN
|
||||
|
||||
# Render the NETCONF SSH key from 1Password to a 0600 temp file (op run can't
|
||||
# write files) — now via the escalated SA.
|
||||
KEYF=$(mktemp); chmod 600 "$KEYF"; trap 'rm -f "$KEYF"' EXIT
|
||||
op read "op://yucca_tf_prod/NET_SWITCHES_TERRAFORM_SSH_PRIVATE_KEY/password" > "$KEYF"
|
||||
|
||||
export TF_VAR_netconf_ssh_key_path="$KEYF"
|
||||
export TF_CLI_CONFIG_FILE="$ROOT/tf/.terraformrc.local"
|
||||
SITE="${SITE:-htz-fsn1}"
|
||||
# With a service-account token in the env, drop OP_ACCOUNT — the onepassword
|
||||
# provider rejects having both set ("service_account_token and account are set").
|
||||
unset OP_ACCOUNT
|
||||
# -parallelism=1: the JTAF junos-qfx provider isn't concurrency-safe — parallel
|
||||
# ApplyResourceChange calls across the VCs crash the plugin ("Plugin did not respond").
|
||||
OP_ENV_FILE=tf/.env.prod "$ROOT/tf/op-run.sh" terragrunt --working-dir "tf/deployment/prod/$SITE" apply -parallelism=1 "$@"
|
||||
Executable
+37
@@ -0,0 +1,37 @@
|
||||
#!/usr/bin/env bash
|
||||
#MISE description="terragrunt plan for a prod deployment stack (builds providers, renders creds from 1Password). SITE selects the stack."
|
||||
set -euo pipefail
|
||||
ROOT=$(git rev-parse --show-toplevel)
|
||||
|
||||
# OpenTofu's Go binary doesn't read the macOS keychain, so the OVH S3 backend's
|
||||
# cert fails to verify locally ("unknown authority"). Point it at the system CA
|
||||
# bundle on macOS; Linux/CI reads its trust store fine and is left alone.
|
||||
if [ -z "${SSL_CERT_FILE:-}" ] && [ "$(uname -s)" = "Darwin" ] && [ -f /etc/ssl/cert.pem ]; then
|
||||
export SSL_CERT_FILE=/etc/ssl/cert.pem AWS_CA_BUNDLE=/etc/ssl/cert.pem
|
||||
fi
|
||||
|
||||
mise run infra:providers
|
||||
|
||||
# Resolve a (read-scoped) service-account token. CI provides one in
|
||||
# OP_SERVICE_ACCOUNT_TOKEN; locally we read it from the vault via an interactive
|
||||
# team-futo sign-in. Exporting it makes every downstream `op`/`op run` use that
|
||||
# SA (and the right account) — plan stays read-only, so no write escalation.
|
||||
if [ -z "${OP_SERVICE_ACCOUNT_TOKEN:-}" ]; then
|
||||
OP_SERVICE_ACCOUNT_TOKEN=$(op read --account "${OP_ACCOUNT:-team-futo}" \
|
||||
"op://yucca_tf_prod/yucca_futo_1pass_service_account/password")
|
||||
export OP_SERVICE_ACCOUNT_TOKEN
|
||||
fi
|
||||
|
||||
# Render the NETCONF SSH key from 1Password to a 0600 temp file (op run can't
|
||||
# write files).
|
||||
KEYF=$(mktemp); chmod 600 "$KEYF"; trap 'rm -f "$KEYF"' EXIT
|
||||
op read "op://yucca_tf_prod/NET_SWITCHES_TERRAFORM_SSH_PRIVATE_KEY/password" > "$KEYF"
|
||||
|
||||
export TF_VAR_netconf_ssh_key_path="$KEYF"
|
||||
export TF_CLI_CONFIG_FILE="$ROOT/tf/.terraformrc.local"
|
||||
SITE="${SITE:-htz-fsn1}"
|
||||
# With a service-account token in the env, drop OP_ACCOUNT — the onepassword
|
||||
# provider rejects having both set ("service_account_token and account are set").
|
||||
unset OP_ACCOUNT
|
||||
# -parallelism=1: the JTAF junos-qfx provider isn't concurrency-safe (see apply).
|
||||
OP_ENV_FILE=tf/.env.prod "$ROOT/tf/op-run.sh" terragrunt --working-dir "tf/deployment/prod/$SITE" plan -parallelism=1 "$@"
|
||||
Executable
+34
@@ -0,0 +1,34 @@
|
||||
#!/usr/bin/env bash
|
||||
#MISE description="Build all of the deployment stack's local providers (junos-qfx + hetzner) into a filesystem mirror + write its terraformrc"
|
||||
set -euo pipefail
|
||||
ROOT=$(git rev-parse --show-toplevel)
|
||||
MIRROR="${PROVIDER_MIRROR:-$ROOT/.mise/.provider-mirror}"
|
||||
|
||||
# Per-provider component builders (each builds one binary into the mirror).
|
||||
mise run fabric:provider-build # junos-qfx (switch fabric)
|
||||
mise run mgmt:provider-build # hetzner (mgmt-host reprovisioning)
|
||||
|
||||
# A single terraformrc points TF/tofu at the mirror for every locally-built
|
||||
# provider, for both registry hosts (OpenTofu -> registry.opentofu.org,
|
||||
# Terraform -> registry.terraform.io). Consumed via TF_CLI_CONFIG_FILE.
|
||||
cat > "$ROOT/tf/.terraformrc.local" <<'EOF'
|
||||
# Generated by 'mise run infra:providers' — do not edit.
|
||||
provider_installation {
|
||||
filesystem_mirror {
|
||||
path = "__MIRROR__"
|
||||
include = [
|
||||
"registry.opentofu.org/hashicorp/junos-qfx", "registry.terraform.io/hashicorp/junos-qfx",
|
||||
"registry.opentofu.org/zack/hetzner", "registry.terraform.io/zack/hetzner",
|
||||
]
|
||||
}
|
||||
direct {
|
||||
exclude = [
|
||||
"registry.opentofu.org/hashicorp/junos-qfx", "registry.terraform.io/hashicorp/junos-qfx",
|
||||
"registry.opentofu.org/zack/hetzner", "registry.terraform.io/zack/hetzner",
|
||||
]
|
||||
}
|
||||
}
|
||||
EOF
|
||||
# shellcheck disable=SC2016
|
||||
sed -i.bak "s#__MIRROR__#${MIRROR}#" "$ROOT/tf/.terraformrc.local" && rm -f "$ROOT/tf/.terraformrc.local.bak"
|
||||
echo "wrote filesystem_mirror -> tf/.terraformrc.local"
|
||||
Executable
+36
@@ -0,0 +1,36 @@
|
||||
#!/usr/bin/env bash
|
||||
#MISE description="Converge a site's management hosts with ansible/mgmt (root via the TF provisioning key from 1Password). SITE selects the inventory."
|
||||
set -euo pipefail
|
||||
ROOT=$(git rev-parse --show-toplevel)
|
||||
SITE="${SITE:-htz-fsn1}"
|
||||
INV="$ROOT/ansible/mgmt/inventories/$SITE"
|
||||
|
||||
# Per-site: only run where an inventory exists (keeps the prod matrix happy).
|
||||
if [ ! -d "$INV" ]; then
|
||||
echo "mgmt:ansible: no ansible/mgmt inventory for site '$SITE' — skipping."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Generate the inventory (hosts, host_vars, users) from Terraform's SoT.
|
||||
mise run mgmt:render-inventory
|
||||
|
||||
# op creds: CI provides a (read-scoped) SA token in OP_SERVICE_ACCOUNT_TOKEN;
|
||||
# locally we sign in to team-futo. Reads only — no write escalation needed.
|
||||
ACCT=(); [ -z "${OP_SERVICE_ACCOUNT_TOKEN:-}" ] && ACCT=(--account "${OP_ACCOUNT:-team-futo}")
|
||||
|
||||
# Provisioning private key (root login) -> 0600 temp file. Item name is
|
||||
# site-derived: htz-fsn1 -> HTZ_FSN1_PROVISIONING_SSH_PRIVATE_KEY (set by mgmt.tf).
|
||||
KEY_ITEM="$(printf '%s' "$SITE" | tr 'a-z-' 'A-Z_')_PROVISIONING_SSH_PRIVATE_KEY"
|
||||
KEYF=$(mktemp); chmod 600 "$KEYF"; trap 'rm -f "$KEYF"' EXIT
|
||||
op read "${ACCT[@]}" "op://yucca_tf_prod/$KEY_ITEM/password" > "$KEYF"
|
||||
|
||||
# NetBird "mgmt" setup key (auto_groups=["mgmt"]) — joins the node to the overlay
|
||||
# as a route peer. Site-derived item: htz-fsn1 -> NETBIRD_YUCCA_PROD_HTZ_FSN1_MGMT_SETUP_KEY.
|
||||
NB_KEY_ITEM="NETBIRD_YUCCA_PROD_$(printf '%s' "$SITE" | tr 'a-z-' 'A-Z_')_MGMT_SETUP_KEY"
|
||||
NB_SETUP_KEY=$(op read "${ACCT[@]}" "op://yucca_tf_prod/$NB_KEY_ITEM/password")
|
||||
|
||||
cd "$ROOT/ansible/mgmt"
|
||||
ansible-galaxy collection install -r requirements.yml >/dev/null
|
||||
ansible-playbook -i "inventories/$SITE" site.yml \
|
||||
--private-key "$KEYF" \
|
||||
--extra-vars "mgmt_netbird_setup_key=$NB_SETUP_KEY" "$@"
|
||||
Executable
+28
@@ -0,0 +1,28 @@
|
||||
#!/usr/bin/env bash
|
||||
#MISE description="Clone + build the Hetzner robot provider (zack/hetzner) into the shared provider mirror"
|
||||
set -euo pipefail
|
||||
ROOT=$(git rev-parse --show-toplevel)
|
||||
MIRROR="${PROVIDER_MIRROR:-$ROOT/.mise/.provider-mirror}"
|
||||
SRC="${HETZNER_PROVIDER_SRC:-$ROOT/.mise/.hetzner-provider-src}"
|
||||
REPO="${HETZNER_PROVIDER_REPO:-https://github.com/zackpollard/terraform-provider-hetzner.git}"
|
||||
REF="${HETZNER_PROVIDER_REF:-v0.1.0}"
|
||||
VER="${HETZNER_PROVIDER_VERSION:-0.1.0}"
|
||||
|
||||
# Pinned shallow clone of the upstream provider (not vendored — it's stock).
|
||||
if [ ! -d "$SRC/.git" ]; then
|
||||
git clone --depth 1 --branch "$REF" "$REPO" "$SRC"
|
||||
else
|
||||
( cd "$SRC" && git fetch --depth 1 origin "$REF" && git checkout -q FETCH_HEAD )
|
||||
fi
|
||||
|
||||
# Build into the unpacked filesystem-mirror layout for both registry hosts
|
||||
# (OpenTofu defaults to registry.opentofu.org; Terraform to registry.terraform.io).
|
||||
# Provider address is registry.<host>/zack/hetzner (see its main.go).
|
||||
GOOS=$(go env GOOS); GOARCH=$(go env GOARCH)
|
||||
BIN="terraform-provider-hetzner_v${VER}"
|
||||
for HOST in registry.opentofu.org registry.terraform.io; do
|
||||
DEST="$MIRROR/$HOST/zack/hetzner/$VER/${GOOS}_${GOARCH}"
|
||||
mkdir -p "$DEST"
|
||||
( cd "$SRC" && go build -o "$DEST/$BIN" . )
|
||||
done
|
||||
echo "built hetzner v$VER (${GOOS}_${GOARCH}) -> $MIRROR"
|
||||
Executable
+10
@@ -0,0 +1,10 @@
|
||||
#!/usr/bin/env bash
|
||||
#MISE description="Render a site's ansible/mgmt inventory from Terraform (tf/render/ansible-mgmt: addressing + identity + mgmt-hosts.yaml). SITE selects the site."
|
||||
set -euo pipefail
|
||||
ROOT=$(git rev-parse --show-toplevel)
|
||||
SITE="${SITE:-htz-fsn1}"
|
||||
|
||||
cd "$ROOT/tf/render/ansible-mgmt"
|
||||
tofu init -input=false >/dev/null
|
||||
tofu apply -input=false -auto-approve -var "site=$SITE" >/dev/null
|
||||
echo "rendered ansible/mgmt/inventories/$SITE/ (hosts.yml, host_vars/, group_vars/all/users.generated.yml)"
|
||||
@@ -0,0 +1,6 @@
|
||||
---
|
||||
# All roles share the mgmt_ variable prefix for consistency across the
|
||||
# stack. The var-naming[no-role-prefix] rule expects each role to use its
|
||||
# own prefix, which doesn't fit our single-product layout.
|
||||
skip_list:
|
||||
- var-naming[no-role-prefix]
|
||||
@@ -0,0 +1,29 @@
|
||||
root = true
|
||||
|
||||
[*]
|
||||
end_of_line = lf
|
||||
insert_final_newline = true
|
||||
trim_trailing_whitespace = true
|
||||
charset = utf-8
|
||||
|
||||
[*.{yml,yaml}]
|
||||
indent_style = space
|
||||
indent_size = 2
|
||||
|
||||
[*.{j2,jinja2}]
|
||||
indent_style = space
|
||||
indent_size = 2
|
||||
|
||||
[*.py]
|
||||
indent_style = space
|
||||
indent_size = 4
|
||||
|
||||
[*.{sh,bash}]
|
||||
indent_style = space
|
||||
indent_size = 2
|
||||
|
||||
[Makefile]
|
||||
indent_style = tab
|
||||
|
||||
[*.md]
|
||||
trim_trailing_whitespace = false
|
||||
@@ -0,0 +1,20 @@
|
||||
# Ansible
|
||||
ansible.log
|
||||
ansible.*.log
|
||||
*.retry
|
||||
.ansible/
|
||||
.ansible_facts_cache/
|
||||
|
||||
# Operator-local host_vars overrides
|
||||
inventories/*/host_vars/*.local.yml
|
||||
|
||||
# Python virtualenv and bytecode
|
||||
.venv/
|
||||
__pycache__/
|
||||
*.pyc
|
||||
|
||||
# OS / editor
|
||||
.DS_Store
|
||||
*.swp
|
||||
*.swo
|
||||
*~
|
||||
@@ -0,0 +1,12 @@
|
||||
---
|
||||
extends: relaxed
|
||||
|
||||
rules:
|
||||
line-length:
|
||||
max: 260
|
||||
allow-non-breakable-inline-mappings: true
|
||||
comments:
|
||||
min-spaces-from-content: 1
|
||||
octal-values:
|
||||
forbid-implicit-octal: true
|
||||
forbid-explicit-octal: true
|
||||
@@ -0,0 +1,143 @@
|
||||
# mgmt — Hetzner FSN1 Management Hosts
|
||||
|
||||
Ansible automation that configures the two Hetzner management hosts after they
|
||||
are reprovisioned to Debian 13 ("trixie"). Mirrors the conventions of the
|
||||
sibling `ansible/ceph/` tree (layout, `ansible.cfg`, role structure,
|
||||
systemd-networkd templating).
|
||||
|
||||
| Host | Public IP | Role |
|
||||
|------|-----------|------|
|
||||
| `htz-fsn-mgmt-1` | `178.63.124.40` | NetBird route peer (routes `10.40.5.0/24` et al.) |
|
||||
| `htz-fsn-mgmt-2` | `178.63.124.41` | NetBird route peer |
|
||||
|
||||
Both are AX41-NVMe, both join the NetBird `mgmt` group and route the site
|
||||
subnets (the routed network is declared in TF, `tf/deployment/prod/htz-fsn1/netbird`).
|
||||
The inventory is **TF-generated** at run time (see "Generated inventory").
|
||||
|
||||
## What it does
|
||||
|
||||
`site.yml` applies these roles in order:
|
||||
|
||||
1. **baseline** — apt cache + base packages, timezone UTC, hostname from
|
||||
inventory, `/etc/hosts`.
|
||||
2. **users** — login users for the members of the identity registry's
|
||||
`server`-mapped groups (`nutgood`, `andy` — both in `server_admins`,
|
||||
`sudo = ALL`). The `mgmt_users` list is **TF-generated** from
|
||||
`tf/shared/modules/identity` (see "Generated inventory" below).
|
||||
3. **security** — nftables firewall, SSH hardening (no password auth,
|
||||
`PermitRootLogin prohibit-password`), unattended-upgrades.
|
||||
4. **networkd** — systemd-networkd VLAN sub-interfaces on the 25G fabric NIC.
|
||||
**Gated on `mgmt_networkd_enabled` (default false)** — see the 25G caveat
|
||||
below.
|
||||
5. **netbird** — install NetBird, `netbird up` with the `mgmt` setup key, enable
|
||||
IP forwarding (the mgmt nodes are the NetBird route peers for the site subnets;
|
||||
the routed network itself is declared in TF).
|
||||
|
||||
## Generated inventory
|
||||
|
||||
The inventory is **generated by Terraform at run time**, not hand-written.
|
||||
`mise run mgmt:render-inventory` (run automatically by `mgmt:ansible`) invokes
|
||||
`tf/render/ansible-mgmt`, which derives everything from the single sources of
|
||||
truth and writes these **gitignored** files:
|
||||
|
||||
| file | generated from |
|
||||
|---|---|
|
||||
| `inventories/<site>/hosts.yml` | `mgmt-hosts.yaml` (host names + public IPs) |
|
||||
| `inventories/<site>/host_vars/*.yml` | `mgmt-hosts.yaml` (NIC) + `fabric-addressing` (VLAN ids/addresses, subnet route) |
|
||||
| `inventories/<site>/group_vars/all/users.generated.yml` | `tf/shared/modules/identity` (`server`-mapped users) |
|
||||
|
||||
Only `group_vars/all/main.yml` (static config) and `roles/**` are committed. To
|
||||
change hosts, addresses, or users, edit the Terraform sources — never the
|
||||
generated files. The render uses only the `local` provider (no backend, no
|
||||
secrets), so it runs anywhere.
|
||||
|
||||
## Reprovision → Ansible flow
|
||||
|
||||
1. Reprovision both hosts to Debian 13 via Hetzner robot auto-install.
|
||||
2. Post-reprovision, root is reachable over SSH with the TF-generated
|
||||
provisioning key (stored in 1Password). Render it and run `site.yml`.
|
||||
|
||||
```bash
|
||||
# Render the provisioning private key from 1Password to a temp file
|
||||
umask 077
|
||||
op read --account team-futo \
|
||||
"op://yucca_tf_prod/HTZ_FSN1_PROVISIONING_SSH_PRIVATE_KEY/password" \
|
||||
> /tmp/htz-fsn1-prov-key
|
||||
chmod 600 /tmp/htz-fsn1-prov-key
|
||||
|
||||
# Run the playbook (root, provisioning key, NetBird mgmt setup key from 1P)
|
||||
ansible-playbook -i inventories/htz-fsn1 site.yml \
|
||||
--private-key /tmp/htz-fsn1-prov-key \
|
||||
--extra-vars "mgmt_netbird_setup_key=$(op read --account team-futo \
|
||||
'op://yucca_tf_prod/NETBIRD_YUCCA_PROD_HTZ_FSN1_MGMT_SETUP_KEY/password')"
|
||||
|
||||
# Clean up
|
||||
shred -u /tmp/htz-fsn1-prov-key
|
||||
```
|
||||
|
||||
`ansible.cfg` sets `inventory = inventories/htz-fsn1/hosts.yml`, so `-i` is
|
||||
optional. The inventory connects as `ansible_user: root`.
|
||||
|
||||
The whole render-key + run flow above is wrapped by `mise run mgmt:ansible`
|
||||
(`SITE` selects the inventory; defaults to `htz-fsn1`), which CI also runs on
|
||||
every **prod** apply — the `Ansible converge (mgmt hosts)` step of
|
||||
`.github/workflows/infra.yml`, right after the Terraform apply. It's idempotent
|
||||
and reaches the hosts over their public IP, so it requires them to already be
|
||||
reprovisioned (provisioning key authorized).
|
||||
|
||||
> The `--private-key` flow is preferred over `ansible_ssh_private_key_file` in
|
||||
> group_vars so the key never has to be persisted to a committed path — it is
|
||||
> rendered to a temp file, used, and shredded.
|
||||
|
||||
## Connection: public IP → NetBird
|
||||
|
||||
A freshly-reprovisioned host is only reachable over its **public IP**, so that's
|
||||
the bootstrap address (`mgmt_public_ip` in host_vars). The first play in
|
||||
`site.yml` probes the overlay from the control node (`netbird status --json`):
|
||||
if the host is a **connected peer**, it switches `ansible_host` to the host's
|
||||
**NetBird IP** for the rest of the run; otherwise it stays on the public IP.
|
||||
|
||||
So the first run provisions over the public IP and brings the host onto the
|
||||
overlay (the `netbird` role); every subsequent run reconnects over NetBird
|
||||
automatically. This needs the control node on the overlay too — the CI
|
||||
`prod-ansible` job joins via the `ci` setup key; locally, be connected to NetBird.
|
||||
For the very first run after a reinstall, pass `-e mgmt_bootstrap=true` to force
|
||||
the public IP (skips any stale peer entry for the host).
|
||||
|
||||
## 25G fabric caveat
|
||||
|
||||
The 25G fabric link (Intel E810, "ice" driver) is **currently physically
|
||||
unreliable**. The `networkd` role that configures its VLAN sub-interfaces is
|
||||
gated off by default (`mgmt_networkd_enabled: false`), so a normal `site.yml`
|
||||
run is a no-op for networking. Once the fabric links:
|
||||
|
||||
1. Confirm the NIC name on each host with `ip link` (prior name:
|
||||
`enp33s0f0np0`) and correct `mgmt_fabric_nic` in the host_vars if it
|
||||
differs.
|
||||
2. Set `mgmt_networkd_enabled: true` (e.g. `--extra-vars` or group_vars).
|
||||
|
||||
VLAN layout (gateways are `.1` on the leaf IRB):
|
||||
|
||||
| VLAN | Network | mgmt-1 | mgmt-2 |
|
||||
|------|---------|--------|--------|
|
||||
| 20 (cluster public) | `10.40.20.0/23` | `10.40.20.2` | `10.40.20.3` |
|
||||
| 22 (cluster private) | `10.40.22.0/23` | `10.40.22.2` | `10.40.22.3` |
|
||||
|
||||
The primary public NIC keeps Hetzner's DHCP default — this tree does not touch
|
||||
it.
|
||||
|
||||
## Setup
|
||||
|
||||
```bash
|
||||
ansible-galaxy collection install -r requirements.yml
|
||||
ansible-playbook -i inventories/htz-fsn1 --syntax-check site.yml
|
||||
```
|
||||
|
||||
## Secrets
|
||||
|
||||
Nothing secret is committed. SSH public keys are public data and are TF-generated
|
||||
into `group_vars/all/users.generated.yml`. Runtime secrets are passed via `op read`:
|
||||
|
||||
- Provisioning private key: `op://yucca_tf_prod/HTZ_FSN1_PROVISIONING_SSH_PRIVATE_KEY/password`
|
||||
- NetBird mgmt setup key: `op://yucca_tf_prod/NETBIRD_YUCCA_PROD_<SITE>_MGMT_SETUP_KEY/password`
|
||||
(the site's reusable `mgmt` key, `auto_groups=["mgmt"]`; minted by the prod netbird stack)
|
||||
@@ -0,0 +1,30 @@
|
||||
[defaults]
|
||||
# Connects as root with the TF-provisioning key (see README). Inventory is
|
||||
# committed (these are two static hosts, not TF-rendered like ceph).
|
||||
inventory = inventories/htz-fsn1/hosts.yml
|
||||
log_path = ansible.log
|
||||
|
||||
# Performance
|
||||
forks = 20
|
||||
gathering = smart
|
||||
fact_caching = jsonfile
|
||||
fact_caching_connection = .ansible_facts_cache
|
||||
fact_caching_timeout = 3600
|
||||
|
||||
# Output
|
||||
stdout_callback = default
|
||||
result_format = yaml
|
||||
callbacks_enabled = ansible.posix.timer, ansible.posix.profile_tasks
|
||||
force_color = True
|
||||
diff_always = True
|
||||
deprecation_warnings = False
|
||||
retry_files_enabled = False
|
||||
display_skipped_hosts = False
|
||||
|
||||
# Security
|
||||
host_key_checking = False
|
||||
timeout = 30
|
||||
|
||||
[ssh_connection]
|
||||
ssh_args = -o ControlMaster=auto -o ControlPersist=60s -o StrictHostKeyChecking=no
|
||||
pipelining = True
|
||||
@@ -0,0 +1,14 @@
|
||||
---
|
||||
# Static shared vars for the htz-fsn1 mgmt hosts.
|
||||
#
|
||||
# Host/user/addressing data is TF-GENERATED at run time (hosts.yml,
|
||||
# host_vars/*.yml, users.generated.yml) by tf/render/ansible-mgmt from the
|
||||
# Terraform sources (mgmt-hosts.yaml, fabric-addressing, identity). Edit those,
|
||||
# not the generated files. Only truly-static config lives here.
|
||||
|
||||
timezone: UTC
|
||||
mgmt_domain: fsn.htz.futo.cloud
|
||||
|
||||
# NetBird "mgmt" setup key — NOT hardcoded; passed at run time by
|
||||
# `mise run mgmt:ansible` from op://yucca_tf_prod/NETBIRD_YUCCA_PROD_<SITE>_MGMT_SETUP_KEY.
|
||||
mgmt_netbird_setup_key: ""
|
||||
@@ -0,0 +1,6 @@
|
||||
---
|
||||
collections:
|
||||
- name: ansible.posix
|
||||
version: ">=2.0.0"
|
||||
- name: community.general
|
||||
version: ">=9.0.0"
|
||||
@@ -0,0 +1,19 @@
|
||||
---
|
||||
# Post-reprovision OS baseline defaults.
|
||||
|
||||
# --- Packages ---
|
||||
mgmt_base_packages:
|
||||
- curl
|
||||
- gnupg
|
||||
- ca-certificates
|
||||
- apt-transport-https
|
||||
- ethtool
|
||||
- jq
|
||||
- htop
|
||||
- tmux
|
||||
- ncdu
|
||||
- sysstat
|
||||
- bsdmainutils
|
||||
|
||||
# --- /etc/hosts ---
|
||||
mgmt_manage_hosts: true
|
||||
@@ -0,0 +1,10 @@
|
||||
---
|
||||
# First role in site.yml — runs on the freshly installed OS.
|
||||
dependencies: []
|
||||
|
||||
galaxy_info:
|
||||
author: FUTO
|
||||
license: AGPL-3.0-only
|
||||
role_name: baseline
|
||||
description: Post-reprovision OS baseline — packages, hostname, timezone, hosts
|
||||
min_ansible_version: "2.18"
|
||||
@@ -0,0 +1,12 @@
|
||||
---
|
||||
# Post-reprovision OS baseline.
|
||||
# Runs as root over SSH on the freshly installed Debian 13 host.
|
||||
# Convergeable: re-running corrects drift in packages, hostname, hosts file.
|
||||
|
||||
- name: Configure system packages
|
||||
ansible.builtin.import_tasks: packages.yml
|
||||
tags: [packages]
|
||||
|
||||
- name: Configure hostname, timezone, /etc/hosts
|
||||
ansible.builtin.import_tasks: system.yml
|
||||
tags: [system]
|
||||
@@ -0,0 +1,12 @@
|
||||
---
|
||||
# Base package set. Convergeable: re-running installs anything missing.
|
||||
|
||||
- name: Update apt cache
|
||||
ansible.builtin.apt:
|
||||
update_cache: true
|
||||
cache_valid_time: 3600
|
||||
|
||||
- name: Install base packages
|
||||
ansible.builtin.apt:
|
||||
name: "{{ mgmt_base_packages }}"
|
||||
state: present
|
||||
@@ -0,0 +1,20 @@
|
||||
---
|
||||
# Hostname, timezone, /etc/hosts.
|
||||
# Convergeable: re-running corrects drift.
|
||||
|
||||
- name: Set hostname from inventory
|
||||
ansible.builtin.hostname:
|
||||
name: "{{ inventory_hostname }}"
|
||||
|
||||
- name: Set timezone
|
||||
community.general.timezone:
|
||||
name: "{{ timezone }}"
|
||||
|
||||
- name: Render /etc/hosts with mgmt node entries
|
||||
ansible.builtin.template:
|
||||
src: hosts.j2
|
||||
dest: /etc/hosts
|
||||
owner: root
|
||||
group: root
|
||||
mode: '0644'
|
||||
when: mgmt_manage_hosts | bool
|
||||
@@ -0,0 +1,13 @@
|
||||
127.0.0.1 localhost
|
||||
127.0.1.1 {{ inventory_hostname }}.{{ mgmt_domain }} {{ inventory_hostname }}
|
||||
|
||||
# Management nodes (25G fabric VLAN 20 — cluster public)
|
||||
{% for host in groups['mgmt'] %}
|
||||
{% for vlan in hostvars[host]['mgmt_fabric_vlans'] | default([]) if vlan.id == 20 %}
|
||||
{{ vlan.address | regex_replace('/.*$', '') }} {{ host }}.{{ mgmt_domain }} {{ host }}
|
||||
{% endfor %}
|
||||
{% endfor %}
|
||||
|
||||
::1 localhost ip6-localhost ip6-loopback
|
||||
ff02::1 ip6-allnodes
|
||||
ff02::2 ip6-allrouters
|
||||
@@ -0,0 +1,9 @@
|
||||
---
|
||||
# NetBird defaults.
|
||||
mgmt_netbird_management_url: "https://api.netbird.io"
|
||||
|
||||
# Setup key — NEVER hardcoded. Passed at run time (op:// ref, see README). The
|
||||
# site's reusable "mgmt" key (auto_groups=["mgmt"]) so the node joins the mgmt
|
||||
# group and becomes a route peer for the site subnets (10.40.5.0/24, api, cluster
|
||||
# nets — the routed network is defined in TF: tf/deployment/prod/<site>/netbird).
|
||||
mgmt_netbird_setup_key: ""
|
||||
@@ -0,0 +1,10 @@
|
||||
---
|
||||
# Install NetBird and join the overlay (mgmt nodes are the site route peers).
|
||||
dependencies: []
|
||||
|
||||
galaxy_info:
|
||||
author: FUTO
|
||||
license: AGPL-3.0-only
|
||||
role_name: netbird
|
||||
description: Install NetBird and bring the node up on the overlay
|
||||
min_ansible_version: "2.18"
|
||||
@@ -0,0 +1,56 @@
|
||||
---
|
||||
# Install NetBird and join the overlay. The mgmt nodes are the NetBird route
|
||||
# peers for the site subnets (the routed network is declared in TF, not here), so
|
||||
# enable IP forwarding. Convergeable: `netbird up` is skipped when already
|
||||
# connected.
|
||||
|
||||
- name: Install NetBird (official script — adds the apt repo + installs)
|
||||
ansible.builtin.shell:
|
||||
cmd: curl -fsSL https://pkgs.netbird.io/install.sh | sh
|
||||
creates: /usr/bin/netbird
|
||||
environment:
|
||||
NETBIRD_SKIP_UP: "true" # don't auto-join during install; we run `up` below
|
||||
|
||||
- name: Enable and start netbird
|
||||
ansible.builtin.systemd:
|
||||
name: netbird
|
||||
enabled: true
|
||||
state: started
|
||||
|
||||
- name: Assert NetBird setup key is provided
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- mgmt_netbird_setup_key | length > 0
|
||||
fail_msg: >-
|
||||
mgmt_netbird_setup_key is empty. Pass it at run time (mgmt:ansible does this
|
||||
from op://yucca_tf_prod/NETBIRD_YUCCA_PROD_<SITE>_MGMT_SETUP_KEY).
|
||||
|
||||
# mgmt nodes route the site subnets to the overlay.
|
||||
- name: Enable IP forwarding (NetBird route peer)
|
||||
ansible.posix.sysctl:
|
||||
name: "{{ item }}"
|
||||
value: '1'
|
||||
sysctl_set: true
|
||||
state: present
|
||||
reload: true
|
||||
loop:
|
||||
- net.ipv4.ip_forward
|
||||
- net.ipv6.conf.all.forwarding
|
||||
|
||||
- name: Check NetBird connection status
|
||||
ansible.builtin.command: netbird status
|
||||
register: mgmt_netbird_status
|
||||
changed_when: false
|
||||
failed_when: false
|
||||
|
||||
- name: Bring the node up on the overlay
|
||||
ansible.builtin.command:
|
||||
cmd: >-
|
||||
netbird up
|
||||
--setup-key {{ mgmt_netbird_setup_key }}
|
||||
--management-url {{ mgmt_netbird_management_url }}
|
||||
--hostname {{ inventory_hostname }}
|
||||
register: mgmt_netbird_up
|
||||
changed_when: mgmt_netbird_up.rc == 0
|
||||
no_log: true
|
||||
when: "'Management: Connected' not in mgmt_netbird_status.stdout"
|
||||
@@ -0,0 +1,14 @@
|
||||
---
|
||||
# systemd-networkd VLAN sub-interfaces on the 25G fabric NIC.
|
||||
#
|
||||
# Prerequisites:
|
||||
# - systemd-networkd present (Debian 13 base)
|
||||
# - mgmt_fabric_nic, mgmt_fabric_vlans defined in host_vars
|
||||
#
|
||||
# Master toggle. False by default because the 25G fabric link is currently
|
||||
# unreliable — flip to true (and confirm mgmt_fabric_nic via `ip link`) once
|
||||
# the fabric is up.
|
||||
mgmt_networkd_enabled: false
|
||||
|
||||
# --- Paths ---
|
||||
mgmt_networkd_config_dir: /etc/systemd/network
|
||||
@@ -0,0 +1,5 @@
|
||||
---
|
||||
- name: Reload systemd for networkd
|
||||
ansible.builtin.systemd:
|
||||
name: systemd-networkd
|
||||
state: reloaded
|
||||
@@ -0,0 +1,11 @@
|
||||
---
|
||||
# 25G fabric VLAN sub-interfaces. Gated on mgmt_networkd_enabled (default
|
||||
# false) — applies once the 25G fabric links.
|
||||
dependencies: []
|
||||
|
||||
galaxy_info:
|
||||
author: FUTO
|
||||
license: AGPL-3.0-only
|
||||
role_name: networkd
|
||||
description: systemd-networkd VLAN sub-interfaces on the 25G fabric NIC
|
||||
min_ansible_version: "2.18"
|
||||
@@ -0,0 +1,38 @@
|
||||
---
|
||||
# Write systemd-networkd config files for the 25G fabric VLANs.
|
||||
# Safe to re-run.
|
||||
|
||||
- name: Deploy fabric parent .network (declares VLANs, no L3 of its own)
|
||||
ansible.builtin.template:
|
||||
src: fabric.network.j2
|
||||
dest: "{{ mgmt_networkd_config_dir }}/20-{{ mgmt_fabric_nic }}.network"
|
||||
owner: root
|
||||
group: root
|
||||
mode: '0644'
|
||||
notify: Reload systemd for networkd
|
||||
|
||||
- name: Deploy VLAN .netdev files
|
||||
ansible.builtin.template:
|
||||
src: vlan.netdev.j2
|
||||
dest: "{{ mgmt_networkd_config_dir }}/30-{{ mgmt_fabric_nic }}.{{ vlan.id }}.netdev"
|
||||
owner: root
|
||||
group: root
|
||||
mode: '0644'
|
||||
loop: "{{ mgmt_fabric_vlans }}"
|
||||
loop_control:
|
||||
loop_var: vlan
|
||||
label: "vlan {{ vlan.id }}"
|
||||
notify: Reload systemd for networkd
|
||||
|
||||
- name: Deploy VLAN .network files
|
||||
ansible.builtin.template:
|
||||
src: vlan.network.j2
|
||||
dest: "{{ mgmt_networkd_config_dir }}/30-{{ mgmt_fabric_nic }}.{{ vlan.id }}.network"
|
||||
owner: root
|
||||
group: root
|
||||
mode: '0644'
|
||||
loop: "{{ mgmt_fabric_vlans }}"
|
||||
loop_control:
|
||||
loop_var: vlan
|
||||
label: "vlan {{ vlan.id }}"
|
||||
notify: Reload systemd for networkd
|
||||
@@ -0,0 +1,17 @@
|
||||
---
|
||||
# systemd-networkd VLAN sub-interfaces on the 25G fabric NIC.
|
||||
#
|
||||
# NOTE: applies once the 25G fabric links — the 25G link is currently
|
||||
# physically unreliable (see notes). Gated on mgmt_networkd_enabled
|
||||
# (default: false) so a normal site.yml run is a no-op until the fabric
|
||||
# is up and the operator opts in.
|
||||
#
|
||||
# 1. write parent .network (DHCP off on fabric, declares VLANs)
|
||||
# 2. write per-VLAN .netdev + .network (static addresses)
|
||||
# 3. reload networkd
|
||||
#
|
||||
# The primary public NIC keeps Hetzner's DHCP default — untouched here.
|
||||
|
||||
- name: Deploy networkd configs
|
||||
ansible.builtin.import_tasks: deploy.yml
|
||||
when: mgmt_networkd_enabled | bool
|
||||
@@ -0,0 +1,15 @@
|
||||
# {{ ansible_managed }}
|
||||
# 25G fabric parent NIC: {{ mgmt_fabric_nic }} (Intel E810, "ice").
|
||||
# No L3 on the parent itself — addresses live on the VLAN sub-interfaces.
|
||||
# The `VLAN=` directives below tell networkd to instantiate the sub-interfaces
|
||||
# declared by the .netdev files (Kind=vlan alone only registers the type).
|
||||
|
||||
[Match]
|
||||
Name={{ mgmt_fabric_nic }}
|
||||
|
||||
[Network]
|
||||
{% for vlan in mgmt_fabric_vlans %}
|
||||
VLAN={{ mgmt_fabric_nic }}.{{ vlan.id }}
|
||||
{% endfor %}
|
||||
LinkLocalAddressing=no
|
||||
IPv6AcceptRA=no
|
||||
@@ -0,0 +1,9 @@
|
||||
# {{ ansible_managed }}
|
||||
# Tagged VLAN {{ vlan.id }} sub-interface on {{ mgmt_fabric_nic }}.
|
||||
|
||||
[NetDev]
|
||||
Name={{ mgmt_fabric_nic }}.{{ vlan.id }}
|
||||
Kind=vlan
|
||||
|
||||
[VLAN]
|
||||
Id={{ vlan.id }}
|
||||
@@ -0,0 +1,11 @@
|
||||
# {{ ansible_managed }}
|
||||
# L3 on VLAN {{ vlan.id }} sub-interface: {{ vlan.address }} (gateway .1 on the leaf IRB).
|
||||
|
||||
[Match]
|
||||
Name={{ mgmt_fabric_nic }}.{{ vlan.id }}
|
||||
|
||||
[Network]
|
||||
Address={{ vlan.address }}
|
||||
|
||||
[Link]
|
||||
RequiredForOnline=no
|
||||
@@ -0,0 +1,25 @@
|
||||
---
|
||||
# Security hardening defaults for management hosts.
|
||||
|
||||
# --- Firewall (nftables) ---
|
||||
mgmt_firewall_enabled: true
|
||||
|
||||
# Trusted source networks allowed to reach management services beyond SSH.
|
||||
# RFC1918 + NetBird/CGNAT (100.64.0.0/10). Loopback is always allowed via "lo".
|
||||
mgmt_firewall_trusted_networks:
|
||||
- 10.0.0.0/8
|
||||
- 172.16.0.0/12
|
||||
- 192.168.0.0/16
|
||||
- 100.64.0.0/10
|
||||
|
||||
# Allow SSH from any source (true) or restrict to trusted networks (false).
|
||||
# Default true: these hosts are reached over the public IP for bootstrapping
|
||||
# and NetBird admin. Tighten once NetBird access is confirmed.
|
||||
mgmt_firewall_ssh_any_source: true
|
||||
|
||||
# --- SSH hardening ---
|
||||
mgmt_ssh_max_auth_tries: 3
|
||||
# Key-based root is required for the initial Ansible run (TF provisioning key).
|
||||
# prohibit-password keeps password root login off while allowing key auth.
|
||||
mgmt_ssh_permit_root_login: prohibit-password
|
||||
mgmt_ssh_allowed_users: "root nutgood andy"
|
||||
@@ -0,0 +1,3 @@
|
||||
// Managed by Ansible (security role)
|
||||
APT::Periodic::Update-Package-Lists "1";
|
||||
APT::Periodic::Unattended-Upgrade "1";
|
||||
@@ -0,0 +1,5 @@
|
||||
---
|
||||
- name: Restart sshd
|
||||
ansible.builtin.systemd:
|
||||
name: ssh
|
||||
state: restarted
|
||||
@@ -0,0 +1,10 @@
|
||||
---
|
||||
# nftables firewall, SSH hardening, unattended-upgrades.
|
||||
dependencies: []
|
||||
|
||||
galaxy_info:
|
||||
author: FUTO
|
||||
license: AGPL-3.0-only
|
||||
role_name: security
|
||||
description: nftables firewall, SSH hardening, unattended-upgrades for mgmt hosts
|
||||
min_ansible_version: "2.18"
|
||||
@@ -0,0 +1,70 @@
|
||||
---
|
||||
# Security hardening for management hosts.
|
||||
# nftables firewall + SSH hardening + unattended-upgrades.
|
||||
|
||||
# --- nftables firewall ---
|
||||
|
||||
- name: Install nftables
|
||||
ansible.builtin.apt:
|
||||
name: nftables
|
||||
state: present
|
||||
|
||||
- name: Deploy nftables ruleset
|
||||
ansible.builtin.template:
|
||||
src: nftables.conf.j2
|
||||
dest: /etc/nftables.conf
|
||||
owner: root
|
||||
group: root
|
||||
mode: '0644'
|
||||
validate: "nft -c -f %s"
|
||||
register: nftables_config
|
||||
when: mgmt_firewall_enabled | bool
|
||||
|
||||
- name: Enable and start nftables
|
||||
ansible.builtin.systemd:
|
||||
name: nftables
|
||||
enabled: true
|
||||
state: "{{ 'restarted' if nftables_config is changed else 'started' }}"
|
||||
when: mgmt_firewall_enabled | bool
|
||||
|
||||
# --- SSH hardening ---
|
||||
|
||||
- name: Deploy SSH hardening config
|
||||
ansible.builtin.template:
|
||||
src: sshd-hardening.conf.j2
|
||||
dest: /etc/ssh/sshd_config.d/50-hardening.conf
|
||||
owner: root
|
||||
group: root
|
||||
mode: '0644'
|
||||
notify: Restart sshd
|
||||
|
||||
- name: Verify full SSH config is valid after drop-in
|
||||
ansible.builtin.command: sshd -t
|
||||
changed_when: false
|
||||
|
||||
# --- unattended-upgrades ---
|
||||
|
||||
- name: Install unattended-upgrades
|
||||
ansible.builtin.apt:
|
||||
name:
|
||||
- unattended-upgrades
|
||||
- apt-listchanges
|
||||
state: present
|
||||
|
||||
- name: Enable unattended-upgrades
|
||||
ansible.builtin.copy:
|
||||
src: 20auto-upgrades
|
||||
dest: /etc/apt/apt.conf.d/20auto-upgrades
|
||||
owner: root
|
||||
group: root
|
||||
mode: '0644'
|
||||
|
||||
# --- Summary ---
|
||||
|
||||
- name: Report security state
|
||||
ansible.builtin.debug:
|
||||
msg:
|
||||
- "Firewall: {{ 'enabled' if mgmt_firewall_enabled else 'disabled' }}"
|
||||
- "Trusted networks: {{ mgmt_firewall_trusted_networks | join(', ') }}"
|
||||
- "SSH AllowUsers: {{ mgmt_ssh_allowed_users }}"
|
||||
- "SSH PermitRootLogin: {{ mgmt_ssh_permit_root_login }}"
|
||||
@@ -0,0 +1,47 @@
|
||||
#!/usr/sbin/nft -f
|
||||
# Management host firewall — managed by Ansible (security role)
|
||||
# Allows SSH + established/related + ICMP. Drops everything else.
|
||||
|
||||
flush ruleset
|
||||
|
||||
table inet filter {
|
||||
chain input {
|
||||
type filter hook input priority 0; policy drop;
|
||||
|
||||
# Loopback
|
||||
iifname "lo" accept
|
||||
|
||||
# NetBird overlay interface (mesh + subnet routing)
|
||||
iifname "wt0" accept
|
||||
|
||||
# Established/related
|
||||
ct state established,related accept
|
||||
|
||||
# ICMP (ping, MTU discovery)
|
||||
ip protocol icmp accept
|
||||
ip6 nexthdr icmpv6 accept
|
||||
|
||||
# SSH
|
||||
{% if mgmt_firewall_ssh_any_source | bool %}
|
||||
tcp dport 22 accept
|
||||
{% else %}
|
||||
{% for net in mgmt_firewall_trusted_networks %}
|
||||
ip saddr {{ net }} tcp dport 22 accept
|
||||
{% endfor %}
|
||||
{% endif %}
|
||||
|
||||
# Log + drop everything else
|
||||
limit rate 5/minute log prefix "nftables-drop: " level warn
|
||||
drop
|
||||
}
|
||||
|
||||
chain forward {
|
||||
# NetBird route peer forwards between the overlay and the site subnets
|
||||
# (10.40.5.0/24 et al.).
|
||||
type filter hook forward priority 0; policy accept;
|
||||
}
|
||||
|
||||
chain output {
|
||||
type filter hook output priority 0; policy accept;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,10 @@
|
||||
# SSH hardening — managed by Ansible (security role)
|
||||
# Dropped into /etc/ssh/sshd_config.d/ to override defaults without
|
||||
# touching the main sshd_config.
|
||||
|
||||
MaxAuthTries {{ mgmt_ssh_max_auth_tries }}
|
||||
PermitRootLogin {{ mgmt_ssh_permit_root_login }}
|
||||
AllowUsers {{ mgmt_ssh_allowed_users }}
|
||||
PasswordAuthentication no
|
||||
PermitEmptyPasswords no
|
||||
X11Forwarding no
|
||||
@@ -0,0 +1,4 @@
|
||||
---
|
||||
# Login users. The real list lives in group_vars/all.yml (mgmt_users),
|
||||
# mirrored from tf/shared/modules/identity. Default empty here.
|
||||
mgmt_users: []
|
||||
@@ -0,0 +1,10 @@
|
||||
---
|
||||
# Creates login users from the identity registry mirror (mgmt_users).
|
||||
dependencies: []
|
||||
|
||||
galaxy_info:
|
||||
author: FUTO
|
||||
license: AGPL-3.0-only
|
||||
role_name: users
|
||||
description: Login users + SSH keys + sudo from the identity registry
|
||||
min_ansible_version: "2.18"
|
||||
@@ -0,0 +1,37 @@
|
||||
---
|
||||
# Login users for members of the identity registry's `server`-mapped groups.
|
||||
# Sourced from mgmt_users (group_vars/all.yml). Convergeable: re-running fixes
|
||||
# drift in group membership, sudo, and authorized_keys.
|
||||
|
||||
- name: Ensure login users exist
|
||||
ansible.builtin.user:
|
||||
name: "{{ item.name }}"
|
||||
shell: /bin/bash
|
||||
groups: sudo
|
||||
append: true
|
||||
create_home: true
|
||||
state: present
|
||||
loop: "{{ mgmt_users }}"
|
||||
loop_control:
|
||||
label: "{{ item.name }}"
|
||||
|
||||
- name: Configure per-user sudo
|
||||
ansible.builtin.copy:
|
||||
content: "{{ item.name }} ALL=(ALL) {{ item.sudo }}\n"
|
||||
dest: "/etc/sudoers.d/{{ item.name }}"
|
||||
mode: '0440'
|
||||
validate: "visudo -cf %s"
|
||||
loop: "{{ mgmt_users }}"
|
||||
loop_control:
|
||||
label: "{{ item.name }}"
|
||||
|
||||
# exclusive: true makes the registry the single source of truth — keys not
|
||||
# listed for the user are removed.
|
||||
- name: Authorize user SSH keys
|
||||
ansible.posix.authorized_key:
|
||||
user: "{{ item.name }}"
|
||||
key: "{{ item.ssh_keys | join('\n') }}"
|
||||
exclusive: true
|
||||
loop: "{{ mgmt_users }}"
|
||||
loop_control:
|
||||
label: "{{ item.name }}"
|
||||
@@ -0,0 +1,60 @@
|
||||
---
|
||||
# Configure the Hetzner FSN1 management hosts after they are reprovisioned
|
||||
# to Debian 13 ("trixie") via Hetzner robot auto-install.
|
||||
#
|
||||
# Connection: freshly-reprovisioned hosts are reached over their PUBLIC IP
|
||||
# (bootstrap); once a host has joined the NetBird overlay (the netbird role),
|
||||
# later runs reconnect over its NetBird IP automatically — see the first play.
|
||||
#
|
||||
# Canonical entrypoint is `mise run mgmt:ansible` (renders the inventory from
|
||||
# Terraform, then runs this). It passes the provisioning key + the NetBird
|
||||
# "mgmt" setup key (op://yucca_tf_prod/NETBIRD_YUCCA_PROD_<SITE>_MGMT_SETUP_KEY).
|
||||
#
|
||||
# NOTE: the networkd role (25G VLAN sub-interfaces) is gated on
|
||||
# mgmt_networkd_enabled (default false) because the 25G fabric link is
|
||||
# currently unreliable. Enable it once the fabric is up.
|
||||
|
||||
# ── Prefer NetBird, fall back to the public IP ───────────────────────────────
|
||||
# Runs entirely on the control node (no target connection), so it's safe before
|
||||
# a host is reachable at all. If the host is a connected NetBird peer, switch the
|
||||
# connection to its NetBird IP; otherwise keep the public IP (bootstrap).
|
||||
- name: Select connection address
|
||||
hosts: mgmt
|
||||
gather_facts: false
|
||||
become: false
|
||||
tasks:
|
||||
- name: Read the NetBird overlay status from the control node
|
||||
ansible.builtin.command: netbird status --json
|
||||
delegate_to: localhost
|
||||
register: _nb_status
|
||||
changed_when: false
|
||||
failed_when: false
|
||||
|
||||
- name: Reconnect over NetBird when the host is a connected peer
|
||||
vars:
|
||||
_nb_ips: >-
|
||||
{{ ((_nb_status.stdout | from_json).peers.details | default([]))
|
||||
| selectattr('fqdn', 'defined')
|
||||
| selectattr('fqdn', 'match', '^' ~ inventory_hostname ~ '([.]|$)')
|
||||
| selectattr('status', 'equalto', 'Connected')
|
||||
| map(attribute='netbirdIp') | select | list }}
|
||||
ansible.builtin.set_fact:
|
||||
ansible_host: "{{ _nb_ips[0] if (_nb_ips | length > 0) else mgmt_public_ip }}"
|
||||
when:
|
||||
# `-e mgmt_bootstrap=true` forces the public IP — use it for the first run
|
||||
# after a reinstall, when a stale OLD peer for this host may still show up.
|
||||
- not (mgmt_bootstrap | default(false) | bool)
|
||||
- _nb_status.rc == 0
|
||||
- (_nb_status.stdout | trim | length) > 0
|
||||
|
||||
- name: Configure management hosts
|
||||
hosts: mgmt
|
||||
become: true
|
||||
gather_facts: true
|
||||
|
||||
roles:
|
||||
- baseline
|
||||
- users
|
||||
- security
|
||||
- networkd
|
||||
- netbird
|
||||
@@ -20,3 +20,8 @@ export TF_VAR_netbox_token=op://yucca_tf/NETBOX_API_TOKEN/password
|
||||
# Stored at op://yucca_tf_prod/NET_SWITCHES_TERRAFORM_SSH_PRIVATE_KEY/password.
|
||||
# NOT exported here as content — the fabric mise task renders it to a 0600 temp
|
||||
# file and exports TF_VAR_netconf_ssh_key_path=<that path>.
|
||||
|
||||
# ── Hetzner Robot API (mgmt-host reprovisioning, zack/hetzner provider) ───────
|
||||
# The provider reads these env vars directly (no provider config block needed).
|
||||
export HETZNER_ROBOT_USERNAME=op://yucca_tf_prod/HETZNER_WEBSERVICE_API_USER/password
|
||||
export HETZNER_ROBOT_PASSWORD=op://yucca_tf_prod/HETZNER_WEBSERVICE_API_PASSWORD/password
|
||||
|
||||
@@ -0,0 +1,10 @@
|
||||
# Generated by 'mise run fabric:provider-build' — do not edit.
|
||||
provider_installation {
|
||||
filesystem_mirror {
|
||||
path = "/Users/leca/src/yucca/.mise/.fabric-provider-mirror"
|
||||
include = ["registry.opentofu.org/hashicorp/junos-qfx", "registry.terraform.io/hashicorp/junos-qfx"]
|
||||
}
|
||||
direct {
|
||||
exclude = ["registry.opentofu.org/hashicorp/junos-qfx", "registry.terraform.io/hashicorp/junos-qfx"]
|
||||
}
|
||||
}
|
||||
@@ -35,25 +35,31 @@ VLAN id == the network's third octet; gateway = `.1` (IRB on the leaf).
|
||||
keys committed; passwords via vars from 1Password). Fed by `modules/identity`.
|
||||
- `modules/fabric-netbox` — mirrors the IP plan into NetBox (prefixes + VLANs).
|
||||
|
||||
## The provider
|
||||
## The providers
|
||||
|
||||
The `junos-qfx` provider is **JTAF-generated and vendored** in `tf/providers/` (not on
|
||||
any registry). It's built into a local filesystem mirror by `mise run fabric:provider-build`.
|
||||
This stack builds two providers locally (neither is on a registry) into a shared
|
||||
filesystem mirror via `mise run infra:providers`:
|
||||
|
||||
- `mise run fabric:provider-gen` — regenerate from device YANG + live config (only
|
||||
when adding new config hierarchies), then commit `tf/providers/`.
|
||||
- `mise run fabric:provider-build` — `go build` the vendored source into the mirror
|
||||
and write `tf/.terraformrc.fabric` (consumed via `TF_CLI_CONFIG_FILE`).
|
||||
- `junos-qfx` — **JTAF-generated and vendored** in `tf/providers/` (the switch fabric).
|
||||
- `hetzner` (`zack/hetzner`) — the Hetzner Robot API, for mgmt-host reprovisioning
|
||||
(`mgmt.tf`); cloned + built (pinned tag) by `mise run mgmt:provider-build`.
|
||||
|
||||
- `mise run fabric:provider-gen` — regenerate the junos-qfx provider from device YANG
|
||||
+ live config (only when adding new config hierarchies), then commit `tf/providers/`.
|
||||
- `mise run infra:providers` — build both providers into the mirror and write
|
||||
`tf/.terraformrc.local` (consumed via `TF_CLI_CONFIG_FILE`).
|
||||
|
||||
## Running
|
||||
|
||||
```sh
|
||||
mise run fabric:plan # builds provider, renders the NETCONF key from 1Password, terragrunt plan
|
||||
mise run fabric:apply # ... apply
|
||||
mise run infra:plan # builds providers, renders creds from 1Password, terragrunt plan
|
||||
mise run infra:apply # ... apply
|
||||
```
|
||||
|
||||
CI: `.github/workflows/fabric.yml` — plan on PR, gated apply on merge behind the
|
||||
site-scoped `prod-fabric-htz-fsn1` GitHub Environment (required reviewers).
|
||||
(`SITE` selects the stack; defaults to `htz-fsn1`.)
|
||||
|
||||
CI: `.github/workflows/infra.yml` — plan on PR, gated apply on merge behind the
|
||||
site-scoped `prod-htz-fsn1` GitHub Environment (required reviewers).
|
||||
|
||||
## Adoption caveat (first run)
|
||||
|
||||
|
||||
@@ -0,0 +1,27 @@
|
||||
# Canonical mgmt-host roster for htz-fsn1 — the single source of truth consumed by
|
||||
# BOTH the reprovision stack (mgmt.tf, uses server_number) and the ansible
|
||||
# inventory render (tf/render/ansible-mgmt, uses everything else).
|
||||
#
|
||||
# site_id — drives addressing (must match the stack's var.site_id).
|
||||
# cluster_id — the cluster whose public/private VLANs these hosts sit on.
|
||||
# host_index — host's offset within each VLAN /23 (gateway is .1 on the leaf);
|
||||
# .2/.3 here -> 10.40.20.2/.3 (public), 10.40.22.2/.3 (private).
|
||||
# fabric_nic — 25G NIC carrying the tagged VLAN sub-interfaces (verify after
|
||||
# reprovision; predictable name may differ on fresh Debian 13).
|
||||
# subnet_router — advertises the mgmt /24 over Tailscale (exactly one host).
|
||||
site_id: 40
|
||||
cluster_id: 1
|
||||
|
||||
hosts:
|
||||
htz-fsn-mgmt-1:
|
||||
server_number: 3008208
|
||||
public_ip: 178.63.124.40
|
||||
host_index: 2
|
||||
fabric_nic: enp33s0f0np0
|
||||
subnet_router: true
|
||||
htz-fsn-mgmt-2:
|
||||
server_number: 3008209
|
||||
public_ip: 178.63.124.41
|
||||
host_index: 3
|
||||
fabric_nic: enp33s0f0np0
|
||||
subnet_router: false
|
||||
@@ -0,0 +1,78 @@
|
||||
# mgmt hosts — Hetzner dedicated-server reprovisioning (zack/hetzner robot API).
|
||||
#
|
||||
# Two-step, operator-gated flow:
|
||||
# 1. Terraform arms a fresh Debian auto-install on the robot (hetzner_boot_linux),
|
||||
# authorizing a TF-owned automation SSH key for root. This is NON-destructive
|
||||
# — the flag only takes effect on the next boot; the running host is untouched.
|
||||
# 2. The operator reboots the host (manually, for now) -> the robot wipes the
|
||||
# disks + installs Debian -> ansible/mgmt configures it (networkd, tailscale,
|
||||
# users from modules/identity, baseline + hardening).
|
||||
#
|
||||
# Only hosts listed in var.mgmt_reprovision_targets are armed, so a normal apply
|
||||
# does nothing to the mgmt hosts. Re-target deliberately for each reprovision.
|
||||
#
|
||||
# Host roster (server numbers = Hetzner robot IDs, GET /server) comes from the
|
||||
# shared mgmt-hosts.yaml — the same SoT the ansible inventory render reads.
|
||||
locals {
|
||||
mgmt_roster = yamldecode(file("${path.module}/mgmt-hosts.yaml"))
|
||||
mgmt_hosts = local.mgmt_roster.hosts
|
||||
|
||||
site_prefix = upper(replace(var.netbox_site_slug, "-", "_")) # HTZ_FSN1
|
||||
provisioning_key_item = "${local.site_prefix}_PROVISIONING_SSH_PRIVATE_KEY"
|
||||
}
|
||||
|
||||
# Guard: the roster's site_id must match the stack's, or addressing diverges.
|
||||
resource "terraform_data" "mgmt_site_id_check" {
|
||||
lifecycle {
|
||||
precondition {
|
||||
condition = local.mgmt_roster.site_id == var.site_id
|
||||
error_message = "mgmt-hosts.yaml site_id (${local.mgmt_roster.site_id}) != var.site_id (${var.site_id})."
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# ── Provisioning keypair (TF-owned, recorded in 1Password) ───────────────────
|
||||
# Generated here, stored in the env-appropriate vault as the source-of-truth
|
||||
# record, and registered in the Hetzner robot. Authorized for root on freshly-
|
||||
# reprovisioned hosts; ansible/mgmt reads the private half from 1Password
|
||||
# (op://<vault>/<title>/password). Survives state loss and is operator-visible.
|
||||
resource "tls_private_key" "mgmt_provisioning" {
|
||||
algorithm = "ED25519"
|
||||
}
|
||||
|
||||
data "onepassword_vault" "env" {
|
||||
name = var.op_vault
|
||||
}
|
||||
|
||||
resource "onepassword_item" "mgmt_provisioning_key" {
|
||||
vault = data.onepassword_vault.env.uuid
|
||||
title = local.provisioning_key_item
|
||||
category = "password"
|
||||
password = tls_private_key.mgmt_provisioning.private_key_openssh
|
||||
|
||||
section {
|
||||
label = "keypair"
|
||||
field {
|
||||
label = "public_key"
|
||||
type = "STRING"
|
||||
value = trimspace(tls_private_key.mgmt_provisioning.public_key_openssh)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
resource "hetzner_ssh_key" "mgmt_automation" {
|
||||
name = "${local.site_prefix}-provisioning"
|
||||
data = trimspace(tls_private_key.mgmt_provisioning.public_key_openssh)
|
||||
}
|
||||
|
||||
# ── Arm a fresh OS install for each targeted host ────────────────────────────
|
||||
# server_number RequiresReplace, so re-targeting recreates cleanly.
|
||||
resource "hetzner_boot_linux" "mgmt" {
|
||||
for_each = toset(var.mgmt_reprovision_targets)
|
||||
|
||||
server_number = local.mgmt_hosts[each.key].server_number
|
||||
dist = var.mgmt_dist
|
||||
lang = "en"
|
||||
arch = 64
|
||||
authorized_key = hetzner_ssh_key.mgmt_automation.fingerprint
|
||||
}
|
||||
@@ -27,6 +27,22 @@ policies = {
|
||||
destinations = ["mgmt", "talos", "k8s_operator"]
|
||||
}]
|
||||
}
|
||||
|
||||
# CI also reaches the routed site subnets (switch vme 10.40.5.0/24 + api +
|
||||
# cluster nets), so the fabric jobs can NETCONF the switches over the overlay
|
||||
# (the switches are routed resources behind the mgmt peers, not peers
|
||||
# themselves). yucca_resource is the shared tag on every routed resource,
|
||||
# resolved from the global layer via external_groups.
|
||||
ci-to-resources = {
|
||||
description = "CI → routed site subnets (yucca_resource)."
|
||||
rules = [{
|
||||
name = "ci-to-resources"
|
||||
protocol = "all"
|
||||
bidirectional = false
|
||||
sources = ["ci"]
|
||||
destinations = ["yucca_resource"]
|
||||
}]
|
||||
}
|
||||
}
|
||||
|
||||
# Site identifier (mirrors prod/htz-fsn1's site_id). Feeds the fabric-addressing
|
||||
|
||||
@@ -23,3 +23,13 @@ provider "netbox" {
|
||||
server_url = var.netbox_url
|
||||
api_token = var.netbox_token
|
||||
}
|
||||
|
||||
# Hetzner Robot API — mgmt-host reprovisioning (mgmt.tf). Credentials come from
|
||||
# the env (HETZNER_ROBOT_USERNAME/PASSWORD), injected by op-run from tf/.env.prod;
|
||||
# no secrets in config.
|
||||
provider "hetzner" {}
|
||||
|
||||
# 1Password — stores the TF-generated provisioning key (mgmt.tf) in the env vault.
|
||||
# Authenticates with OP_SERVICE_ACCOUNT_TOKEN (the `infra:apply` task escalates to
|
||||
# the write-capable SA pulled from the vault); no Connect host needed.
|
||||
provider "onepassword" {}
|
||||
|
||||
@@ -11,3 +11,13 @@ cls1_leaf_serials = ["XH4925470753", "XH4925460012"]
|
||||
|
||||
# spine (corenetsw) VC member serials.
|
||||
spine_vc_serials = ["WH3622440738", "WH0220510012"]
|
||||
|
||||
# ── mgmt-host reprovisioning (mgmt.tf) ───────────────────────────────────────
|
||||
# The provisioning keypair is GENERATED by Terraform and stored in 1Password
|
||||
# (op_vault, default yucca_tf_prod) as HTZ_FSN1_PROVISIONING_SSH_PRIVATE_KEY;
|
||||
# ansible/mgmt reads the private half from there. Nothing to set here.
|
||||
|
||||
# Hosts to ARM for a fresh OS install on next boot (DESTRUCTIVE on reboot).
|
||||
# Testing the flow on mgmt-2 first (mgmt-1 is the tailscale subnet router).
|
||||
# Set back to [] once reprovisioning is complete.
|
||||
mgmt_reprovision_targets = ["htz-fsn-mgmt-2"]
|
||||
|
||||
@@ -53,3 +53,26 @@ variable "spine_vc_serials" {
|
||||
type = list(string)
|
||||
description = "Spine (corenetsw) VC member chassis serials (member 0, member 1)."
|
||||
}
|
||||
|
||||
# ── mgmt-host reprovisioning (mgmt.tf) ───────────────────────────────────────
|
||||
variable "op_vault" {
|
||||
type = string
|
||||
default = "yucca_tf_prod"
|
||||
description = "1Password vault (env-appropriate) the TF-generated provisioning key is written to."
|
||||
}
|
||||
|
||||
variable "mgmt_dist" {
|
||||
type = string
|
||||
default = "Debian 13 base"
|
||||
description = "Hetzner robot Linux auto-install image (must match an available `dist` exactly; see GET /boot/<n>/linux)."
|
||||
}
|
||||
|
||||
variable "mgmt_reprovision_targets" {
|
||||
type = list(string)
|
||||
default = []
|
||||
description = <<-EOT
|
||||
mgmt host keys (see local.mgmt_hosts in mgmt.tf) to ARM for a fresh OS install
|
||||
on next boot. DESTRUCTIVE once the host is rebooted. Keep empty except during a
|
||||
planned reprovision; set to the host(s) being reprovisioned, apply, then reboot.
|
||||
EOT
|
||||
}
|
||||
|
||||
@@ -2,7 +2,7 @@ terraform {
|
||||
required_version = "~> 1.11"
|
||||
required_providers {
|
||||
# JTAF-generated, vendored in tf/providers/terraform-provider-junos-qfx and
|
||||
# supplied via dev_overrides (see mise `fabric:provider-build` + the GH workflow).
|
||||
# supplied via dev_overrides (built by `infra:providers`; supplied via TF_CLI_CONFIG_FILE).
|
||||
junos-qfx = {
|
||||
source = "hashicorp/junos-qfx"
|
||||
}
|
||||
@@ -10,5 +10,23 @@ terraform {
|
||||
source = "e-breuninger/netbox"
|
||||
version = "~> 4.0"
|
||||
}
|
||||
# Hetzner Robot (dedicated-server) API — mgmt-host reprovisioning (mgmt.tf).
|
||||
# Built locally + supplied via the same filesystem_mirror as junos-qfx
|
||||
# (mise `mgmt:provider-build`, invoked by `infra:providers`).
|
||||
hetzner = {
|
||||
source = "zack/hetzner"
|
||||
}
|
||||
# Generates the mgmt-host provisioning keypair (mgmt.tf).
|
||||
tls = {
|
||||
source = "hashicorp/tls"
|
||||
version = "~> 4.0"
|
||||
}
|
||||
# Stores the generated provisioning key in 1Password (the env vault). Auth via
|
||||
# the service-account token in OP_SERVICE_ACCOUNT_TOKEN (apply escalates to the
|
||||
# write-capable SA stored in the vault; see the `infra:*` mise tasks).
|
||||
onepassword = {
|
||||
source = "1Password/onepassword"
|
||||
version = "~> 2.1"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,67 @@
|
||||
# Render the ansible/mgmt inventory for a site from Terraform's sources of truth:
|
||||
# • fabric-addressing — VLAN ids + per-host VLAN addresses
|
||||
# • identity — server login users (+ keys + sudo)
|
||||
# • mgmt-hosts.yaml — the host roster (IPs, NIC, host index, subnet router)
|
||||
#
|
||||
# Lightweight by design (local provider only, no backend/secrets), so it can run
|
||||
# just-in-time in the ansible CI job (and locally) via `mise run mgmt:ansible`.
|
||||
# Outputs are gitignored — TF is the single source of truth, not the rendered files.
|
||||
|
||||
locals {
|
||||
cfg = yamldecode(file("${path.module}/../../deployment/prod/${var.site}/mgmt-hosts.yaml"))
|
||||
hosts = local.cfg.hosts
|
||||
inv_dir = "${path.module}/../../../ansible/mgmt/inventories/${var.site}"
|
||||
|
||||
pub_mask = split("/", module.addressing.public_cidr)[1]
|
||||
priv_mask = split("/", module.addressing.private_cidr)[1]
|
||||
|
||||
header = "# GENERATED by `mise run mgmt:render-inventory` (tf/render/ansible-mgmt).\n# Do not edit — edit the Terraform sources (mgmt-hosts.yaml, fabric-addressing, identity).\n"
|
||||
}
|
||||
|
||||
module "addressing" {
|
||||
source = "../../shared/modules/fabric-addressing"
|
||||
site_id = local.cfg.site_id
|
||||
cluster_id = local.cfg.cluster_id
|
||||
}
|
||||
|
||||
module "identity" {
|
||||
source = "../../shared/modules/identity"
|
||||
}
|
||||
|
||||
# Inventory: hosts + public IPs (ansible connects as root over the public IP).
|
||||
resource "local_file" "hosts" {
|
||||
filename = "${local.inv_dir}/hosts.yml"
|
||||
content = "${local.header}${yamlencode({
|
||||
mgmt = {
|
||||
hosts = { for name, h in local.hosts : name => { ansible_host = h.public_ip } }
|
||||
vars = {
|
||||
ansible_user = "root"
|
||||
ansible_python_interpreter = "/usr/bin/python3"
|
||||
}
|
||||
}
|
||||
})}"
|
||||
}
|
||||
|
||||
# Per-host: 25G NIC + the tagged VLAN sub-interfaces (ids + addresses from the
|
||||
# addressing module) + the subnet-router's advertised route.
|
||||
resource "local_file" "host_vars" {
|
||||
for_each = local.hosts
|
||||
filename = "${local.inv_dir}/host_vars/${each.key}.yml"
|
||||
content = "${local.header}${yamlencode({
|
||||
# Bootstrap address — site.yml reconnects over NetBird once the host joins.
|
||||
# (NetBird routes the site subnets via the mgmt peer group; that's declared in
|
||||
# tf/deployment/prod/<site>/netbird, so there's no per-host advertise flag.)
|
||||
mgmt_public_ip = each.value.public_ip
|
||||
mgmt_fabric_nic = each.value.fabric_nic
|
||||
mgmt_fabric_vlans = [
|
||||
{ id = module.addressing.public_vlan_id, address = "${cidrhost(module.addressing.public_cidr, each.value.host_index)}/${local.pub_mask}" },
|
||||
{ id = module.addressing.private_vlan_id, address = "${cidrhost(module.addressing.private_cidr, each.value.host_index)}/${local.priv_mask}" },
|
||||
]
|
||||
})}"
|
||||
}
|
||||
|
||||
# group_vars: login users from the identity registry (server-mapped groups).
|
||||
resource "local_file" "users" {
|
||||
filename = "${local.inv_dir}/group_vars/all/users.generated.yml"
|
||||
content = "${local.header}${yamlencode({ mgmt_users = module.identity.server_login })}"
|
||||
}
|
||||
@@ -0,0 +1,8 @@
|
||||
variable "site" {
|
||||
type = string
|
||||
default = "htz-fsn1"
|
||||
description = <<-EOT
|
||||
Site slug. Selects the roster at tf/deployment/prod/<site>/mgmt-hosts.yaml and
|
||||
the inventory rendered under ansible/mgmt/inventories/<site>/.
|
||||
EOT
|
||||
}
|
||||
@@ -0,0 +1,11 @@
|
||||
terraform {
|
||||
required_version = "~> 1.11"
|
||||
required_providers {
|
||||
# Only the local provider — this render writes files from static, public data
|
||||
# (addressing + identity). No backend, no secrets, no remote state.
|
||||
local = {
|
||||
source = "hashicorp/local"
|
||||
version = "~> 2.5"
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -16,6 +16,21 @@ output "server_authorized_keys" {
|
||||
])))
|
||||
}
|
||||
|
||||
output "server_login" {
|
||||
description = "Server login users: members of any `server`-mapped group, with their SSH keys + effective sudo. Consumed by server provisioning (ansible/mgmt)."
|
||||
value = [
|
||||
for uname, u in var.users : {
|
||||
name = uname
|
||||
ssh_keys = sort(distinct(concat(u.ssh_ed25519_keys, u.ssh_rsa_keys)))
|
||||
sudo = try([
|
||||
for g in u.groups : var.groups[g].server.sudo
|
||||
if try(var.groups[g].server, null) != null
|
||||
][0], "ALL")
|
||||
}
|
||||
if length([for g in u.groups : g if try(var.groups[g].server, null) != null]) > 0
|
||||
]
|
||||
}
|
||||
|
||||
output "members_of" {
|
||||
description = "Group name -> sorted member usernames. Lets a consumer (e.g. servers) provision a group's people."
|
||||
value = {
|
||||
|
||||
Reference in New Issue
Block a user