feat(prod): deploy prod apps (#284)

feat(prod); deploy prod apps
This commit is contained in:
Antoine Lecompte
2026-07-21 12:02:57 -07:00
committed by GitHub
parent 017da30a6e
commit bd74b16c7c
35 changed files with 613 additions and 54 deletions
@@ -95,6 +95,11 @@ chrony_ntp_servers:
networkd_bond_vlans:
- id: "{{ fabric_public_vlan }}" # 120 - Ceph public (10.40.20.0/23)
address: "10.40.20.{{ host_index }}/23"
# Return route to the kube workers (michael S3 clients): the spine holds a
# second IRB on this VLAN (10.40.21.254, fabric stack public_routing) and
# routes kube<->cls1-public. The leaf keeps the .1 host gateway.
routes:
- { to: "10.40.10.0/24", via: "10.40.21.254" }
- id: "{{ fabric_private_vlan }}" # 122 - Ceph private (10.40.22.0/23)
address: "10.40.22.{{ host_index }}/23"
mtu: 9000
@@ -11,6 +11,12 @@ Address={{ vlan.address }}
{% if vlan.dns is defined %}
DNS={{ vlan.dns }}
{% endif %}
{% for route in vlan.routes | default([]) %}
[Route]
Destination={{ route.to }}
Gateway={{ route.via }}
{% endfor %}
[Link]
RequiredForOnline=no
@@ -35,12 +35,13 @@ spec:
value:
metadata:
labels:
# Pools with a serviceSelector (e.g. father's lb-internal) require
# this; ignored where the pool matches by IP annotation (staging).
lb: internal
# Pool selection where pools use a serviceSelector (father:
# internal → lb-internal, public → lb-public-a); ignored where
# the pool matches by IP annotation (staging).
lb: ${INGRESS_LB_LABEL}
annotations:
# Cilium LB-IPAM assigns this specific VIP to the Service.
lbipam.cilium.io/ips: "${INGRESS_INTERNAL_IP}"
lbipam.cilium.io/ips: "${INGRESS_VIP}"
telemetry:
metrics:
prometheus: {}
+4 -2
View File
@@ -1,13 +1,15 @@
---
# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/gateway.networking.k8s.io/httproute_v1.json
# michael — the restic gateway. Exposed at gw.<domain>.
# michael — the restic gateway. Exposed at gw.<domain>. The parent is
# per-cluster: staging's single app gateway (envoy) vs father's dedicated
# high-throughput gateway (gw) — see GW_PARENT_GATEWAY in cluster-settings.
apiVersion: gateway.networking.k8s.io/v1
kind: HTTPRoute
metadata:
name: gw
spec:
parentRefs:
- name: envoy
- name: ${GW_PARENT_GATEWAY}
namespace: envoy-system
sectionName: https
hostnames:
@@ -21,6 +21,8 @@ spec:
remediation:
retries: 3
values:
# Per-cluster fleet size (staging 2, father 12 — the 60+ Gbps data plane).
replicas: ${MICHAEL_REPLICAS}
image:
repository: ghcr.io/immich-app/yucca/michael
tag: ${YUCCA_IMAGE_TAG}
@@ -21,6 +21,8 @@ spec:
remediation:
retries: 3
values:
# Per-cluster replica count (staging 2, father 3).
replicas: ${API_REPLICAS}
image:
repository: ghcr.io/immich-app/yucca/yucca-api
# Substituted by Flux postBuild from the image-versions ConfigMap (CI-bumped).
@@ -0,0 +1,43 @@
# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/gateway.envoyproxy.io/envoyproxy_v1alpha1.json
# Data-plane config for the DEDICATED michael (restic) gateway — sized for the
# backup data plane (60+ Gbps target), separate from the 2-replica app gateway.
# One replica per worker; the LB Service draws its VIP from the public pool-a
# (lb: public matches its NotIn selector) and runs externalTrafficPolicy: Local,
# so the spine ECMPs the /32 across every worker with a local envoy and traffic
# stays on the landing node (no second SNAT hop).
apiVersion: gateway.envoyproxy.io/v1alpha1
kind: EnvoyProxy
metadata:
name: envoy-gw
spec:
logging:
level:
default: info
provider:
type: Kubernetes
kubernetes:
envoyDeployment:
replicas: ${GW_PROXY_REPLICAS}
pod:
topologySpreadConstraints:
- maxSkew: 1
topologyKey: kubernetes.io/hostname
whenUnsatisfiable: DoNotSchedule
labelSelector:
matchLabels:
app.kubernetes.io/name: envoy
gateway.envoyproxy.io/owning-gateway-name: gw
envoyService:
type: LoadBalancer
externalTrafficPolicy: Local
patch:
type: StrategicMerge
value:
metadata:
labels:
lb: public
annotations:
lbipam.cilium.io/ips: "${GW_VIP}"
telemetry:
metrics:
prometheus: {}
@@ -0,0 +1,30 @@
# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/gateway.networking.k8s.io/gateway_v1.json
# The michael (restic) Gateway: HTTPS-only, scoped to ${GW_HOST}. Shares the
# `envoy` GatewayClass but attaches its own EnvoyProxy via
# infrastructure.parametersRef (Envoy Gateway creates one proxy fleet per
# Gateway). TLS terminates with the shared wildcard cert (*.APP_DOMAIN covers
# gw.); no :80 listener — restic clients speak HTTPS only.
apiVersion: gateway.networking.k8s.io/v1
kind: Gateway
metadata:
name: gw
spec:
gatewayClassName: envoy
infrastructure:
parametersRef:
group: gateway.envoyproxy.io
kind: EnvoyProxy
name: envoy-gw
listeners:
- name: https
protocol: HTTPS
port: 443
hostname: "${GW_HOST}"
allowedRoutes:
namespaces:
from: All
tls:
mode: Terminate
certificateRefs:
- kind: Secret
name: staging-backups-tls
@@ -0,0 +1,6 @@
---
apiVersion: kustomize.config.k8s.io/v1beta1
kind: Kustomization
resources:
- ./envoyproxy.yaml
- ./gateway.yaml
@@ -0,0 +1,28 @@
# CNPG operator on father — CHERRY-PICKED from components/infra/cnpg-system
# (the full infra component stays off: its storage layer would collide with the
# prod-owned openebs install, see kustomization.yaml). CRD provider for the
# yucca-database Cluster CR, so it lives in the cluster-infra layer.
# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/kustomize.toolkit.fluxcd.io/kustomization_v1.json
---
apiVersion: kustomize.toolkit.fluxcd.io/v1
kind: Kustomization
metadata:
name: cnpg-operator
namespace: flux-system
spec:
healthChecks:
- apiVersion: helm.toolkit.fluxcd.io/v2
kind: HelmRelease
name: cloudnative-pg
namespace: cnpg-system
interval: 1h
retryInterval: 2m
timeout: 10m
path: ./kubernetes/apps/base/cloudnative-pg
prune: true
wait: true
sourceRef:
kind: GitRepository
name: flux-system
namespace: flux-system
targetNamespace: cnpg-system
@@ -1,7 +1,7 @@
# Envoy Gateway on father — CHERRY-PICKED from components/infra/network (same
# caveat as cert-manager.yaml: REMOVE this file when components/infra is enabled,
# or the Kustomization names collide). Deploys the operator + the APP gateway
# (pre-staged for the yucca launch: VIP ${INGRESS_INTERNAL_IP}, cert for
# (pre-staged for the yucca launch: VIP ${INGRESS_VIP}, cert for
# ${APP_DOMAIN}); the NETOPS gateway lives in netops/gateway.yaml.
# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/kustomize.toolkit.fluxcd.io/kustomization_v1.json
---
@@ -11,6 +11,8 @@ kind: Kustomization
resources:
- ./namespace-envoy-system.yaml
- ./namespace-openebs.yaml
- ./namespace-cnpg-system.yaml
- ./cert-manager.yaml
- ./envoy.yaml
- ./openebs.yaml
- ./cnpg.yaml
@@ -0,0 +1,5 @@
---
apiVersion: v1
kind: Namespace
metadata:
name: cnpg-system
@@ -4,15 +4,12 @@
# `cluster-infra` Flux Kustomization that this layer dependsOn — see
# infra/kustomization.yaml for why (fresh-cluster dry-run deadlock).
#
# CURRENT SCOPE: Flux owns the cluster BASELINE (coredns, the Cilium BGP LB
# config, the netops stack, OpenEBS pools). The platform/infra components and
# the yucca app set are DELIBERATELY not enabled yet:
# - components/infra needs real cluster-settings (RGW endpoint, o11y vmauth)
# and would collide with the in-cluster OpenEBS install (infra/openebs.yaml
# owns it via HelmRelease instead).
# - components/roles/primary is the yucca WORKLOAD set — explicitly held back
# until prod launch.
# Re-enable by uncommenting `components:` below.
# SCOPE: cluster baseline (coredns, Cilium BGP LB config, netops, OpenEBS
# pools) + the yucca platform/app set. components/infra stays off WHOLESALE —
# its storage layer collides with the prod-owned openebs (infra/openebs.yaml,
# targetNamespace `openebs`) and netbird-operator has no prod service user —
# so the platform pieces are cherry-picked: operators in ./infra
# (cluster-infra layer), issuer/proxies/routes/observability in ./platform.
apiVersion: kustomize.config.k8s.io/v1beta1
kind: Kustomization
resources:
@@ -22,6 +19,6 @@ resources:
- ./lb-return-route.yaml
- ./diskpools.yaml
- ./netops
# components:
# - ../../../components/infra
# - ../../../components/roles/primary
- ./platform
components:
- ../../../components/roles/primary
@@ -0,0 +1,24 @@
# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/kustomize.toolkit.fluxcd.io/kustomization_v1.json
---
# Dedicated michael (restic) gateway — see ../gw-proxy/. Separate from the app
# gateway so the backup data plane (60+ Gbps target) scales independently.
apiVersion: kustomize.toolkit.fluxcd.io/v1
kind: Kustomization
metadata:
name: gw-proxy
namespace: flux-system
spec:
dependsOn:
- name: envoy-gateway
- name: cert-manager-issuer
interval: 1h
retryInterval: 2m
timeout: 10m
path: ./kubernetes/apps/prod/htz-fsn1/gw-proxy
prune: true
wait: true
sourceRef:
kind: GitRepository
name: flux-system
namespace: flux-system
targetNamespace: envoy-system
@@ -0,0 +1,25 @@
# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/kustomize.toolkit.fluxcd.io/kustomization_v1.json
---
# HTTPRoutes (web/api on the app gateway, gw on the dedicated michael gateway —
# GW_PARENT_GATEWAY selects the parent). CHERRY-PICKED from
# components/infra/network/httproutes.yaml.
apiVersion: kustomize.toolkit.fluxcd.io/v1
kind: Kustomization
metadata:
name: httproutes
namespace: flux-system
spec:
dependsOn:
- name: envoy-proxy
- name: gw-proxy
interval: 1h
retryInterval: 2m
timeout: 5m
path: ./kubernetes/apps/base/httproutes
prune: true
wait: true
sourceRef:
kind: GitRepository
name: flux-system
namespace: flux-system
targetNamespace: yucca
@@ -0,0 +1,11 @@
---
# prod@htz-fsn1 PLATFORM slice — the app-layer platform pieces cherry-picked
# from components/infra (which stays off wholesale: its storage layer collides
# with the prod-owned openebs, and netbird-operator has no prod service user
# yet). cert-manager/envoy/issuer already live in ../infra (cluster-infra).
apiVersion: kustomize.config.k8s.io/v1beta1
kind: Kustomization
resources:
- ./observability.yaml
- ./httproutes.yaml
- ./gw-proxy.yaml
@@ -0,0 +1,22 @@
# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/kustomize.toolkit.fluxcd.io/kustomization_v1.json
---
# Observability agents (vmagent + victoria-logs-collector) — the shared
# components/infra/observability slice (namespace, netpols, the two nested
# Kustomizations), reused as a plain kustomize path. The vmagent-remote-write
# Secret is TF-provisioned (talos stack secrets.tf).
apiVersion: kustomize.toolkit.fluxcd.io/v1
kind: Kustomization
metadata:
name: observability
namespace: flux-system
spec:
interval: 1h
retryInterval: 2m
timeout: 10m
path: ./kubernetes/components/infra/observability
prune: true
wait: true
sourceRef:
kind: GitRepository
name: flux-system
namespace: flux-system
@@ -3,7 +3,7 @@
# layout is per-cluster: father uses BGP pools in its cilium-bgp.yaml, and a
# shared base pool would CIDR-overlap its lb-internal /24 → Conflicting pools).
# Applied directly by cluster-apps (no postBuild substitution on raw overlay
# resources), so the VIP is LITERAL here — keep in sync with INGRESS_INTERNAL_IP
# resources), so the VIP is LITERAL here — keep in sync with INGRESS_VIP
# in clusters/staging/austin/cluster-settings.yaml.
#
# Single-address LB-IPAM pool for the ingress VIP. The envoy Service requests it
@@ -1,10 +1,7 @@
---
# TF-RENDERED — do not hand-edit. Authored ahead of the cluster (prod's
# Talos/flux stack isn't built yet — tf/deployment/prod/htz-fsn1 is fabric-only),
# to a valid value matching what TF will regenerate once that stack lands. Keys
# are DISJOINT from the human cluster-settings + CI image-versions ConfigMaps.
# TODO(prod): real RGW endpoint + observability vmauth host when the cluster is
# provisioned.
# TF-RENDERED — do not hand-edit (hand-maintained until the talos stack renders
# it; values match what TF would derive). Keys are DISJOINT from the human
# cluster-settings + CI image-versions ConfigMaps.
apiVersion: v1
kind: ConfigMap
metadata:
@@ -16,16 +13,19 @@ data:
CLUSTER_ROLE: primary
METRICS_CLUSTER_LABEL: yucca_prod_htz_fsn1
# TODO: real production ingress domain + restic-gateway host.
APP_DOMAIN: yucca.futo.cloud
GW_HOST: gw.yucca.futo.cloud
# Per-service ingress hostnames. web (apex; /api routes to yucca-api) on the
# app gateway; gw. (michael, restic) on the dedicated gw-proxy gateway.
APP_DOMAIN: backups.futo.cloud
GW_HOST: gw.backups.futo.cloud
# Bare-metal RGW (Ceph S3) gateway. Prod uses a COMPLETELY SEPARATE Ceph from
# staging's sietch cluster. S3_HOST is the same gateway without the scheme
# (michael's S3_BACKEND_DNS_HOST). TODO: real production RGW endpoint.
S3_ENDPOINT: https://s3.prod.futo.cloud
S3_HOST: s3.prod.futo.cloud
# Spice RGW (Ceph S3): round-robin DNS across all 48 nodes' fabric VLAN-120
# IPs (prod/global/dns), reached over the FABRIC — the spine routes
# kube↔cls1-public (fabric stack public_routing). Self-signed cert, so
# consumers skip TLS verify. S3_HOST is the schemeless form for michael's
# DNS-based backend LB (S3_BACKEND_DNS_HOST).
S3_ENDPOINT: https://s3.prod.fsn1.htz.futo.cloud
S3_HOST: s3.prod.fsn1.htz.futo.cloud
# Observability egress. TODO: real prod vmauth host.
# Observability egress (apps -> agents -> o11y prod vmauth).
VMAGENT_OTLP: vmagent-yucca.observability.svc:8429
VLOGS_REMOTE_URL: https://vmauth.prod.futostatus.com/insert/native
@@ -1,22 +1,33 @@
---
# Per-cluster, HUMAN-managed settings for prod@htz-fsn1. Authored ahead of the
# cluster; activates once the prod flux stack syncs this path. Neither CI nor TF
# touch this file. Keys here are disjoint from cluster-settings.generated.yaml
# (TF) and image-versions.yaml (CI). TODO(prod): real ingress VIPs + ops contact.
# Per-cluster, HUMAN-managed settings for prod@htz-fsn1. Neither CI nor TF
# touch this file. Keys are disjoint from cluster-settings.generated.yaml (TF)
# and image-versions.yaml (CI).
apiVersion: v1
kind: ConfigMap
metadata:
name: cluster-settings
namespace: flux-system
data:
# TODO: real production OIDC issuer (Zitadel prod instance).
OIDC_ISSUER: https://external-dev-gkhk8b.us1.zitadel.cloud
# Production Zitadel instance; clients in 1P (CUSTOMER_ZITADEL_OAUTH_*).
OIDC_ISSUER: https://external-prod-r5dlf0.us1.zitadel.cloud
# ─── Ingress entry point ─────────────────────────────────────────────
# App gateway VIP (lb_internal range; the netops gateway pins .16 itself).
# TODO: INGRESS_PUBLIC_IP once the public app frontend is wired (lb-public pool).
INGRESS_INTERNAL_IP: 10.40.12.20
INGRESS_PUBLIC_IP: 0.0.0.0
# ─── Ingress entry points ────────────────────────────────────────────
# App gateway (web/api): public pool-a VIP, BGP-advertised /32 covered by the
# 69.48.224.0/24 transit aggregate. lb: public selects pool-a.
INGRESS_LB_LABEL: public
INGRESS_VIP: 69.48.224.5
# Dedicated michael (restic) gateway — see apps/prod/htz-fsn1/gw-proxy.
# externalTrafficPolicy: Local + one envoy per worker → spine ECMP.
GW_PARENT_GATEWAY: gw
GW_VIP: 69.48.224.6
GW_PROXY_REPLICAS: "3"
# (netops keeps its own internal VIP, pinned at 10.40.12.16 in netops/.)
# ─── App fleet sizing ────────────────────────────────────────────────
# michael sized for the 60+ Gbps backup data plane (~4 per worker today;
# scale with the worker count). API is stateless behind the app gateway.
MICHAEL_REPLICAS: "12"
API_REPLICAS: "3"
# ACME registration contact for Let's Encrypt. TODO: real ops address.
ACME_EMAIL: fubar-prod@futo.org
@@ -15,9 +15,18 @@ data:
# ─── Ingress entry point ─────────────────────────────────────────────
# Internal VIP the in-cluster LB/Gateway announces (Cilium L2; distinct from
# the 10.10.10.15 control-plane API VIP). Public IP is the NAT in front of it,
# which staging.backups.futo.cloud resolves to publicly.
INGRESS_INTERNAL_IP: 10.10.10.16
# which staging.backups.futo.cloud resolves to publicly. The lb label is
# ignored here (the pool matches by IP annotation, no serviceSelector).
INGRESS_LB_LABEL: internal
INGRESS_VIP: 10.10.10.16
INGRESS_PUBLIC_IP: 97.77.242.205
# Single shared gateway: michael's gw. route attaches to the app gateway
# (father runs a dedicated gw-proxy instead).
GW_PARENT_GATEWAY: envoy
# ─── App fleet sizing ────────────────────────────────────────────────
MICHAEL_REPLICAS: "2"
API_REPLICAS: "2"
# ACME registration contact for Let's Encrypt. TODO: real ops address.
ACME_EMAIL: fubar-prod@futo.org
+20
View File
@@ -51,3 +51,23 @@ export TF_VAR_flux_github_app_private_key="op://shared_tf/GITHUB_APP_IMMICH_PUSH
# sees no zones. TODO: mint a least-privilege token (Zone:Read + DNS:Edit on
# futo.network only) and swap this ref.
export TF_VAR_cloudflare_api_token="op://shared_tf/FUTO_BOOTSTRAP_CLOUDFLARE_API_TOKEN/password"
# ─── App secrets (prod/htz-fsn1/talos secrets.tf) ────────────────────────────
# yucca-api OIDC client (prod Zitadel) + device-flow public client.
export TF_VAR_yucca_oidc_client_id="op://yucca_tf_prod/CUSTOMER_ZITADEL_OAUTH_CLIENT_ID_YUCCA_WEB/password"
export TF_VAR_yucca_oidc_client_secret="op://yucca_tf_prod/CUSTOMER_ZITADEL_OAUTH_CLIENT_SECRET_YUCCA_WEB/password"
export TF_VAR_yucca_oidc_device_client_id="op://yucca_tf_prod/CUSTOMER_ZITADEL_OAUTH_CLIENT_ID_YUCCA_ORCHESTRATOR/password"
# yucca-admin-api client not registered yet (mirrors staging):
# export TF_VAR_yucca_oidc_admin_client_id="op://yucca_tf_prod/.../password"
# export TF_VAR_yucca_oidc_admin_client_secret="op://yucca_tf_prod/.../password"
# michael → spice RGW (svc-yucca-restic, out-of-band contract items).
export TF_VAR_yucca_rgw_access_key_id="op://yucca_tf_prod/SPICE_CEPH_S3_SVC_YUCCA_RESTIC_ACCESS_KEY/password"
export TF_VAR_yucca_rgw_secret_access_key="op://yucca_tf_prod/SPICE_CEPH_S3_SVC_YUCCA_RESTIC_SECRET_KEY/password"
# yucca-metrics-worker → spice RGW admin API.
export TF_VAR_spice_metrics_worker_access_key="op://yucca_tf_prod/SPICE_METRICS_WORKER_ACCESS_KEY/password"
export TF_VAR_spice_metrics_worker_secret_key="op://yucca_tf_prod/SPICE_METRICS_WORKER_SECRET_KEY/password"
# vmagent/logs remote-write bearer for o11y prod vmauth.
export TF_VAR_vmauth_remote_write_password="op://shared_tf_prod/O11Y_VICTORIAMETRICS_VMAUTH_PASSWORD/password"
@@ -16,6 +16,22 @@ zone_id = "474fbfd96bf49879054a493f126c4071"
# the ceph roster are kept in sync by tf/scripts/check-s3-dns-roster.py (CI gate
# s3-dns-roster-validate); it fails if a node is added/removed on one side only.
records = {
# Yucca prod ingress: app gateway (web + /api) and the dedicated michael
# (restic) gateway. Public pool-a VIPs (Cilium LB-IPAM pins, cluster-settings
# INGRESS_VIP / GW_VIP), BGP-advertised /32s covered by the 69.48.224.0/24
# transit aggregate. proxied false: restic long-lived uploads and the API
# don't want Cloudflare in the path.
"backups.futo.cloud" = {
type = "A"
values = ["69.48.224.5"]
comment = "Yucca prod web/api gateway (tf/deployment/prod/global/dns)"
}
"gw.backups.futo.cloud" = {
type = "A"
values = ["69.48.224.6"]
comment = "Yucca prod restic gateway - michael (tf/deployment/prod/global/dns)"
}
"s3.prod.fsn1.htz.futo.cloud" = {
type = "A"
values = [
@@ -36,6 +36,13 @@ module "core" {
ae2 = ["et-0/0/2:3", "et-1/0/2:2"]
ae3 = ["et-0/0/2:1", "et-1/0/2:1"]
}
# Spine routes kube ↔ cls1-public (same pattern as kube↔kube-cp): the workers'
# fabric path to the spice RGW frontend (michael S3 traffic, not via NetBird).
# IRB = last usable /23 address; the leaf keeps the .1 host gateway.
public_routing = {
cidr = module.addr_cls1.public_cidr
ip = "${cidrhost(module.addr_cls1.public_cidr, 510)}/${module.addr_cls1.prefixlen}"
}
# father's bare-metal control planes: port-0 breakout legs at 10G (xe-), one leg
# per VC member, trunking the kube-cp VLAN. Pairing VERIFIED 2026-07-15 via MAC
@@ -19,6 +19,11 @@ groups = {
mgmt = { resource = true } # management nodes (ansible); also the route peers
talos = { resource = true } # Talos cluster nodes → yucca-prod-htz-fsn1-talos
resources = { resource = true } # routed-subnet tag → yucca-prod-htz-fsn1-resources (Network resources tag in)
# cls1 (ceph) nets only — split from `resources` so talos-to-resources does NOT
# grant them: the workers reach the RGW frontend over the FABRIC (spine-routed),
# and a NetBird client route would shadow that path. yucca users still get
# access via resource = true.
ceph_nets = { resource = true }
# CP-only subset of `talos` — the ROUTER peer group for the kube-cp network. Only
# the CPs sit on the kube-cp VLAN, so only they can route it; if the router were
# the whole `talos` group the bare-metal WORKERS (also `talos`) would be
@@ -19,6 +19,11 @@ locals {
# NB: the `kube-cp` VLAN is deliberately NOT in this map — it's routed by its
# own network below (via the CPs, the talos_cp group), keeping the API plane's
# mesh path independent of the mgmt routers.
# The cls1 (ceph) networks are tagged `ceph_nets`, NOT `resources`: yucca users
# still reach them (resource=true group ⇒ yucca→resources policy destination),
# but the talos-to-resources policy doesn't — the workers' RGW path is the
# FABRIC (spine routes kube↔cls1-public), and a NetBird client route here
# would shadow the machineconfig fabric route (policy-routing table wins).
routed = {
mgmt = { address = module.addr_site.mgmt_cidr, description = "OOB / vme management network" }
# Internal LB VIPs (Grafana + netops UIs): NetBird peer -> mgmt router -> spine
@@ -26,9 +31,9 @@ locals {
# this range via the spine IRB (10.40.10.1).
lb_internal = { address = module.addr_site.lb_internal_cidr, description = "father internal LoadBalancer VIPs (netops UIs)" }
kube = { address = module.addr_site.kube_cidr, description = "Site-global kube node network (fabric)" }
cls1_public = { address = module.addr_cls1.public_cidr, description = "cls1 public cluster network" }
cls1_private = { address = module.addr_cls1.private_cidr, description = "cls1 private cluster network" }
cls1_host_mgmt = { address = module.addr_cls1.host_mgmt_cidr, description = "cls1 host-management network" }
cls1_public = { address = module.addr_cls1.public_cidr, description = "cls1 public cluster network", groups = ["ceph_nets"] }
cls1_private = { address = module.addr_cls1.private_cidr, description = "cls1 private cluster network", groups = ["ceph_nets"] }
cls1_host_mgmt = { address = module.addr_cls1.host_mgmt_cidr, description = "cls1 host-management network", groups = ["ceph_nets"] }
}
netbird_networks = {
@@ -39,7 +44,7 @@ locals {
for name, r in local.routed : name => {
address = r.address
description = r.description
groups = ["resources"]
groups = try(r.groups, ["resources"])
}
}
}
+23 -1
View File
@@ -64,6 +64,29 @@ provider "registry.opentofu.org/hashicorp/kubernetes" {
]
}
provider "registry.opentofu.org/hashicorp/tls" {
version = "4.3.0"
constraints = "~> 4.0"
hashes = [
"h1:ZxKvDInYHzss9rv75M778pInFm08ME6hY31XMyFP4IA=",
"zh:07bb8c6e64124dada7dff57a38a46f2f323b3fd77920404c0c550293d1cf6188",
"zh:0b3bfda2df39c52f1c5452d05cf3107bedd5d20ab6977c90ede540c695fb6c3e",
"zh:110a055289f0400a63ac172bedb0e671d059b7a5ba22d4a3f5f246ccac0ad676",
"zh:15e532d8c711377499dece832e60170a8bef39830125b8154f4bda81d9721d29",
"zh:22ca65d96e9fc1be5605372d855c9e1eba2d86d510f7ac8593968f5649435e47",
"zh:36df38dfd03e8c1298c5704fd85e28b69a3927ed0b339f9628d0b56dac99c6b5",
"zh:429e2bfcb81656e1fe90b7b284767d1453c1a4100b16d27e4b29c34aa12f0ce1",
"zh:5b6679953065f0279bf018426c6fb06dd93a851a7a9369f2e3a1fec5bc417e83",
"zh:6a72c88d5aa945ddb32041350755377c96681563136decfe7e05c7cdea7988f1",
"zh:6f05757c50da9f8354a735b5756bd63a71126fcd142129525b90c56bfd081d61",
"zh:751703b7a4d40c3a111c4ed0d5da3ec91c14f880faf6f010a5000a2eb5366011",
"zh:87a5279e61b8198798a2fe86cfe3b74e5340bb486f4e148bb5b4d46f860cf1db",
"zh:942af95e9fd73327a7e9ab0803c4d701b782ddacd78c9b7ce9c91e38b3051522",
"zh:a457d0efea3c404178a182d240ba21cdeb0c620ffabeeb9a8977b024a85e1360",
"zh:d5eac8f4f0ae1ff41cbcc1008e6a74a8491dc27f4c6e5a0c32c5c4b6ef2e4087",
]
}
provider "registry.opentofu.org/siderolabs/talos" {
version = "0.11.0"
constraints = "~> 0.11"
@@ -85,4 +108,3 @@ provider "registry.opentofu.org/siderolabs/talos" {
"zh:d218bab0f67a2a8b15add9b51df3d30f514b57e9a7c1d733ebe97966ea132acb",
]
}
@@ -20,4 +20,8 @@ locals {
kube_gateway = cidrhost(module.addr_site.kube_cidr, 1) # .1 IRB on the spine
kube_cp_cidr = module.addr_site.kube_cp_cidr # 10.40.11.0/24 (fabric VLAN 11)
kube_cp_gw = module.addr_site.kube_cp_gateway # .1 IRB on the spine
# cls1 public network (spice RGW frontend) — routed by the spine (fabric stack
# public_routing); michael's S3 path.
cls1_public_cidr = module.addr_cls1.public_cidr # 10.40.20.0/23 (fabric VLAN 120)
}
@@ -49,3 +49,148 @@ resource "kubernetes_secret_v1" "cloudflare_api_token" {
}
}
}
# ─── App secrets (yucca workload set) ────────────────────────────────────────
# Mirrors staging/austin/talos/secrets.tf: this stack is the 1P integration
# point, so it owns the app secret material. Each Secret is named after its
# chart's fullnameOverride (the chart's dev `secretData` fixture is nulled in
# the HelmReleases) so the apps' envFrom picks these up unchanged.
# ES256 (P-256) JWT keypair: yucca-api signs, michael verifies. The 1P record
# is the survives-state-loss source of truth; prevent_destroy because rotating
# it invalidates every issued token AND michael's verification of them.
resource "tls_private_key" "yucca_jwt" {
algorithm = "ECDSA"
ecdsa_curve = "P256"
lifecycle {
prevent_destroy = true
}
}
resource "onepassword_item" "yucca_jwt" {
vault = data.onepassword_vault.prod.uuid
title = "YUCCA_JWT_KEYPAIR"
category = "password"
password = tls_private_key.yucca_jwt.private_key_pem_pkcs8
section {
label = "keypair"
field {
label = "public_key"
type = "STRING"
value = tls_private_key.yucca_jwt.public_key_pem
}
}
}
# Namespaces created here so the Secrets have a home before Flux reconciles;
# the Flux overlays declare them too (bare Namespace is safe under dual SSA).
resource "kubernetes_namespace_v1" "yucca" {
metadata {
name = "yucca"
}
}
resource "kubernetes_namespace_v1" "observability" {
metadata {
name = "observability"
}
}
# yucca-api: signs JWTs with the generated private key + its OIDC client creds
# (prod Zitadel, CUSTOMER_ZITADEL_OAUTH_*_YUCCA_WEB / _YUCCA_ORCHESTRATOR).
resource "kubernetes_secret_v1" "yucca_api" {
metadata {
name = "yucca-api"
namespace = kubernetes_namespace_v1.yucca.metadata[0].name
}
data = {
JWT_PRIVATE_KEY = tls_private_key.yucca_jwt.private_key_pem_pkcs8
OIDC_CLIENT_ID = var.yucca_oidc_client_id
OIDC_CLIENT_SECRET = var.yucca_oidc_client_secret
OIDC_DEVICE_CLIENT_ID = var.yucca_oidc_device_client_id
}
lifecycle {
precondition {
condition = length(var.yucca_oidc_client_id) > 0 && length(var.yucca_oidc_client_secret) > 0
error_message = "yucca OIDC client creds are empty — run applies through tf/op-run.sh with OP_ENV_FILE=tf/.env.prod."
}
}
}
# yucca-admin-api: its own OIDC client — not registered yet (mirrors staging:
# empty creds until the admin console launches).
resource "kubernetes_secret_v1" "yucca_admin_api" {
metadata {
name = "yucca-admin-api"
namespace = kubernetes_namespace_v1.yucca.metadata[0].name
}
data = {
OIDC_ADMIN_CLIENT_ID = var.yucca_oidc_admin_client_id
OIDC_ADMIN_CLIENT_SECRET = var.yucca_oidc_admin_client_secret
}
}
# michael: verifies yucca-api's JWTs + the spice RGW svc-yucca-restic S3 keys
# (out-of-band contract items, seeded into the RGW by the ceph ansible).
resource "kubernetes_secret_v1" "yucca_michael" {
metadata {
name = "yucca-michael"
namespace = kubernetes_namespace_v1.yucca.metadata[0].name
}
data = {
JWT_PUBLIC_KEY = tls_private_key.yucca_jwt.public_key_pem
S3_ACCESS_KEY_ID = var.yucca_rgw_access_key_id
S3_SECRET_ACCESS_KEY = var.yucca_rgw_secret_access_key
}
lifecycle {
precondition {
condition = length(var.yucca_rgw_access_key_id) > 0 && length(var.yucca_rgw_secret_access_key) > 0
error_message = "michael RGW S3 keys are empty — run applies through tf/op-run.sh with OP_ENV_FILE=tf/.env.prod."
}
}
}
# yucca-metrics-worker: separate RGW user WITH admin caps (per-bucket usage via
# the spice RGW admin API). Keys are AccessKey/SecretKey to match the chart's
# radosSecretName lookup.
resource "kubernetes_secret_v1" "yucca_metrics_rgw" {
metadata {
name = "yucca-metrics-rgw"
namespace = kubernetes_namespace_v1.yucca.metadata[0].name
}
data = {
AccessKey = var.spice_metrics_worker_access_key
SecretKey = var.spice_metrics_worker_secret_key
}
lifecycle {
precondition {
condition = length(var.spice_metrics_worker_access_key) > 0 && length(var.spice_metrics_worker_secret_key) > 0
error_message = "spice metrics-worker RGW keys are empty — run applies through tf/op-run.sh with OP_ENV_FILE=tf/.env.prod."
}
}
}
# ─── Observability Secret (namespace: observability) ─────────────────────────
# Bearer token vmagent + the logs collector present to o11y's prod vmauth.
resource "kubernetes_secret_v1" "vmagent_remote_write" {
metadata {
name = "vmagent-remote-write"
namespace = kubernetes_namespace_v1.observability.metadata[0].name
}
data = {
token = var.vmauth_remote_write_password
}
lifecycle {
precondition {
condition = length(var.vmauth_remote_write_password) > 0
error_message = "vmauth_remote_write_password is empty — run applies through tf/op-run.sh with OP_ENV_FILE=tf/.env.prod."
}
}
}
@@ -149,3 +149,74 @@ variable "cloudflare_api_token" {
sensitive = true
default = ""
}
# ─── App secrets (secrets.tf) ────────────────────────────────────────────────
# Externally-issued / human-managed creds from yucca_tf_prod (+ shared_tf_prod
# for the vmauth token), injected via TF_VAR from op:// refs in tf/.env.prod.
# Empty defaults keep credential-less `tofu validate` clean; the Secret
# preconditions refuse to ship empties to the cluster.
variable "yucca_oidc_client_id" {
description = "OIDC client ID for yucca-api (prod Zitadel, CUSTOMER_ZITADEL_OAUTH_CLIENT_ID_YUCCA_WEB)."
type = string
default = ""
}
variable "yucca_oidc_client_secret" {
description = "OIDC client secret for yucca-api (prod Zitadel)."
type = string
sensitive = true
default = ""
}
variable "yucca_oidc_device_client_id" {
description = "Public OIDC client ID for yucca-api's device flow (CUSTOMER_ZITADEL_OAUTH_CLIENT_ID_YUCCA_ORCHESTRATOR)."
type = string
default = ""
}
variable "yucca_oidc_admin_client_id" {
description = "OIDC client ID for yucca-admin-api — not registered yet (empty mirrors staging)."
type = string
default = ""
}
variable "yucca_oidc_admin_client_secret" {
description = "OIDC client secret for yucca-admin-api — not registered yet."
type = string
sensitive = true
default = ""
}
variable "yucca_rgw_access_key_id" {
description = "Spice RGW (S3) access key for michael (svc-yucca-restic, out-of-band contract item)."
type = string
default = ""
}
variable "yucca_rgw_secret_access_key" {
description = "Spice RGW (S3) secret key for michael (svc-yucca-restic)."
type = string
sensitive = true
default = ""
}
variable "spice_metrics_worker_access_key" {
description = "Spice RGW admin access key for yucca-metrics-worker (per-bucket usage via the RGW admin API)."
type = string
default = ""
}
variable "spice_metrics_worker_secret_key" {
description = "Spice RGW admin secret key for yucca-metrics-worker."
type = string
sensitive = true
default = ""
}
variable "vmauth_remote_write_password" {
description = "o11y prod vmauth bearer token for vmagent/logs remote-write (shared_tf_prod/O11Y_VICTORIAMETRICS_VMAUTH_PASSWORD)."
type = string
sensitive = true
default = ""
}
@@ -22,5 +22,10 @@ terraform {
source = "1Password/onepassword"
version = "~> 2.1"
}
# App JWT keypair generation (secrets.tf).
tls = {
source = "hashicorp/tls"
version = "~> 4.0"
}
}
}
@@ -6,6 +6,7 @@
# bond0 (2×25G LACP) → vlan 10 (kube) = fabric_ip — nodeIP + worker east-west (50G)
# route to kube-cp (apiserver + VIP) via the kube IRB (10.40.10.1) — the fabric
# path to the control plane; the old wt0 (NetBird) route is retired
# route to cls1-public (spice RGW) via the same IRB — michael's S3 data path
# default route via the Hetzner public NIC (DHCP) for egress
#
# Workers are PROVISIONED to maintenance mode out of band — see the runbook
@@ -49,6 +50,8 @@ locals {
# kubelet→apiserver + geneve to the CPs ride the fabric, not the mesh.
routes = concat(
[{ network = local.kube_cp_cidr, gateway = local.kube_gateway }],
# spice RGW frontend — spine routes kube↔cls1-public (michael S3 path).
[{ network = local.cls1_public_cidr, gateway = local.kube_gateway }],
var.cluster.worker_default_route_via_fabric ? [
{ network = "0.0.0.0/0", gateway = local.kube_gateway },
] : [],
@@ -71,6 +71,20 @@ variable "node_lags" {
EOT
}
variable "public_routing" {
type = object({
cidr = string # the cluster public network (e.g. 10.40.20.0/23)
ip = string # spine IRB address on that VLAN, CIDR form (e.g. 10.40.21.254/23)
})
default = null
description = <<-EOT
When set, the spine gets an IRB on the cluster public VLAN and routes
kube ↔ cls-public between its own IRBs (same pattern as kube↔kube-cp) —
the workers' fabric path to the Ceph RGW frontend. Hosts on the public
VLAN reach the kube net back via `ip` (the ceph ansible adds that route).
EOT
}
variable "cp_node_lags" {
type = map(list(string))
default = {}
+12 -1
View File
@@ -4,7 +4,7 @@
# kube↔kube-cp). Other gateways live on the leaves.
locals {
spine_vlans = merge({
"vlan${var.public_vlan_id}" = { id = var.public_vlan_id, l3 = null }
"vlan${var.public_vlan_id}" = { id = var.public_vlan_id, l3 = var.public_routing == null ? null : "irb.${var.public_vlan_id}" }
"vlan${var.private_vlan_id}" = { id = var.private_vlan_id, l3 = null }
"vlan${var.kube_vlan_id}" = { id = var.kube_vlan_id, l3 = var.node_bgp == null ? null : "irb.${var.kube_vlan_id}" }
"vlan${var.mgmt_vlan_id}" = { id = var.mgmt_vlan_id, l3 = null }
@@ -30,3 +30,14 @@ resource "junos_interface_logical" "kube_cp_irb" {
address { cidr_ip = "${cidrhost(var.kube_cp.cidr, 1)}/${split("/", var.kube_cp.cidr)[1]}" }
}
}
# Public-VLAN IRB — NOT the .1 gateway (that's the leaf); a second L3 presence
# so the spine routes kube↔cls-public for the workers' RGW path. Public-VLAN
# hosts route the kube net back via this address.
resource "junos_interface_logical" "public_irb" {
count = var.public_routing == null ? 0 : 1
name = "irb.${var.public_vlan_id}"
family_inet {
address { cidr_ip = var.public_routing.ip }
}
}