feat(k8s): fleet topology and meta discovery groundwork (#380)

This commit is contained in:
Antoine Lecompte
2026-07-30 08:29:28 -04:00
committed by GitHub
parent 7306bc4b4d
commit 046aacb6b1
37 changed files with 643 additions and 9 deletions
+66 -2
View File
@@ -12,8 +12,8 @@ set -euo pipefail
# Charts are role-grouped: apps/* (services), platform/* (operators/CRs),
# lib/yucca-common (shared library), dev/mock-oidc (dev-only). Paths below are
# relative to charts/.
LIB_CONSUMERS=(apps/yucca-api apps/yucca-admin-api apps/yucca-metrics-worker apps/web apps/michael dev/mock-oidc)
ALL_CHARTS=(apps/yucca-api apps/yucca-admin-api apps/yucca-metrics-worker apps/web apps/michael dev/mock-oidc platform/cnpg-cluster platform/ceph-objectuser platform/rook-ceph-cluster)
LIB_CONSUMERS=(apps/yucca-api apps/yucca-admin-api apps/yucca-metrics-worker apps/web apps/meta apps/michael dev/mock-oidc)
ALL_CHARTS=(apps/yucca-api apps/yucca-admin-api apps/yucca-metrics-worker apps/web apps/meta apps/michael dev/mock-oidc platform/cnpg-cluster platform/ceph-objectuser platform/rook-ceph-cluster)
echo "==> helm dependency build (yucca-common consumers)"
for c in "${LIB_CONSUMERS[@]}"; do
@@ -40,6 +40,70 @@ echo "==> kustomize build (Flux entrypoints)"
# restrictor would reject even though Flux allows it.
kb() { kustomize build --load-restrictor=LoadRestrictionsNone "$@"; }
CLUSTERS=(staging/austin prod/htz-fsn1 dev/local)
echo "==> topology substitution safety"
# The real-cluster topology is a JSON document after Flux postBuild
# substitution. Reject characters that would escape its JSON string literals,
# and enforce the immutable internal-code convention before Flux sees them.
node <<'NODE'
const fs = require('node:fs');
const environments = [
'kubernetes/clusters/staging/austin',
'kubernetes/clusters/prod/htz-fsn1',
];
for (const directory of environments) {
const files = [`${directory}/cluster-settings.yaml`, `${directory}/cluster-settings.generated.yaml`];
const text = files.map((file) => fs.readFileSync(file, 'utf8')).join('\n');
const value = (key) => {
const match = text.match(new RegExp(`^\\s*${key}:\\s*(.+)$`, 'm'));
if (!match) throw new Error(`${directory}: missing ${key}`);
const raw = match[1].trim().replace(/\s+#.*$/, '');
return raw.startsWith('"') ? JSON.parse(raw) : raw;
};
const site = value('SITE_CODE');
const cluster = value('STORAGE_CLUSTER_CODE');
const legacySite = value('LEGACY_SITE_CODE');
const legacyCluster = value('LEGACY_STORAGE_CLUSTER_CODE');
if (!/^[a-z0-9][a-z0-9-]{0,63}$/.test(site)) throw new Error(`${directory}: invalid SITE_CODE ${site}`);
if (!cluster.startsWith(`${site}-`) || !/^[a-z0-9][a-z0-9-]{0,63}$/.test(cluster)) {
throw new Error(`${directory}: STORAGE_CLUSTER_CODE ${cluster} must start with ${site}-`);
}
if (!/^[a-z0-9][a-z0-9-]{0,63}$/.test(legacySite)) {
throw new Error(`${directory}: invalid LEGACY_SITE_CODE ${legacySite}`);
}
if (
!legacyCluster.startsWith(`${legacySite}-`) ||
!/^[a-z0-9][a-z0-9-]{0,63}$/.test(legacyCluster)
) {
throw new Error(`${directory}: LEGACY_STORAGE_CLUSTER_CODE ${legacyCluster} must start with ${legacySite}-`);
}
for (const key of [
'SITE_DISPLAY_NAME',
'SITE_DESCRIPTION',
'STORAGE_CLUSTER_DISPLAY_NAME',
'GW_HOST',
'S3_ENDPOINT',
'S3_HOST',
]) {
const label = value(key);
if (/["\\\u0000-\u001f]/.test(label)) {
throw new Error(`${directory}: ${key} contains a character that is unsafe for topology JSON substitution`);
}
}
for (const key of ['GW_HOST', 'S3_HOST']) {
const host = value(key);
if (!/^[a-z0-9](?:[a-z0-9.-]*[a-z0-9])?$/.test(host)) {
throw new Error(`${directory}: invalid ${key} ${host}`);
}
}
const s3Url = new URL(value('S3_ENDPOINT'));
if (!['http:', 'https:'].includes(s3Url.protocol)) {
throw new Error(`${directory}: S3_ENDPOINT must use HTTP(S)`);
}
}
NODE
echo " OK topology identifiers and labels"
for c in "${CLUSTERS[@]}"; do
kb "kubernetes/clusters/$c" >/dev/null && echo " OK kubernetes/clusters/$c"
kb "kubernetes/apps/$c" >/dev/null && echo " OK kubernetes/apps/$c"
+23 -5
View File
@@ -72,6 +72,17 @@ if DEV_ENV:
}))
k8s_resource(objects=['yucca-dev-env:secret'], new_name='dev-env', labels=['helm'])
# ---------------------------------------------------------------------------
# Fleet topology. The real clusters get this ConfigMap from Flux with the
# per-cluster values substituted in (kubernetes/apps/base/topology); Tilt
# deploys only HelmReleases, so the dev copy — the same file the dev-mirror
# Flux tree applies — is applied here by hand. yucca-api, yucca-admin-api and
# yucca-metrics-worker mount it and parse it at boot, so it has to exist before
# they start (see their APP_WIRING deps below).
# ---------------------------------------------------------------------------
k8s_yaml('kubernetes/apps/dev/local/yucca/topology/app/configmap.yaml')
k8s_resource(objects=['yucca-topology:configmap'], new_name='yucca-topology', labels=['app'])
# ---------------------------------------------------------------------------
# Images. Built locally and injected into each app's Helm release via image_keys
# (image.repository/image.tag). Edits are live-synced into the running pods.
@@ -228,12 +239,13 @@ docker_build(
# ---------------------------------------------------------------------------
local_resource(
'helm-deps',
cmd='rm -rf charts/apps/yucca-api/charts charts/apps/yucca-admin-api/charts charts/apps/yucca-metrics-worker/charts charts/apps/web/charts charts/apps/michael/charts charts/dev/mock-oidc/charts && for d in charts/apps/yucca-api charts/apps/yucca-admin-api charts/apps/yucca-metrics-worker charts/apps/web charts/apps/michael charts/dev/mock-oidc; do (cd $d && helm dependency build); done',
cmd='rm -rf charts/apps/yucca-api/charts charts/apps/yucca-admin-api/charts charts/apps/yucca-metrics-worker/charts charts/apps/web/charts charts/apps/meta/charts charts/apps/michael/charts charts/dev/mock-oidc/charts && for d in charts/apps/yucca-api charts/apps/yucca-admin-api charts/apps/yucca-metrics-worker charts/apps/web charts/apps/meta charts/apps/michael charts/dev/mock-oidc; do (cd $d && helm dependency build); done',
deps=[
'charts/apps/yucca-api',
'charts/apps/yucca-admin-api',
'charts/apps/yucca-metrics-worker',
'charts/apps/web',
'charts/apps/meta',
'charts/apps/michael',
'charts/dev/mock-oidc',
'charts/lib/yucca-common',
@@ -260,10 +272,14 @@ APP_WIRING = {
# dev_env: receives the .env override Secret (see load_dev_env above).
# dev_keypair: render the well-known dev JWT fixture into the chart Secret
# (the chart default is useDevKeypair=false so real overlays fail loudly).
'yucca-api': {'build': 'yucca-api', 'deps': ['yucca-database', 'yucca-mock-oidc', 'yucca-michael'], 'dev_env': True, 'dev_keypair': True},
'yucca-admin-api': {'build': 'yucca-admin-api', 'deps': ['yucca-database', 'yucca-mock-oidc'], 'dev_keypair': True},
'yucca-metrics-worker': {'build': 'yucca-metrics-worker', 'deps': ['yucca-database', 'yucca-metrics-object-user'], 'dev_env': True},
'yucca-api': {'build': 'yucca-api', 'deps': ['yucca-database', 'yucca-mock-oidc', 'yucca-michael', 'yucca-topology'], 'dev_env': True, 'dev_keypair': True},
'yucca-admin-api': {'build': 'yucca-admin-api', 'deps': ['yucca-database', 'yucca-mock-oidc', 'yucca-topology'], 'dev_keypair': True},
'yucca-metrics-worker': {'build': 'yucca-metrics-worker', 'deps': ['yucca-database', 'yucca-metrics-object-user', 'yucca-topology'], 'dev_env': True},
'yucca-web': {'build': 'web', 'deps': ['yucca-api']},
# Stock upstream nginx serving the .well-known pointer — nothing to build,
# nothing to wait for (it's a static file, deliberately independent of the
# API whose URL it advertises).
'yucca-meta': {'build': None, 'deps': []},
'yucca-michael': {'build': 'michael', 'deps': ['yucca-object-user'], 'dev_keypair': True},
'yucca-mock-oidc': {'build': 'mock-oidc-provider', 'deps': []},
'yucca-database': {'build': None, 'deps': ['cloudnative-pg']},
@@ -456,19 +472,21 @@ local_resource(
kubectl port-forward -n yucca svc/yucca-api 3020:3020 &
kubectl port-forward -n yucca svc/yucca-admin-api 3030:3030 &
kubectl port-forward -n yucca svc/yucca-web 5173:5173 &
kubectl port-forward -n yucca svc/yucca-meta 8081:8080 &
kubectl port-forward -n yucca svc/yucca-michael 3010:3010 &
kubectl port-forward -n yucca svc/yucca-mock-oidc 8092:8092 &
kubectl port-forward -n rook-ceph svc/rook-ceph-rgw-yucca 9000:80 &
kubectl port-forward -n yucca svc/victoria-metrics 8428:8428 &
kubectl port-forward -n yucca svc/victoria-logs 9428:9428 &
wait''',
resource_deps=['yucca-api', 'yucca-web', 'yucca-michael', 'yucca-mock-oidc'],
resource_deps=['yucca-api', 'yucca-web', 'yucca-michael', 'yucca-mock-oidc', 'yucca-meta'],
labels=['app'],
links=[
link('http://localhost:5173', 'web'),
link('http://localhost:3020', 'yucca-api'),
link('http://localhost:3030', 'yucca-admin-api'),
link('http://localhost:3010', 'michael'),
link('http://localhost:8081/.well-known/yucca.json', 'meta (.well-known)'),
link('http://localhost:8092', 'mock-oidc'),
link('http://localhost:9000', 'ceph rgw (s3)'),
link('http://localhost:8428', 'victoria-metrics'),
+13
View File
@@ -0,0 +1,13 @@
apiVersion: v2
name: meta
description: Yucca .well-known discovery pointer (static nginx)
type: application
version: 0.1.0
# Upstream nginx, not a yucca image — this chart is deliberately OUT of the
# release-please single-version scheme (release-please-config.json lists the
# other apps' Chart.yaml, not this one). The tag in values.yaml is what ships.
appVersion: "1.29"
dependencies:
- name: yucca-common
version: 0.2.0
repository: "file://../../lib/yucca-common"
+15
View File
@@ -0,0 +1,15 @@
{{- /*
The served document. Chart-owned (rather than a raw manifest in the Flux tree)
so `helm template charts/apps/meta` is a complete, runnable rendering and dev
needs no second source of truth: the per-cluster value arrives through the
HelmRelease, which postBuild-substitutes ${APP_DOMAIN} exactly like a raw
ConfigMap would have.
*/}}
apiVersion: v1
kind: ConfigMap
metadata:
name: {{ include "yucca-common.fullname" . }}
labels:
{{- include "yucca-common.labels" . | nindent 4 }}
data:
yucca.json: {{ .Values.wellKnown | toJson | quote }}
@@ -0,0 +1 @@
{{- include "yucca-common.deployment" . }}
+1
View File
@@ -0,0 +1 @@
{{- include "yucca-common.pdb" . }}
+1
View File
@@ -0,0 +1 @@
{{- include "yucca-common.service" . }}
+82
View File
@@ -0,0 +1,82 @@
# The discovery pointer: a static nginx serving exactly
# /.well-known/yucca.json — the one URL a client can hard-code. The JSON names
# the API root, and yucca-api's /meta serves everything else (sites, rest_url,
# config). Deliberately a separate pod from yucca-api: the pointer must stay
# reachable on a stable vendor host (meta.futo.cloud) independent of whichever
# domain the API currently lives on.
# 2 replicas + a PDB (templates/pdb.yaml): the pointer is the first thing every
# client touches, and it costs ~nothing to run two.
replicas: 2
# Tiny: nginx serving one 60-byte file. The limit looks oversized for that
# because nginx's stock `worker_processes auto` forks one worker per HOST cpu
# (~48 on the father workers, measured ~0.6MiB private each) — the image's
# autotune entrypoint can't help, it bails on a read-only rootfs since it
# rewrites /etc/nginx/nginx.conf. Headroom is cheaper than a custom nginx.conf.
resources:
requests: { cpu: 10m, memory: 32Mi }
limits: { memory: 256Mi }
# Stable in-cluster name, independent of the Helm release name (dev == prod).
fullnameOverride: yucca-meta
# Unprivileged upstream nginx: listens on 8080 as uid 101 and relocates every
# runtime path (pid, *_temp) under /tmp, so it runs unchanged with the chart's
# read-only rootfs. Pinned to the 1.29 minor — patch releases roll in on pull.
image:
repository: docker.io/nginxinc/nginx-unprivileged
tag: 1.29-alpine
pullPolicy: IfNotPresent
ports:
- name: http
containerPort: 8080
service:
type: ClusterIP
# WHOLESALE override of the library chart's uid/gid 1000 default: this image's
# user is 101, and kubelet can't verify runAsNonRoot against a named user.
podSecurityContext:
runAsNonRoot: true
runAsUser: 101
runAsGroup: 101
seccompProfile:
type: RuntimeDefault
# Serialized verbatim (as JSON) into the ConfigMap that becomes
# /.well-known/yucca.json. The real clusters override meta_url from the
# HelmRelease (https://${APP_DOMAIN}/api/meta); this default points at the dev
# port-forward, which is where a browser-side client reaches yucca-api in k3d.
wellKnown:
meta_url: http://localhost:3020/api/meta
volumes:
# Mounted as a DIRECTORY (not subPath) so a ConfigMap edit propagates into
# the running pod — nginx reopens the file per request, no restart needed.
- name: well-known
configMap:
name: yucca-meta
# nginx-unprivileged keeps its temp paths in /tmp (already an emptyDir from
# the library chart), but the image still creates /var/cache/nginx and the
# rootfs is read-only. A tmpfs here keeps startup quiet.
- name: cache
emptyDir: {}
volumeMounts:
- name: well-known
mountPath: /usr/share/nginx/html/.well-known
readOnly: true
- name: cache
mountPath: /var/cache/nginx
# Probe the real artifact, not the port: a missing/unreadable ConfigMap mount
# would otherwise leave a pod Ready while serving 404 to every client.
startupProbe:
httpGet: { path: /.well-known/yucca.json, port: http }
periodSeconds: 2
failureThreshold: 30
readinessProbe:
httpGet: { path: /.well-known/yucca.json, port: http }
periodSeconds: 10
@@ -4,4 +4,5 @@ kind: Kustomization
resources:
- ./web.yaml
- ./gw.yaml
- ./meta.yaml
- ./redirect.yaml
+27
View File
@@ -0,0 +1,27 @@
---
# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/gateway.networking.k8s.io/httproute_v1.json
# The discovery pointer at ${META_HOST} — a host that deliberately does NOT
# move with the product domain, so shipped clients can hard-code it. Only
# /.well-known/ is routed: everything else on this host (including nginx's
# stock welcome page) falls through to the gateway's 404.
#
# The :80 half is covered by redirect.yaml's catch-all 301 (no hostname filter).
apiVersion: gateway.networking.k8s.io/v1
kind: HTTPRoute
metadata:
name: meta
spec:
parentRefs:
- name: envoy
namespace: envoy-system
sectionName: https
hostnames:
- "${META_HOST}"
rules:
- matches:
- path:
type: PathPrefix
value: /.well-known/
backendRefs:
- name: yucca-meta
port: 8080
@@ -0,0 +1,34 @@
---
# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/helm.toolkit.fluxcd.io/helmrelease_v2.json
apiVersion: helm.toolkit.fluxcd.io/v2
kind: HelmRelease
metadata:
name: yucca-meta
spec:
interval: 1h
chart:
spec:
chart: charts/apps/meta
# Repackage on every git revision — the in-repo charts keep a static
# version, so the default ChartVersion strategy never ships template edits.
reconcileStrategy: Revision
sourceRef:
kind: GitRepository
name: ${CHART_SOURCE:=flux-system}
namespace: flux-system
install:
remediation:
retries: 3
upgrade:
cleanupOnFail: true
remediation:
retries: 3
values:
# Stock upstream nginx — no image override here (nothing to stamp with
# ${YUCCA_IMAGE_TAG}); the chart pins the tag.
#
# The whole point of the pod: hand a client the API root it should ask for
# everything else. Rendered into the chart's ConfigMap and served at
# /.well-known/yucca.json.
wellKnown:
meta_url: https://${APP_DOMAIN}/api/meta
@@ -0,0 +1,5 @@
---
apiVersion: kustomize.config.k8s.io/v1beta1
kind: Kustomization
resources:
- ./helmrelease.yaml
@@ -0,0 +1,55 @@
---
# The fleet topology, GitOps-owned: which sites exist, the restic gateway that
# fronts each one, and the Ceph clusters behind it. yucca-api, yucca-admin-api,
# yucca-metrics-worker and michael all mount this at /etc/yucca, validate it at
# boot, and are rolled on change by their charts' topology checksum annotation.
#
# Today every cluster is single-site: the site IS this region, and its one
# storage cluster is the active one. A second Ceph cluster in a region means a
# second `clusters` entry (active: false until it takes writes); a second REGION
# means a second `sites` entry, which is a cross-cluster edit and doesn't belong
# in a per-cluster substitution — expect this file to grow a real template then.
#
# NB: the ${VAR} tokens are Flux postBuild substitutions, not shell/JSON — the
# literal JSON braces around them pass through untouched.
apiVersion: v1
kind: ConfigMap
metadata:
name: yucca-topology
data:
topology.json: |
{
"schema_version": 1,
"default_site": "${SITE_CODE}",
"sites": [
{
"code": "${SITE_CODE}",
"display_name": "${SITE_DISPLAY_NAME}",
"description": "${SITE_DESCRIPTION}",
"rest_url": "https://${GW_HOST}",
"default_cluster": "${STORAGE_CLUSTER_CODE}",
"clusters": [
{
"code": "${STORAGE_CLUSTER_CODE}",
"display_name": "${STORAGE_CLUSTER_DISPLAY_NAME}",
"active": true,
"rgw_admin_endpoint": "${S3_ENDPOINT}",
"s3": {
"endpoint": "${S3_ENDPOINT}",
"region": "us-east-1",
"force_path_style": true,
"access_key_env": "S3_ACCESS_KEY_ID",
"secret_key_env": "S3_SECRET_ACCESS_KEY",
"backend_source": "dns",
"backend_dns_host": "${S3_HOST}",
"pin_host": true,
"tls_skip_verify": true,
"probe_bucket": "michael-rgw-healthcheck",
"eject_threshold": 3,
"reconcile_interval_ms": 5000
}
}
]
}
]
}
@@ -0,0 +1,5 @@
---
apiVersion: kustomize.config.k8s.io/v1beta1
kind: Kustomization
resources:
- ./configmap.yaml
@@ -2,6 +2,7 @@ apiVersion: kustomize.config.k8s.io/v1beta1
kind: Kustomization
resources:
- ./namespace.yaml
- ./topology/ks.yaml
- ./database/ks.yaml
- ./object-user/ks.yaml
- ./metrics-object-user/ks.yaml
@@ -13,3 +14,4 @@ resources:
- ./yucca-api/ks.yaml
- ./yucca-admin-api/ks.yaml
- ./web/ks.yaml
- ./meta/ks.yaml
@@ -0,0 +1,25 @@
---
apiVersion: helm.toolkit.fluxcd.io/v2
kind: HelmRelease
metadata:
name: yucca-meta
namespace: yucca
spec:
interval: 1h
chart:
spec:
chart: charts/apps/meta
sourceRef:
kind: GitRepository
name: yucca
namespace: flux-system
install:
remediation:
retries: 3
upgrade:
cleanupOnFail: true
remediation:
retries: 3
# No values: stock upstream nginx (nothing Tilt builds, so none of the
# dev rootfs/securityContext relaxations the other releases need), and the
# chart's default wellKnown.meta_url already points at the dev port-forward.
@@ -0,0 +1,4 @@
apiVersion: kustomize.config.k8s.io/v1beta1
kind: Kustomization
resources:
- ./helmrelease.yaml
@@ -0,0 +1,20 @@
---
apiVersion: kustomize.toolkit.fluxcd.io/v1
kind: Kustomization
metadata:
name: yucca-meta
namespace: flux-system
spec:
targetNamespace: yucca
commonMetadata:
labels:
app.kubernetes.io/name: yucca-meta
path: ./kubernetes/apps/dev/local/yucca/meta/app
prune: true
wait: true
interval: 1h
retryInterval: 2m
timeout: 5m
sourceRef:
kind: GitRepository
name: yucca
@@ -21,3 +21,5 @@ spec:
dependsOn:
- name: yucca-database
- name: yucca-object-user
# ConfigMap mounted at /etc/yucca and parsed at boot.
- name: yucca-topology
@@ -0,0 +1,51 @@
---
# Dev copy of the fleet topology that Flux renders per cluster in the real
# clusters (kubernetes/apps/base/topology). Written out LITERALLY because the
# dev/local entry point runs no postBuild substitution — there is no
# cluster-settings ConfigMap here, so ${VAR} tokens would reach the services
# verbatim and fail the schema at boot.
#
# Not the same as the in-image packages/*/topology.dev.json fixture, which
# targets `mise dev` (compose, everything on localhost): in k3d the services
# talk to each other over cluster DNS.
#
# Tilt applies this file directly too (see the Tiltfile) — k3d-with-Flux and
# k3d-with-Tilt must agree, so keep it the single dev source of truth.
apiVersion: v1
kind: ConfigMap
metadata:
name: yucca-topology
namespace: yucca
data:
topology.json: |
{
"schema_version": 1,
"default_site": "local",
"sites": [
{
"code": "local",
"display_name": "Local development",
"description": "Compose/k3d development site",
"rest_url": "http://yucca-michael:3010",
"default_cluster": "local-dev",
"clusters": [
{
"code": "local-dev",
"display_name": "Local development storage",
"active": true,
"rgw_admin_endpoint": "http://rook-ceph-rgw-yucca.rook-ceph.svc:80",
"s3": {
"endpoint": "http://rook-ceph-rgw-yucca.rook-ceph.svc:80",
"region": "us-east-1",
"force_path_style": true,
"access_key_env": "S3_ACCESS_KEY_ID",
"secret_key_env": "S3_SECRET_ACCESS_KEY",
"backend_source": "dns",
"backend_dns_host": "rook-ceph-rgw-yucca.rook-ceph.svc",
"probe_bucket": "michael-rgw-healthcheck"
}
}
]
}
]
}
@@ -0,0 +1,4 @@
apiVersion: kustomize.config.k8s.io/v1beta1
kind: Kustomization
resources:
- ./configmap.yaml
@@ -0,0 +1,20 @@
---
apiVersion: kustomize.toolkit.fluxcd.io/v1
kind: Kustomization
metadata:
name: yucca-topology
namespace: flux-system
spec:
targetNamespace: yucca
commonMetadata:
labels:
app.kubernetes.io/name: yucca-topology
path: ./kubernetes/apps/dev/local/yucca/topology/app
prune: true
wait: true
interval: 1h
retryInterval: 2m
timeout: 5m
sourceRef:
kind: GitRepository
name: yucca
@@ -21,3 +21,5 @@ spec:
dependsOn:
- name: yucca-database
- name: yucca-mock-oidc
# ConfigMap mounted at /etc/yucca and parsed at boot.
- name: yucca-topology
@@ -22,3 +22,5 @@ spec:
- name: yucca-database
- name: yucca-mock-oidc
- name: yucca-michael
# ConfigMap mounted at /etc/yucca and parsed at boot.
- name: yucca-topology
@@ -50,3 +50,28 @@ spec:
name: ${MANIFEST_SOURCE:=flux-system}
namespace: flux-system
targetNamespace: envoy-system
# PROD-ONLY: the discovery pointer lives on ${META_HOST} = meta.futo.cloud,
# which is outside the *.${APP_DOMAIN} wildcard the base cert covers (staging's
# META_HOST is under its app domain, so staging needs none of this).
#
# Deliberately an extra SAN on the EXISTING cert rather than a second
# certificateRef on the gateway listener: Gateway API makes a listener whose
# certificateRef can't be resolved Programmed=False, so a new ref would take
# the whole https listener (web + api) down for as long as it takes
# cert-manager to issue — an outage designed into a routine merge. Editing
# dnsNames instead just triggers a reissue; the old Secret keeps serving
# throughout and the listener never changes.
#
# NB: kustomize applies this patch, THEN postBuild substitutes — which is why
# a ${VAR} in the patch value resolves. Coupling accepted: a meta.futo.cloud
# DNS-01 failure now blocks renewal of the app cert too, but both names sit in
# the same Cloudflare zone behind the same token, so they fail together anyway.
patches:
- target:
group: cert-manager.io
kind: Certificate
name: app-domain
patch: |-
- op: add
path: /spec/dnsNames/-
value: ${META_HOST}
@@ -26,6 +26,28 @@ data:
# admin OIDC redirect URIs derive from this (see base yucca-admin-api HR).
YUCCA_ADMIN_HOST: admin.father.fsn.htz.yucca.futo.network
# ─── Fleet topology ──────────────────────────────────────────────────
# Rendered into the yucca-topology ConfigMap (apps/base/topology), which
# yucca-api/admin-api/metrics-worker parse at boot. SITE_CODE and
# STORAGE_CLUSTER_CODE are IDENTIFIERS, not labels: they end up in repository
# rows and restic JWT claims, so changing one after launch orphans data.
# Codes are immutable internal identifiers; display names are presentation.
SITE_CODE: father
SITE_DISPLAY_NAME: "Hetzner Falkenstein-1"
SITE_DESCRIPTION: "Located in Falkenstein, Germany"
STORAGE_CLUSTER_CODE: father-spice
STORAGE_CLUSTER_DISPLAY_NAME: "Spice"
# Immutable origin for pre-placement database rows. Never repoint these when
# the active write cluster changes.
LEGACY_SITE_CODE: father
LEGACY_STORAGE_CLUSTER_CODE: father-spice
# The vendor-stable discovery host: clients hard-code
# https://${META_HOST}/.well-known/yucca.json and learn everything else from
# there, so it must NOT move with the product domain. Own cert
# (meta-domain.yaml) — futo.cloud is outside the *.backups.futo.cloud
# wildcard; A record in tf/deployment/prod/global/dns.
META_HOST: meta.futo.cloud
# ─── Ingress entry points ────────────────────────────────────────────
# App gateway (web/api): public pool-a VIP, BGP-advertised /32 covered by the
# 69.48.224.0/24 transit aggregate. lb: public selects pool-a.
@@ -20,6 +20,23 @@ data:
# from this (see base yucca-admin-api HR).
YUCCA_ADMIN_HOST: admin.luke.aus.int.yucca.futo.network
# ─── Fleet topology ──────────────────────────────────────────────────
# Rendered into the yucca-topology ConfigMap (apps/base/topology), which
# Codes are immutable internal identifiers; display names are presentation.
SITE_CODE: luke
SITE_DISPLAY_NAME: "Austin"
SITE_DESCRIPTION: "Located in Austin, Texas, USA"
STORAGE_CLUSTER_CODE: luke-sietch
STORAGE_CLUSTER_DISPLAY_NAME: "Sietch"
# Immutable origin for pre-placement database rows. Never repoint these when
# the active write cluster changes.
LEGACY_SITE_CODE: luke
LEGACY_STORAGE_CLUSTER_CODE: luke-sietch
# Discovery host for the staging fleet. Unlike prod's meta.futo.cloud this
# one sits UNDER the app domain, so the existing *.${APP_DOMAIN} wildcard
# cert and DNS record already cover it — no extra cert/record needed.
META_HOST: meta.staging.backups.futo.cloud
# ─── Ingress entry point ─────────────────────────────────────────────
# Internal VIP the in-cluster LB/Gateway announces (Cilium L2; distinct from
# the 10.10.10.15 control-plane API VIP). Public IP is the NAT in front of it,
+26
View File
@@ -0,0 +1,26 @@
# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/kustomize.toolkit.fluxcd.io/kustomization_v1.json
---
apiVersion: kustomize.toolkit.fluxcd.io/v1
kind: Kustomization
metadata:
name: yucca-meta
namespace: flux-system
spec:
# No dependsOn: the pointer is a static file, and it should come up (and stay
# up) even when the API behind the URL it advertises is down.
healthChecks:
- apiVersion: helm.toolkit.fluxcd.io/v2
kind: HelmRelease
name: yucca-meta
namespace: yucca
interval: 1h
retryInterval: 2m
timeout: 10m
path: ./kubernetes/apps/base/meta
prune: true
wait: true
sourceRef:
kind: GitRepository
name: ${MANIFEST_SOURCE:=flux-system}
namespace: flux-system
targetNamespace: yucca
+9
View File
@@ -7,6 +7,15 @@ metadata:
namespace: flux-system
spec:
# michael load-balances across the RGW daemons itself now (no HAProxy hop).
dependsOn:
- name: yucca-topology
# A merely Ready dependency may still be on the previous Git revision.
# Michael loads topology only at boot, so its Helm render (and checksum)
# must happen after the matching topology revision has been applied.
readyExpr: >-
dep.metadata.generation == dep.status.observedGeneration &&
dep.status.conditions.filter(e, e.type == 'Ready').all(e, e.status == 'True') &&
dep.status.lastAppliedRevision == self.status.lastAttemptedRevision
healthChecks:
- apiVersion: helm.toolkit.fluxcd.io/v2
kind: HelmRelease
@@ -4,7 +4,8 @@
# from any other compromised pod must not reach them. Egress is deliberately
# NOT restricted in this pass (external RGW/OIDC/Polar egress needs FQDN rules
# — a CiliumNetworkPolicy follow-up). Flow inventory:
# envoy (envoy-system) → yucca-api:3020, web:5173, michael:3010 (HTTPRoutes)
# envoy (envoy-system) → yucca-api:3020, web:5173, michael:3010,
# meta:8080 (HTTPRoutes)
# yucca-api → michael = hairpin via the ingress VIP, so it ARRIVES as
# envoy traffic — no direct pod-to-pod allow needed
# web (SSR) → yucca-api:3020
@@ -76,6 +77,27 @@ spec:
---
apiVersion: networking.k8s.io/v1
kind: NetworkPolicy
metadata:
name: allow-ingress-meta
namespace: yucca
spec:
podSelector:
matchLabels:
app.kubernetes.io/name: meta
policyTypes: [Ingress]
ingress:
- from:
- namespaceSelector:
matchLabels:
kubernetes.io/metadata.name: envoy-system
podSelector:
matchLabels:
app.kubernetes.io/name: envoy
ports:
- { port: 8080, protocol: TCP }
---
apiVersion: networking.k8s.io/v1
kind: NetworkPolicy
metadata:
name: allow-ingress-michael
namespace: yucca
+25
View File
@@ -0,0 +1,25 @@
# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/kustomize.toolkit.fluxcd.io/kustomization_v1.json
---
# Just the yucca-topology ConfigMap, but a Kustomization of its own so the
# services that parse it at boot can dependsOn it — a pod that starts before
# the ConfigMap exists sits in ContainerCreating, and one that starts against a
# stale one silently serves the wrong rest_url.
apiVersion: kustomize.toolkit.fluxcd.io/v1
kind: Kustomization
metadata:
name: yucca-topology
namespace: flux-system
spec:
interval: 1h
retryInterval: 2m
timeout: 5m
path: ./kubernetes/apps/base/topology
prune: true
# No healthChecks: `wait` alone is the gate (kstatus reports a ConfigMap
# Current as soon as it applies).
wait: true
sourceRef:
kind: GitRepository
name: ${MANIFEST_SOURCE:=flux-system}
namespace: flux-system
targetNamespace: yucca
@@ -8,6 +8,12 @@ metadata:
spec:
dependsOn:
- name: yucca-database
# Mounted at /etc/yucca and validated on access — it must exist first.
- name: yucca-topology
readyExpr: >-
dep.metadata.generation == dep.status.observedGeneration &&
dep.status.conditions.filter(e, e.type == 'Ready').all(e, e.status == 'True') &&
dep.status.lastAppliedRevision == self.status.lastAttemptedRevision
healthChecks:
- apiVersion: helm.toolkit.fluxcd.io/v2
kind: HelmRelease
@@ -8,6 +8,12 @@ metadata:
spec:
dependsOn:
- name: yucca-database
# Mounted at /etc/yucca and validated on access — it must exist first.
- name: yucca-topology
readyExpr: >-
dep.metadata.generation == dep.status.observedGeneration &&
dep.status.conditions.filter(e, e.type == 'Ready').all(e, e.status == 'True') &&
dep.status.lastAppliedRevision == self.status.lastAttemptedRevision
healthChecks:
- apiVersion: helm.toolkit.fluxcd.io/v2
kind: HelmRelease
@@ -6,9 +6,16 @@ metadata:
name: yucca-metrics-worker
namespace: flux-system
spec:
# Shares yucca-api's database; needs the CNPG app Secret to exist.
# Shares yucca-api's database; needs the CNPG app Secret to exist. The
# topology ConfigMap is mounted and validated on access (rgw_admin_endpoint
# per storage cluster), so it has to land first too.
dependsOn:
- name: yucca-database
- name: yucca-topology
readyExpr: >-
dep.metadata.generation == dep.status.observedGeneration &&
dep.status.conditions.filter(e, e.type == 'Ready').all(e, e.status == 'True') &&
dep.status.lastAppliedRevision == self.status.lastAttemptedRevision
healthChecks:
- apiVersion: helm.toolkit.fluxcd.io/v2
kind: HelmRelease
@@ -12,9 +12,11 @@ kind: Component
resources:
- ../../apps/namespace.yaml
- ../../apps/networkpolicies.yaml
- ../../apps/topology.yaml
- ../../apps/yucca-database.yaml
- ../../apps/yucca-api.yaml
- ../../apps/yucca-admin-api.yaml
- ../../apps/web.yaml
- ../../apps/meta.yaml
- ../../apps/michael.yaml
- ../../apps/yucca-metrics-worker.yaml
@@ -18,6 +18,7 @@ apiVersion: kustomize.config.k8s.io/v1alpha1
kind: Component
resources:
- ../../apps/namespace.yaml
- ../../apps/topology.yaml
# Policies referencing apps a secondary doesn't run are inert (selectors
# match nothing); the default-deny + michael/db rules are what matter here.
- ../../apps/networkpolicies.yaml
@@ -31,6 +31,18 @@ records = {
values = ["69.48.224.6"]
comment = "Yucca prod restic gateway - michael (tf/deployment/prod/global/dns)"
}
# Discovery pointer: /.well-known/yucca.json, served by the yucca-meta pod off
# the SAME app gateway VIP as backups.futo.cloud. Deliberately not under
# backups. — shipped clients hard-code this host, so it must survive the
# product domain changing. Outside the *.backups.futo.cloud wildcard, so the
# app-domain cert carries it as an extra SAN (kubernetes/apps/prod/htz-fsn1/
# infra/envoy.yaml patches it in from META_HOST). proxied false: same DNS-01
# + no-Cloudflare-in-the-path reasoning as the two records above.
"meta.futo.cloud" = {
type = "A"
values = ["69.48.224.5"]
comment = "Yucca meta discovery pointer (tf/deployment/prod/global/dns)"
}
"s3.prod.fsn1.htz.futo.cloud" = {
type = "A"