Files
yucca/kubernetes/apps/base/michael/helmrelease.yaml
T

109 lines
4.3 KiB
YAML

---
# yaml-language-server: $schema=https://k8s-schemas.home-operations.com/helm.toolkit.fluxcd.io/helmrelease_v2.json
apiVersion: helm.toolkit.fluxcd.io/v2
kind: HelmRelease
metadata:
name: michael
spec:
interval: 1h
# Well above helm's 5m default: with maxUnavailable 0 a terminating pod holds
# its slot until its in-flight restic requests drain, so father's 12 gateways
# roll at the pace of DRAIN_DELAY_MS + SHUTDOWN_TIMEOUT_MS per wave.
timeout: 15m
chart:
spec:
chart: charts/apps/michael
# Repackage on every git revision — the in-repo charts keep a static
# version, so the default ChartVersion strategy never ships template edits.
reconcileStrategy: Revision
sourceRef:
kind: GitRepository
name: ${CHART_SOURCE:=flux-system}
namespace: flux-system
install:
remediation:
retries: 3
upgrade:
cleanupOnFail: true
remediation:
retries: 3
values:
# Per-cluster fleet size (staging 2, father 12 — the 60+ Gbps data plane).
replicas: ${MICHAEL_REPLICAS}
image:
repository: ghcr.io/immich-app/yucca/michael
# Substituted from the image-versions ConfigMap: staging's is rewritten
# in-cluster to the latest CI build (flux-operator ResourceSet), prod's is
# release-please-stamped in git. If unset, the empty default lets the
# chart fall back to v<appVersion> — the matching release image.
tag: ${YUCCA_IMAGE_TAG:=}
# secretData nulled: the chart's dev JWT_PUBLIC_KEY fixture is replaced by
# the TF-provisioned `yucca-michael` Secret (JWT_PUBLIC_KEY + S3 creds),
# which the chart's `envFrom: secretRef: yucca-michael` picks up unchanged.
secretData: null
env:
- name: RESTIC_API_PORT
value: "3010"
- name: LOG_LEVEL
value: info
# Rollout drain. DRAIN_DELAY_MS is the window between /readyz going 503
# and the listener closing — it has to outlast readiness detection (2s
# probe x 2 failures) plus EndpointSlice and Envoy EDS propagation, or the
# gateway is still steering fresh restic requests at a closing socket.
# SHUTDOWN_TIMEOUT_MS then caps how long an in-flight blob transfer gets
# to finish. The chart's terminationGracePeriodSeconds must stay above
# their sum.
- name: DRAIN_DELAY_MS
value: "15000"
- name: SHUTDOWN_TIMEOUT_MS
value: "120000"
# Must match the cluster's STORAGE_CLUSTER_CODE (topology ConfigMap):
# yucca-api mints restic tokens with a storageCluster claim, and michael
# fails closed on cluster codes it does not front.
- name: S3_DEFAULT_CLUSTER
value: ${STORAGE_CLUSTER_CODE}
- name: SITE_CODE
value: ${SITE_CODE}
- name: S3_TOPOLOGY_FILE
value: /etc/yucca/topology.json
- name: S3_ENDPOINT
value: ${S3_ENDPOINT}
- name: S3_REGION
value: us-east-1
- name: S3_FORCE_PATH_STYLE
value: "true"
# --- built-in RGW load balancing (replaces rgw-haproxy) ---
- name: S3_BACKEND_SOURCE
value: dns
- name: S3_BACKEND_DNS_HOST
value: ${S3_HOST}
# S3_BACKEND_SCHEME/PORT are intentionally omitted: michael derives them
# from S3_ENDPOINT (https -> 443), so a DNS source only needs the host.
- name: S3_BACKEND_PIN_HOST
value: "true"
- name: S3_TLS_SKIP_VERIFY
value: "true"
- name: S3_PROBE_BUCKET
value: michael-rgw-healthcheck
# Ejection / health tuning (these are michael's defaults, surfaced here so
# they're tunable per-env): eject a gateway after this many consecutive
# transport failures, and re-resolve DNS + probe every interval (also how
# fast an ejected gateway is reinstated).
- name: S3_EJECT_THRESHOLD
value: "3"
- name: S3_RECONCILE_INTERVAL_MS
value: "5000"
# Cross-backend retry budget (michael's defaults, surfaced likewise): a
# retry spends COST tokens, every success earns EARN back, CAP bounds how
# many retries a burst can spend before the budget must refill.
- name: S3_RETRY_TOKEN_EARN
value: "1"
- name: S3_RETRY_TOKEN_COST
value: "10"
- name: S3_RETRY_BUDGET_CAP
value: "200"
- name: OTLP_METRICS_ENDPOINT
value: ${VMAGENT_OTLP}
- name: OTLP_METRICS_URL_PATH
value: /opentelemetry/v1/metrics