mirror of
https://github.com/immich-app/yucca.git
synced 2026-09-30 13:33:00 +08:00
fix(ci): give each e2e ceph cluster its own osd loop device (#681)
This commit is contained in:
@@ -100,6 +100,10 @@ jobs:
|
||||
# rest_url; rendering it that way up front spares a yucca-api rollout.
|
||||
YUCCA_TOPOLOGY_LOCAL_REST: '1'
|
||||
YUCCA_E2E_PREBUILT: '1'
|
||||
# Loop devices belong to the host kernel, so e2e runs sharing a
|
||||
# pokedex-large host must not all attach Ceph's OSD to the chart's
|
||||
# /dev/loop100; the run number is unique across concurrent runs.
|
||||
YUCCA_CEPH_LOOP_DEVICE: /dev/loop${{ github.run_number }}
|
||||
steps:
|
||||
- uses: actions/checkout@1af3b93b6815bc44a9784bd300feb67ff0d1eeb3 # v6.0.0
|
||||
with:
|
||||
@@ -169,4 +173,5 @@ jobs:
|
||||
| while read -r ns name; do kubectl describe pod -n "$ns" "$name" || true; done
|
||||
echo "::endgroup::"
|
||||
echo "::group::ceph cluster"; kubectl -n rook-ceph get cephcluster,cephobjectstore -o wide; echo "::endgroup::"
|
||||
echo "::group::host loop devices"; kubectl -n rook-ceph exec ds/rook-ceph-loop-device -- losetup -a || true; echo "::endgroup::"
|
||||
echo "::group::recent events"; kubectl get events -A --sort-by=.lastTimestamp | tail -80; echo "::endgroup::"
|
||||
|
||||
@@ -536,6 +536,8 @@ for app in LOCAL_APPS:
|
||||
flags += ['--set', 'replicas=1']
|
||||
if wiring.get('dev_keypair'):
|
||||
flags += ['--set', 'useDevKeypair=true']
|
||||
if app.name == 'rook-ceph-cluster' and os.getenv('YUCCA_CEPH_LOOP_DEVICE'):
|
||||
flags += ['--set', 'storage.device=%s' % os.getenv('YUCCA_CEPH_LOOP_DEVICE')]
|
||||
if wiring.get('dev_values'):
|
||||
for key in sorted(app.values.keys()):
|
||||
flags += ['--set-json', '%s=%s' % (key, str(encode_json(app.values[key])).rstrip('\n'))]
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
{{- if .Values.loopDevice.enabled }}
|
||||
# DEV ONLY. k3d nodes have no spare block device, and Rook OSDs need raw block
|
||||
# storage (k3d's local-path provisioner is filesystem-only). This privileged
|
||||
# DaemonSet creates a sparse image on the node and losetup-attaches it to a
|
||||
# FIXED loop device (storage.device / loopDevice.device) that the CephCluster
|
||||
# names explicitly. The container stays alive so the attachment persists.
|
||||
# DaemonSet creates a sparse image on the node and losetup-attaches it to the
|
||||
# loop device (storage.device) that the CephCluster names explicitly. The
|
||||
# container stays alive so the attachment persists.
|
||||
#
|
||||
# It runs on cephImage rather than something small like alpine because every
|
||||
# OSD in the cluster is blocked on this attachment: a registry the node has to
|
||||
@@ -37,19 +37,24 @@ spec:
|
||||
- |
|
||||
set -eu
|
||||
IMG="{{ .Values.loopDevice.path }}"
|
||||
DEV="{{ .Values.loopDevice.device }}"
|
||||
DEV="{{ .Values.storage.device }}"
|
||||
mkdir -p "$(dirname "$IMG")"
|
||||
[ -f "$IMG" ] || truncate -s {{ .Values.loopDevice.sizeGiB }}G "$IMG"
|
||||
# Attach the image to the FIXED device the CephCluster references.
|
||||
# Loop attachments live in the HOST kernel (shared by every k3d
|
||||
# node), so they survive cluster re-creation: after a k3d:reset,
|
||||
# DEV is typically still bound to the PREVIOUS cluster's now-
|
||||
# deleted image — which also makes Rook skip it (stale bluestore
|
||||
# signature). Clean all stale state before attaching:
|
||||
if losetup "$DEV" >/dev/null 2>&1; then
|
||||
back="$(losetup --noheadings --output BACK-FILE "$DEV" 2>/dev/null | xargs || true)"
|
||||
# detach unless it's a live attachment of exactly our image
|
||||
[ "$back" = "$IMG" ] || losetup -d "$DEV" || true
|
||||
# Loop attachments live in the HOST kernel, shared by every k3d
|
||||
# cluster on it, and outlive the cluster that made them. One whose
|
||||
# image is deleted belongs to a torn-down cluster (a k3d:reset, or
|
||||
# a finished CI run on the same host) and would also make Rook skip
|
||||
# the device for its stale bluestore signature.
|
||||
for d in $(losetup -a | grep -F "($IMG (deleted))" | cut -d: -f1); do
|
||||
losetup -d "$d" || true
|
||||
done
|
||||
# Every cluster's image has the same path, so BACK-FILE cannot tell
|
||||
# ours from a live neighbour's; -j matches by inode. Sharing a
|
||||
# neighbour's device leaves this cluster with no OSD and surfaces
|
||||
# half an hour later as an unrelated timeout, so fail here instead.
|
||||
if losetup "$DEV" >/dev/null 2>&1 && ! losetup -j "$IMG" | cut -d: -f1 | grep -qx "$DEV"; then
|
||||
echo "$DEV is attached to another live cluster's image; give this cluster its own storage.device" >&2
|
||||
exit 1
|
||||
fi
|
||||
# our image attached on some other device = corruption hazard
|
||||
for d in $(losetup -j "$IMG" | cut -d: -f1); do
|
||||
@@ -62,6 +67,13 @@ spec:
|
||||
echo "loop device for Rook OSD: $(losetup -j "$IMG" | cut -d: -f1) ($IMG)"
|
||||
# keep the container (and thus the loop attachment) alive
|
||||
while true; do sleep 3600; done
|
||||
readinessProbe:
|
||||
exec:
|
||||
command:
|
||||
- /bin/sh
|
||||
- -c
|
||||
- 'losetup -j "{{ .Values.loopDevice.path }}" | cut -d: -f1 | grep -qx "{{ .Values.storage.device }}"'
|
||||
periodSeconds: 5
|
||||
volumeMounts:
|
||||
- name: dev
|
||||
mountPath: /dev
|
||||
|
||||
@@ -29,9 +29,10 @@ probes:
|
||||
storage:
|
||||
# Rook does NOT auto-select loop devices via useAllDevices (even with
|
||||
# allowLoopDevices on the operator) — they must be named explicitly. This is
|
||||
# the loop device attached by the DaemonSet below. A HIGH minor number: the
|
||||
# the loop device the DaemonSet below attaches. A HIGH minor number: the
|
||||
# kernel hands out the lowest free loop devices, so k3s/containerd will have
|
||||
# claimed loop0..N on a fresh node — loop100 is reliably ours.
|
||||
# claimed loop0..N on a fresh node. Loop devices are host-global, so clusters
|
||||
# sharing a host need different ones (CI sets YUCCA_CEPH_LOOP_DEVICE per run).
|
||||
device: /dev/loop100
|
||||
|
||||
# S3 object store (CephObjectStore + RGW).
|
||||
@@ -54,5 +55,3 @@ loopDevice:
|
||||
sizeGiB: 20
|
||||
# where the backing image file lives on the node (under dataDirHostPath)
|
||||
path: /var/lib/rook/osd-loop.img
|
||||
# fixed device the image is attached to (must match storage.device above)
|
||||
device: /dev/loop100
|
||||
|
||||
Reference in New Issue
Block a user