fix(ci): de-flake the e2e job (loop-device image, fail-fast wait, diagnostics, hook timeout) (#621)

This commit is contained in:
Antoine Lecompte
2026-09-14 23:36:27 +01:00
committed by GitHub
parent 64b8a63d95
commit 29a0b653e9
4 changed files with 43 additions and 8 deletions
@@ -4,6 +4,13 @@
# DaemonSet creates a sparse image on the node and losetup-attaches it to a
# FIXED loop device (storage.device / loopDevice.device) that the CephCluster
# names explicitly. The container stays alive so the attachment persists.
#
# It runs on cephImage rather than something small like alpine because every
# OSD in the cluster is blocked on this attachment: a registry the node has to
# reach for this pod alone is a single point of failure, and a hung pull of one
# surfaced half an hour later as an unrelated yucca-michael config error. This
# image the node must pull anyway, and Tilt prepulls it the moment the cluster
# exists. It carries losetup/truncate/mknod, so nothing has to be installed.
apiVersion: apps/v1
kind: DaemonSet
metadata:
@@ -22,14 +29,13 @@ spec:
hostPID: true
containers:
- name: loop-device
image: {{ .Values.loopDevice.image }}
image: {{ .Values.cephImage }}
securityContext:
privileged: true
command: ["/bin/sh", "-c"]
args:
- |
set -eu
apk add --no-cache util-linux >/dev/null 2>&1 || true
IMG="{{ .Values.loopDevice.path }}"
DEV="{{ .Values.loopDevice.device }}"
mkdir -p "$(dirname "$IMG")"
@@ -51,7 +51,6 @@ objectStore:
# it; Rook's OSD then consumes it. Disable if your nodes have real disks.
loopDevice:
enabled: true
image: alpine:3.21@sha256:48b0309ca019d89d40f670aa1bc06e426dc0931948452e8491e3d65087abc07d
sizeGiB: 20
# where the backing image file lives on the node (under dataDirHostPath)
path: /var/lib/rook/osd-loop.img