mirror of
https://github.com/immich-app/yucca.git
synced 2026-09-30 13:33:00 +08:00
178 lines
7.9 KiB
YAML
178 lines
7.9 KiB
YAML
name: ci
|
|
|
|
on:
|
|
pull_request:
|
|
push:
|
|
branches: [main]
|
|
|
|
# Every job only reads the repo (checkout + tool downloads); nothing writes
|
|
# back through the token.
|
|
permissions:
|
|
contents: read
|
|
|
|
concurrency:
|
|
group: ci-${{ github.ref }}
|
|
cancel-in-progress: true
|
|
|
|
jobs:
|
|
checks:
|
|
name: Checks & Unit Tests
|
|
runs-on: ubuntu-latest
|
|
steps:
|
|
- uses: actions/checkout@1af3b93b6815bc44a9784bd300feb67ff0d1eeb3 # v6.0.0
|
|
with:
|
|
persist-credentials: false
|
|
|
|
# Advisory only — this runs against origin/main at PR time, so a branch
|
|
# that goes stale after a migration re-date on main slips past it; the
|
|
# deploy workflow's gate is the authoritative check.
|
|
- name: Migration ordering vs main
|
|
run: .mise/tasks/yucca-api/check-migration-order origin/main
|
|
|
|
- name: Setup Mise
|
|
uses: jdx/mise-action@c2a87611a18de5b3828c5652fe268e992400cb5c # v4.3.0
|
|
- run: mise run prepare
|
|
|
|
- name: Run checks
|
|
run: mise run check
|
|
|
|
k8s-validate:
|
|
name: Validate Kubernetes Surface
|
|
runs-on: ubuntu-latest
|
|
timeout-minutes: 15
|
|
steps:
|
|
- uses: actions/checkout@1af3b93b6815bc44a9784bd300feb67ff0d1eeb3 # v6.0.0
|
|
with:
|
|
persist-credentials: false
|
|
|
|
- name: Setup Mise
|
|
uses: jdx/mise-action@c2a87611a18de5b3828c5652fe268e992400cb5c # v4.3.0
|
|
|
|
# helm template + kubeconform per chart, then flux-local builds the whole
|
|
# kubernetes/ tree the way Flux would. No cluster involved. One retry:
|
|
# flux-local downloads third-party charts (rook is still a classic
|
|
# HelmRepository) and a transient network reset shouldn't fail the build.
|
|
- name: Validate charts + Flux tree
|
|
run: mise run k8s:validate || { echo "::warning::k8s:validate failed once (transient chart fetch?); retrying"; mise run k8s:validate; }
|
|
|
|
integration:
|
|
name: Integration Tests
|
|
runs-on: ubuntu-latest
|
|
timeout-minutes: 40
|
|
steps:
|
|
- uses: actions/checkout@1af3b93b6815bc44a9784bd300feb67ff0d1eeb3 # v6.0.0
|
|
with:
|
|
persist-credentials: false
|
|
|
|
- name: Setup Mise
|
|
uses: jdx/mise-action@c2a87611a18de5b3828c5652fe268e992400cb5c # v4.3.0
|
|
# The stack (dev images, k3s) is heavy; the stock runner has ~14GB free,
|
|
# so drop the biggest unused toolchains up front.
|
|
- name: Free runner disk space
|
|
run: sudo rm -rf /usr/local/lib/android /usr/share/dotnet /opt/ghc || true
|
|
|
|
# Cluster bring-up and the workspace install touch none of the same files,
|
|
# so they overlap. Only the @common/* libs are built: the suites run from
|
|
# TypeScript source via ts-jest, so `mise prepare`'s other packages are
|
|
# never loaded here.
|
|
- name: Stand up k3d infra, install deps and build shared libs
|
|
run: |
|
|
( mise k3d:up && mise tilt:ci-infra ) > /tmp/k3d-infra.log 2>&1 &
|
|
infra=$!
|
|
mise run install:frozen
|
|
mise run common:server:build
|
|
mise run common:emails:build
|
|
wait "$infra" || { echo "::error::k3d infra failed"; cat /tmp/k3d-infra.log; exit 1; }
|
|
echo "::group::k3d infra log"; cat /tmp/k3d-infra.log; echo "::endgroup::"
|
|
|
|
- name: Run integration tests against the cluster
|
|
run: mise test:integration:k3d
|
|
|
|
e2e:
|
|
name: End-to-end Tests
|
|
# This job hosts the whole k3d stack — Ceph, postgres, michael, a vite dev
|
|
# server — and runs restic against it. Both halves are CPU-bound, so the
|
|
# 4-vCPU hosted runner starved them; pokedex-large is 8 cores / 32 GB.
|
|
runs-on: pokedex-large
|
|
timeout-minutes: 60
|
|
env:
|
|
# restic runs on the runner, so yucca-api must advertise a localhost
|
|
# rest_url; rendering it that way up front spares a yucca-api rollout.
|
|
YUCCA_TOPOLOGY_LOCAL_REST: '1'
|
|
YUCCA_E2E_PREBUILT: '1'
|
|
# Loop devices belong to the host kernel, so e2e runs sharing a
|
|
# pokedex-large host must not all attach Ceph's OSD to the chart's
|
|
# /dev/loop100; the run number is unique across concurrent runs.
|
|
YUCCA_CEPH_LOOP_DEVICE: /dev/loop${{ github.run_number }}
|
|
steps:
|
|
- uses: actions/checkout@1af3b93b6815bc44a9784bd300feb67ff0d1eeb3 # v6.0.0
|
|
with:
|
|
persist-credentials: false
|
|
|
|
- name: Setup Mise
|
|
uses: jdx/mise-action@c2a87611a18de5b3828c5652fe268e992400cb5c # v4.3.0
|
|
# Paths from the GitHub-hosted image; absent on a self-hosted runner, and
|
|
# -n so a runner without passwordless sudo fails instead of hanging.
|
|
- name: Free runner disk space
|
|
run: sudo -n rm -rf /usr/local/lib/android /usr/share/dotnet /opt/ghc || true
|
|
|
|
- name: Create the k3d cluster
|
|
run: mise k3d:up
|
|
|
|
# Rook-Ceph converging is the long pole of this job and is nearly all
|
|
# waiting. `tilt:ci-ceph` builds no images, so unlike the full stack it
|
|
# never tars the workspace and cannot race the build writing into it —
|
|
# which is what makes running these together safe. Each of the four
|
|
# annotates its own failure: unattributed, a workspace build error read
|
|
# as "Ceph failed to converge" for most of a week.
|
|
- name: Build the workspace while Ceph converges
|
|
run: |
|
|
( mise tilt:ci-ceph ) > /tmp/k3d-ceph.log 2>&1 &
|
|
ceph=$!
|
|
mise run install:frozen || { echo "::error::workspace install failed"; exit 1; }
|
|
pnpm --filter web exec playwright install --with-deps chromium > /tmp/playwright.log 2>&1 &
|
|
playwright=$!
|
|
mise run build || { echo "::error::workspace build failed"; exit 1; }
|
|
wait "$playwright" || { echo "::error::playwright install failed"; cat /tmp/playwright.log; exit 1; }
|
|
wait "$ceph" || { echo "::error::Ceph failed to converge"; cat /tmp/k3d-ceph.log; exit 1; }
|
|
echo "::group::Ceph bring-up log"; cat /tmp/k3d-ceph.log; echo "::endgroup::"
|
|
|
|
# Every OSD in the cluster is blocked on this DaemonSet attaching the
|
|
# loop device it consumes. Waiting on it here costs nothing on the happy
|
|
# path (Rook is still converging either way) and turns its failure into
|
|
# a five-minute error that names the cause, instead of the half-hour
|
|
# `yucca-michael:runtime CreateContainerConfigError` timeout it reaches
|
|
# by way of no OSDs -> no RGW -> no object-user Secret.
|
|
- name: Wait for the OSD loop device
|
|
run: |
|
|
kubectl wait --for=create daemonset/rook-ceph-loop-device -n rook-ceph --timeout=2m
|
|
kubectl rollout status daemonset/rook-ceph-loop-device -n rook-ceph --timeout=5m
|
|
|
|
# Images build here while Rook finishes; yucca-michael mounts the
|
|
# object-user Secret, so this is where the job blocks on Ceph serving S3.
|
|
- name: Deploy the app stack
|
|
run: mise tilt:ci-e2e
|
|
|
|
# Here rather than in the integration job because this stack already runs
|
|
# Ceph, and standing a second one up there cost more than its whole suite.
|
|
- name: Run S3-backed integration tests against the cluster
|
|
run: mise test:integration:s3
|
|
|
|
- name: Run end-to-end tests against the cluster
|
|
run: mise test:e2e:k3d
|
|
|
|
# The cluster dies with the runner, so anything not captured here is gone
|
|
# by the time anyone reads the log.
|
|
- name: Dump cluster state
|
|
if: failure()
|
|
run: |
|
|
echo "::group::pods"; kubectl get pods -A -o wide; echo "::endgroup::"
|
|
echo "::group::not-running pods"
|
|
kubectl get pods -A --field-selector=status.phase!=Running \
|
|
-o jsonpath='{range .items[*]}{.metadata.namespace} {.metadata.name}{"\n"}{end}' \
|
|
| while read -r ns name; do kubectl describe pod -n "$ns" "$name" || true; done
|
|
echo "::endgroup::"
|
|
echo "::group::ceph cluster"; kubectl -n rook-ceph get cephcluster,cephobjectstore -o wide; echo "::endgroup::"
|
|
echo "::group::host loop devices"; kubectl -n rook-ceph exec ds/rook-ceph-loop-device -- losetup -a || true; echo "::endgroup::"
|
|
echo "::group::recent events"; kubectl get events -A --sort-by=.lastTimestamp | tail -80; echo "::endgroup::"
|