Files
yucca/.github/workflows/ci.yml
T

178 lines
7.9 KiB
YAML

name: ci
on:
pull_request:
push:
branches: [main]
# Every job only reads the repo (checkout + tool downloads); nothing writes
# back through the token.
permissions:
contents: read
concurrency:
group: ci-${{ github.ref }}
cancel-in-progress: true
jobs:
checks:
name: Checks & Unit Tests
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@1af3b93b6815bc44a9784bd300feb67ff0d1eeb3 # v6.0.0
with:
persist-credentials: false
# Advisory only — this runs against origin/main at PR time, so a branch
# that goes stale after a migration re-date on main slips past it; the
# deploy workflow's gate is the authoritative check.
- name: Migration ordering vs main
run: .mise/tasks/yucca-api/check-migration-order origin/main
- name: Setup Mise
uses: jdx/mise-action@c2a87611a18de5b3828c5652fe268e992400cb5c # v4.3.0
- run: mise run prepare
- name: Run checks
run: mise run check
k8s-validate:
name: Validate Kubernetes Surface
runs-on: ubuntu-latest
timeout-minutes: 15
steps:
- uses: actions/checkout@1af3b93b6815bc44a9784bd300feb67ff0d1eeb3 # v6.0.0
with:
persist-credentials: false
- name: Setup Mise
uses: jdx/mise-action@c2a87611a18de5b3828c5652fe268e992400cb5c # v4.3.0
# helm template + kubeconform per chart, then flux-local builds the whole
# kubernetes/ tree the way Flux would. No cluster involved. One retry:
# flux-local downloads third-party charts (rook is still a classic
# HelmRepository) and a transient network reset shouldn't fail the build.
- name: Validate charts + Flux tree
run: mise run k8s:validate || { echo "::warning::k8s:validate failed once (transient chart fetch?); retrying"; mise run k8s:validate; }
integration:
name: Integration Tests
runs-on: ubuntu-latest
timeout-minutes: 40
steps:
- uses: actions/checkout@1af3b93b6815bc44a9784bd300feb67ff0d1eeb3 # v6.0.0
with:
persist-credentials: false
- name: Setup Mise
uses: jdx/mise-action@c2a87611a18de5b3828c5652fe268e992400cb5c # v4.3.0
# The stack (dev images, k3s) is heavy; the stock runner has ~14GB free,
# so drop the biggest unused toolchains up front.
- name: Free runner disk space
run: sudo rm -rf /usr/local/lib/android /usr/share/dotnet /opt/ghc || true
# Cluster bring-up and the workspace install touch none of the same files,
# so they overlap. Only the @common/* libs are built: the suites run from
# TypeScript source via ts-jest, so `mise prepare`'s other packages are
# never loaded here.
- name: Stand up k3d infra, install deps and build shared libs
run: |
( mise k3d:up && mise tilt:ci-infra ) > /tmp/k3d-infra.log 2>&1 &
infra=$!
mise run install:frozen
mise run common:server:build
mise run common:emails:build
wait "$infra" || { echo "::error::k3d infra failed"; cat /tmp/k3d-infra.log; exit 1; }
echo "::group::k3d infra log"; cat /tmp/k3d-infra.log; echo "::endgroup::"
- name: Run integration tests against the cluster
run: mise test:integration:k3d
e2e:
name: End-to-end Tests
# This job hosts the whole k3d stack — Ceph, postgres, michael, a vite dev
# server — and runs restic against it. Both halves are CPU-bound, so the
# 4-vCPU hosted runner starved them; pokedex-large is 8 cores / 32 GB.
runs-on: pokedex-large
timeout-minutes: 60
env:
# restic runs on the runner, so yucca-api must advertise a localhost
# rest_url; rendering it that way up front spares a yucca-api rollout.
YUCCA_TOPOLOGY_LOCAL_REST: '1'
YUCCA_E2E_PREBUILT: '1'
# Loop devices belong to the host kernel, so e2e runs sharing a
# pokedex-large host must not all attach Ceph's OSD to the chart's
# /dev/loop100; the run number is unique across concurrent runs.
YUCCA_CEPH_LOOP_DEVICE: /dev/loop${{ github.run_number }}
steps:
- uses: actions/checkout@1af3b93b6815bc44a9784bd300feb67ff0d1eeb3 # v6.0.0
with:
persist-credentials: false
- name: Setup Mise
uses: jdx/mise-action@c2a87611a18de5b3828c5652fe268e992400cb5c # v4.3.0
# Paths from the GitHub-hosted image; absent on a self-hosted runner, and
# -n so a runner without passwordless sudo fails instead of hanging.
- name: Free runner disk space
run: sudo -n rm -rf /usr/local/lib/android /usr/share/dotnet /opt/ghc || true
- name: Create the k3d cluster
run: mise k3d:up
# Rook-Ceph converging is the long pole of this job and is nearly all
# waiting. `tilt:ci-ceph` builds no images, so unlike the full stack it
# never tars the workspace and cannot race the build writing into it —
# which is what makes running these together safe. Each of the four
# annotates its own failure: unattributed, a workspace build error read
# as "Ceph failed to converge" for most of a week.
- name: Build the workspace while Ceph converges
run: |
( mise tilt:ci-ceph ) > /tmp/k3d-ceph.log 2>&1 &
ceph=$!
mise run install:frozen || { echo "::error::workspace install failed"; exit 1; }
pnpm --filter web exec playwright install --with-deps chromium > /tmp/playwright.log 2>&1 &
playwright=$!
mise run build || { echo "::error::workspace build failed"; exit 1; }
wait "$playwright" || { echo "::error::playwright install failed"; cat /tmp/playwright.log; exit 1; }
wait "$ceph" || { echo "::error::Ceph failed to converge"; cat /tmp/k3d-ceph.log; exit 1; }
echo "::group::Ceph bring-up log"; cat /tmp/k3d-ceph.log; echo "::endgroup::"
# Every OSD in the cluster is blocked on this DaemonSet attaching the
# loop device it consumes. Waiting on it here costs nothing on the happy
# path (Rook is still converging either way) and turns its failure into
# a five-minute error that names the cause, instead of the half-hour
# `yucca-michael:runtime CreateContainerConfigError` timeout it reaches
# by way of no OSDs -> no RGW -> no object-user Secret.
- name: Wait for the OSD loop device
run: |
kubectl wait --for=create daemonset/rook-ceph-loop-device -n rook-ceph --timeout=2m
kubectl rollout status daemonset/rook-ceph-loop-device -n rook-ceph --timeout=5m
# Images build here while Rook finishes; yucca-michael mounts the
# object-user Secret, so this is where the job blocks on Ceph serving S3.
- name: Deploy the app stack
run: mise tilt:ci-e2e
# Here rather than in the integration job because this stack already runs
# Ceph, and standing a second one up there cost more than its whole suite.
- name: Run S3-backed integration tests against the cluster
run: mise test:integration:s3
- name: Run end-to-end tests against the cluster
run: mise test:e2e:k3d
# The cluster dies with the runner, so anything not captured here is gone
# by the time anyone reads the log.
- name: Dump cluster state
if: failure()
run: |
echo "::group::pods"; kubectl get pods -A -o wide; echo "::endgroup::"
echo "::group::not-running pods"
kubectl get pods -A --field-selector=status.phase!=Running \
-o jsonpath='{range .items[*]}{.metadata.namespace} {.metadata.name}{"\n"}{end}' \
| while read -r ns name; do kubectl describe pod -n "$ns" "$name" || true; done
echo "::endgroup::"
echo "::group::ceph cluster"; kubectl -n rook-ceph get cephcluster,cephobjectstore -o wide; echo "::endgroup::"
echo "::group::host loop devices"; kubectl -n rook-ceph exec ds/rook-ceph-loop-device -- losetup -a || true; echo "::endgroup::"
echo "::group::recent events"; kubectl get events -A --sort-by=.lastTimestamp | tail -80; echo "::endgroup::"