name: ci on: pull_request: push: branches: [main] # Every job only reads the repo (checkout + tool downloads); nothing writes # back through the token. permissions: contents: read concurrency: group: ci-${{ github.ref }} cancel-in-progress: true jobs: checks: name: Checks & Unit Tests runs-on: ubuntu-latest steps: - uses: actions/checkout@1af3b93b6815bc44a9784bd300feb67ff0d1eeb3 # v6.0.0 with: persist-credentials: false # Advisory only — this runs against origin/main at PR time, so a branch # that goes stale after a migration re-date on main slips past it; the # deploy workflow's gate is the authoritative check. - name: Migration ordering vs main run: packages/yucca-api/.mise/tasks/check-migration-order origin/main - name: Setup Mise uses: jdx/mise-action@c2a87611a18de5b3828c5652fe268e992400cb5c # v4.3.0 - run: mise run prepare - name: Run checks run: mise run check k8s-validate: name: Validate Kubernetes Surface runs-on: ubuntu-latest timeout-minutes: 15 steps: - uses: actions/checkout@1af3b93b6815bc44a9784bd300feb67ff0d1eeb3 # v6.0.0 with: persist-credentials: false - name: Setup Mise uses: jdx/mise-action@c2a87611a18de5b3828c5652fe268e992400cb5c # v4.3.0 # helm template + kubeconform per chart, then flux-local builds the whole # kubernetes/ tree the way Flux would. No cluster involved. One retry: # flux-local downloads third-party charts (rook is still a classic # HelmRepository) and a transient network reset shouldn't fail the build. - name: Validate charts + Flux tree run: mise run k8s:validate || { echo "::warning::k8s:validate failed once (transient chart fetch?); retrying"; mise run k8s:validate; } integration: name: Integration Tests runs-on: ubuntu-latest timeout-minutes: 40 steps: - uses: actions/checkout@1af3b93b6815bc44a9784bd300feb67ff0d1eeb3 # v6.0.0 with: persist-credentials: false - name: Setup Mise uses: jdx/mise-action@c2a87611a18de5b3828c5652fe268e992400cb5c # v4.3.0 # The stack (dev images, k3s) is heavy; the stock runner has ~14GB free, # so drop the biggest unused toolchains up front. - name: Free runner disk space run: sudo rm -rf /usr/local/lib/android /usr/share/dotnet /opt/ghc || true # Cluster bring-up and the workspace install touch none of the same files, # so they overlap. Only the @common/* libs are built: the suites run from # TypeScript source via ts-jest, so `mise prepare`'s other packages are # never loaded here. - name: Stand up k3d infra, install deps and build shared libs run: | ( mise k3d:up && mise tilt:ci-infra ) > /tmp/k3d-infra.log 2>&1 & infra=$! mise run install:frozen mise run //packages/common:build mise run //packages/emails:build wait "$infra" || { echo "::error::k3d infra failed"; cat /tmp/k3d-infra.log; exit 1; } echo "::group::k3d infra log"; cat /tmp/k3d-infra.log; echo "::endgroup::" - name: Run integration tests against the cluster run: mise test:integration:k3d e2e: name: End-to-end Tests # This job hosts the whole k3d stack — Ceph, postgres, michael, a vite dev # server — and runs restic against it. Both halves are CPU-bound, so the # 4-vCPU hosted runner starved them; pokedex-large is 8 cores / 32 GB. runs-on: pokedex-large timeout-minutes: 60 env: # restic runs on the runner, so yucca-api must advertise a localhost # rest_url; rendering it that way up front spares a yucca-api rollout. YUCCA_TOPOLOGY_LOCAL_REST: '1' YUCCA_E2E_PREBUILT: '1' # Loop devices belong to the host kernel, so e2e runs sharing a # pokedex-large host must not all attach Ceph's OSD to the chart's # /dev/loop100. The attempt is part of it because a re-run keeps the run # number and can land on the node still tearing down the failed attempt. YUCCA_CEPH_LOOP_DEVICE: /dev/loop${{ github.run_number }}${{ github.run_attempt }} steps: - uses: actions/checkout@1af3b93b6815bc44a9784bd300feb67ff0d1eeb3 # v6.0.0 with: persist-credentials: false - name: Setup Mise uses: jdx/mise-action@c2a87611a18de5b3828c5652fe268e992400cb5c # v4.3.0 # Paths from the GitHub-hosted image; absent on a self-hosted runner, and # -n so a runner without passwordless sudo fails instead of hanging. - name: Free runner disk space run: sudo -n rm -rf /usr/local/lib/android /usr/share/dotnet /opt/ghc || true - name: Create the k3d cluster run: mise k3d:up # Rook-Ceph converging is the long pole of this job and is nearly all # waiting. `tilt:ci-ceph` builds no images, so unlike the full stack it # never tars the workspace and cannot race the build writing into it — # which is what makes running these together safe. Each of the four # annotates its own failure: unattributed, a workspace build error read # as "Ceph failed to converge" for most of a week. - name: Build the workspace while Ceph converges run: | ( mise tilt:ci-ceph ) > /tmp/k3d-ceph.log 2>&1 & ceph=$! mise run install:frozen || { echo "::error::workspace install failed"; exit 1; } pnpm --filter web exec playwright install --with-deps chromium > /tmp/playwright.log 2>&1 & playwright=$! mise run build || { echo "::error::workspace build failed"; exit 1; } wait "$playwright" || { echo "::error::playwright install failed"; cat /tmp/playwright.log; exit 1; } wait "$ceph" || { echo "::error::Ceph failed to converge"; cat /tmp/k3d-ceph.log; exit 1; } echo "::group::Ceph bring-up log"; cat /tmp/k3d-ceph.log; echo "::endgroup::" # Every OSD in the cluster is blocked on this DaemonSet attaching the # loop device it consumes. Waiting on it here costs nothing on the happy # path (Rook is still converging either way) and turns its failure into # a five-minute error that names the cause, instead of the half-hour # `yucca-michael:runtime CreateContainerConfigError` timeout it reaches # by way of no OSDs -> no RGW -> no object-user Secret. - name: Wait for the OSD loop device run: | kubectl wait --for=create daemonset/rook-ceph-loop-device -n rook-ceph --timeout=2m kubectl rollout status daemonset/rook-ceph-loop-device -n rook-ceph --timeout=5m # Images build here while Rook finishes; yucca-michael mounts the # object-user Secret, so this is where the job blocks on Ceph serving S3. - name: Deploy the app stack run: mise tilt:ci-e2e # Here rather than in the integration job because this stack already runs # Ceph, and standing a second one up there cost more than its whole suite. - name: Run S3-backed integration tests against the cluster run: mise test:integration:s3 - name: Run end-to-end tests against the cluster run: mise test:e2e:k3d # The cluster dies with the runner, so anything not captured here is gone # by the time anyone reads the log. - name: Dump cluster state if: failure() run: | echo "::group::pods"; kubectl get pods -A -o wide; echo "::endgroup::" echo "::group::not-running pods" kubectl get pods -A --field-selector=status.phase!=Running \ -o jsonpath='{range .items[*]}{.metadata.namespace} {.metadata.name}{"\n"}{end}' \ | while read -r ns name; do kubectl describe pod -n "$ns" "$name" || true; done echo "::endgroup::" echo "::group::ceph cluster"; kubectl -n rook-ceph get cephcluster,cephobjectstore -o wide; echo "::endgroup::" echo "::group::host loop devices"; kubectl -n rook-ceph exec ds/rook-ceph-loop-device -- losetup -a || true; echo "::endgroup::" echo "::group::recent events"; kubectl get events -A --sort-by=.lastTimestamp | tail -80; echo "::endgroup::"