mirror of
https://github.com/immich-app/yucca.git
synced 2026-09-30 13:33:00 +08:00
196 lines
6.9 KiB
YAML
196 lines
6.9 KiB
YAML
---
|
|
# Drift detection: compares expected Ansible state against live cluster.
|
|
# Read-only. No changes. Reports mismatches.
|
|
#
|
|
# Each role owns its own checks in roles/<role>/tasks/drift.yml and appends to
|
|
# the drift_results accumulator; this play sequences them and renders the
|
|
# report. Adding a tunable means touching one role directory, not two files.
|
|
#
|
|
# Usage:
|
|
# scripts/ansible-play.sh drift.yml
|
|
# mise run drift
|
|
|
|
- name: Drift detection
|
|
hosts: ceph_nodes
|
|
become: true
|
|
gather_facts: false
|
|
|
|
vars:
|
|
drift_results: []
|
|
|
|
tasks:
|
|
# --- Per-role checks ---
|
|
#
|
|
# include_role, NOT vars_files. vars_files loads a role's defaults at
|
|
# precedence 14, above inventory group_vars (4) and host_vars (9), so every
|
|
# check compared the live host against the role default and a cluster that
|
|
# legitimately overrode a value would be reported as drifting. include_role
|
|
# loads defaults at precedence 2, where normal layering applies.
|
|
#
|
|
# Order is deliberate: it reproduces the report sequence this play emitted
|
|
# when the checks were inline, so the two can be diffed. security before
|
|
# baseline, and ceph_tuning after the cluster-state block below, for the
|
|
# same reason. Reorder freely once that comparison is no longer useful.
|
|
|
|
- name: OS tuning checks
|
|
ansible.builtin.include_role:
|
|
name: os_tuning
|
|
tasks_from: drift
|
|
|
|
- name: Hardware tuning checks
|
|
ansible.builtin.include_role:
|
|
name: hardware_tuning
|
|
tasks_from: drift
|
|
|
|
- name: Security checks
|
|
ansible.builtin.include_role:
|
|
name: security
|
|
tasks_from: drift
|
|
|
|
- name: Baseline checks
|
|
ansible.builtin.include_role:
|
|
name: baseline
|
|
tasks_from: drift
|
|
|
|
# --- Cluster state (bootstrap only) ---
|
|
#
|
|
# These assert live cluster shape rather than a declared config value, so
|
|
# they read no role defaults and stay at play level.
|
|
|
|
- name: Check OSD status
|
|
ansible.builtin.shell: |
|
|
set -o pipefail
|
|
ceph osd stat --format json 2>/dev/null | python3 -c "
|
|
import sys, json
|
|
d = json.load(sys.stdin)
|
|
print(d.get('num_osds', 0), d.get('num_up_osds', 0))
|
|
"
|
|
args:
|
|
executable: /bin/bash
|
|
register: osd_stat
|
|
when: inventory_hostname in groups['ceph_bootstrap']
|
|
changed_when: false
|
|
|
|
- name: Record OSD drift
|
|
ansible.builtin.set_fact:
|
|
drift_results: >-
|
|
{{ drift_results + [{
|
|
'category': 'ceph',
|
|
'item': 'OSDs up/total',
|
|
'expected': (osd_stat.stdout.split()[0]) ~ '/'
|
|
~ (osd_stat.stdout.split()[0]),
|
|
'actual': (osd_stat.stdout.split()[1]) ~ '/'
|
|
~ (osd_stat.stdout.split()[0]),
|
|
'match': osd_stat.stdout.split()[0]
|
|
== osd_stat.stdout.split()[1]
|
|
}] }}
|
|
when: inventory_hostname in groups['ceph_bootstrap']
|
|
|
|
- name: Check MON quorum
|
|
ansible.builtin.shell: |
|
|
set -o pipefail
|
|
ceph mon stat --format json 2>/dev/null | python3 -c "
|
|
import sys, json
|
|
print(json.load(sys.stdin).get('num_mons', 0))
|
|
"
|
|
args:
|
|
executable: /bin/bash
|
|
register: mon_count
|
|
when: inventory_hostname in groups['ceph_bootstrap']
|
|
changed_when: false
|
|
|
|
# Expect the declared quorum, not one MON per node. ceph_mon holds the hosts
|
|
# whose `roles` include "mon": every node on a small cluster, a pinned subset
|
|
# on a large one (spice runs 5 across 48). Comparing against ceph_nodes read
|
|
# as permanent drift on any cluster that pins its quorum.
|
|
- name: Record MON drift
|
|
ansible.builtin.set_fact:
|
|
drift_results: >-
|
|
{{ drift_results + [{
|
|
'category': 'ceph',
|
|
'item': 'MON count',
|
|
'expected': groups['ceph_mon'] | length | string,
|
|
'actual': mon_count.stdout | trim,
|
|
'match': (mon_count.stdout | trim)
|
|
== (groups['ceph_mon'] | length | string)
|
|
}] }}
|
|
when: inventory_hostname in groups['ceph_bootstrap']
|
|
|
|
- name: Check RGW daemons
|
|
ansible.builtin.shell: |
|
|
set -o pipefail
|
|
ceph orch ls --service-type rgw --format json 2>/dev/null \
|
|
| python3 -c "
|
|
import sys, json
|
|
svcs = json.load(sys.stdin)
|
|
print(sum(s.get('status',{}).get('running',0) for s in svcs))
|
|
"
|
|
args:
|
|
executable: /bin/bash
|
|
register: rgw_count
|
|
when: inventory_hostname in groups['ceph_bootstrap']
|
|
changed_when: false
|
|
|
|
- name: Record RGW drift
|
|
ansible.builtin.set_fact:
|
|
drift_results: >-
|
|
{{ drift_results + [{
|
|
'category': 'ceph',
|
|
'item': 'RGW daemons',
|
|
'expected': groups['ceph_nodes'] | length | string,
|
|
'actual': rgw_count.stdout | trim,
|
|
'match': (rgw_count.stdout | trim)
|
|
== (groups['ceph_nodes'] | length | string)
|
|
}] }}
|
|
when: inventory_hostname in groups['ceph_bootstrap']
|
|
|
|
- name: Check cluster health
|
|
ansible.builtin.command: ceph health --format json
|
|
register: health_check
|
|
when: inventory_hostname in groups['ceph_bootstrap']
|
|
changed_when: false
|
|
|
|
- name: Record health drift
|
|
ansible.builtin.set_fact:
|
|
drift_results: >-
|
|
{{ drift_results + [{
|
|
'category': 'ceph',
|
|
'item': 'cluster health',
|
|
'expected': 'HEALTH_OK',
|
|
'actual': (health_check.stdout | from_json).status,
|
|
'match': (health_check.stdout | from_json).status
|
|
== 'HEALTH_OK'
|
|
}] }}
|
|
when: inventory_hostname in groups['ceph_bootstrap']
|
|
|
|
# Last, to match the previous report order (see the ordering note above).
|
|
- name: Ceph tuning checks
|
|
ansible.builtin.include_role:
|
|
name: ceph_tuning
|
|
tasks_from: drift
|
|
|
|
# --- Report ---
|
|
|
|
- name: Build drift report
|
|
ansible.builtin.set_fact:
|
|
drift_report: |
|
|
=== Drift Report: {{ inventory_hostname }} @ {{ lookup('pipe', 'date -u +%Y-%m-%dT%H:%M:%SZ') }} ===
|
|
{% for r in drift_results %}
|
|
{% if r.match %}
|
|
{{ '%-14s' | format(r.category) }} {{ '%-30s' | format(r.item) }} expected: {{ '%-12s' | format(r.expected) }} actual: {{ '%-12s' | format(r.actual) }} OK
|
|
{% else %}
|
|
{{ '%-14s' | format(r.category) }} {{ '%-30s' | format(r.item) }} expected: {{ '%-12s' | format(r.expected) }} actual: {{ '%-12s' | format(r.actual) }} !! DRIFT
|
|
{% endif %}
|
|
{% endfor %}
|
|
|
|
{% set drifts = drift_results | selectattr('match', 'equalto', false) | list %}
|
|
{% if drifts | length == 0 %}
|
|
No drift detected. Cluster matches expected state.
|
|
{% else %}
|
|
{{ drifts | length }} drift(s) detected. Run 'mise run deploy' to converge.
|
|
{% endif %}
|
|
|
|
- name: Display drift report
|
|
ansible.builtin.debug:
|
|
msg: "{{ drift_report.split('\n') }}"
|