--- # Drift detection: compares expected Ansible state against live cluster. # Read-only. No changes. Reports mismatches. # # Each role owns its own checks in roles//tasks/drift.yml and appends to # the drift_results accumulator; this play sequences them and renders the # report. Adding a tunable means touching one role directory, not two files. # # Usage: # scripts/ansible-play.sh drift.yml # mise run drift - name: Drift detection hosts: ceph_nodes become: true gather_facts: false vars: drift_results: [] tasks: # --- Per-role checks --- # # include_role, NOT vars_files. vars_files loads a role's defaults at # precedence 14, above inventory group_vars (4) and host_vars (9), so every # check compared the live host against the role default and a cluster that # legitimately overrode a value would be reported as drifting. include_role # loads defaults at precedence 2, where normal layering applies. # # Order is deliberate: it reproduces the report sequence this play emitted # when the checks were inline, so the two can be diffed. security before # baseline, and ceph_tuning after the cluster-state block below, for the # same reason. Reorder freely once that comparison is no longer useful. - name: OS tuning checks ansible.builtin.include_role: name: os_tuning tasks_from: drift - name: Hardware tuning checks ansible.builtin.include_role: name: hardware_tuning tasks_from: drift - name: Security checks ansible.builtin.include_role: name: security tasks_from: drift - name: Baseline checks ansible.builtin.include_role: name: baseline tasks_from: drift # --- Cluster state (bootstrap only) --- # # These assert live cluster shape rather than a declared config value, so # they read no role defaults and stay at play level. - name: Check OSD status ansible.builtin.shell: | set -o pipefail ceph osd stat --format json 2>/dev/null | python3 -c " import sys, json d = json.load(sys.stdin) print(d.get('num_osds', 0), d.get('num_up_osds', 0)) " args: executable: /bin/bash register: osd_stat when: inventory_hostname in groups['ceph_bootstrap'] changed_when: false - name: Record OSD drift ansible.builtin.set_fact: drift_results: >- {{ drift_results + [{ 'category': 'ceph', 'item': 'OSDs up/total', 'expected': (osd_stat.stdout.split()[0]) ~ '/' ~ (osd_stat.stdout.split()[0]), 'actual': (osd_stat.stdout.split()[1]) ~ '/' ~ (osd_stat.stdout.split()[0]), 'match': osd_stat.stdout.split()[0] == osd_stat.stdout.split()[1] }] }} when: inventory_hostname in groups['ceph_bootstrap'] - name: Check MON quorum ansible.builtin.shell: | set -o pipefail ceph mon stat --format json 2>/dev/null | python3 -c " import sys, json print(json.load(sys.stdin).get('num_mons', 0)) " args: executable: /bin/bash register: mon_count when: inventory_hostname in groups['ceph_bootstrap'] changed_when: false # Expect the declared quorum, not one MON per node. ceph_mon holds the hosts # whose `roles` include "mon": every node on a small cluster, a pinned subset # on a large one (spice runs 5 across 48). Comparing against ceph_nodes read # as permanent drift on any cluster that pins its quorum. - name: Record MON drift ansible.builtin.set_fact: drift_results: >- {{ drift_results + [{ 'category': 'ceph', 'item': 'MON count', 'expected': groups['ceph_mon'] | length | string, 'actual': mon_count.stdout | trim, 'match': (mon_count.stdout | trim) == (groups['ceph_mon'] | length | string) }] }} when: inventory_hostname in groups['ceph_bootstrap'] - name: Check RGW daemons ansible.builtin.shell: | set -o pipefail ceph orch ls --service-type rgw --format json 2>/dev/null \ | python3 -c " import sys, json svcs = json.load(sys.stdin) print(sum(s.get('status',{}).get('running',0) for s in svcs)) " args: executable: /bin/bash register: rgw_count when: inventory_hostname in groups['ceph_bootstrap'] changed_when: false - name: Record RGW drift ansible.builtin.set_fact: drift_results: >- {{ drift_results + [{ 'category': 'ceph', 'item': 'RGW daemons', 'expected': groups['ceph_nodes'] | length | string, 'actual': rgw_count.stdout | trim, 'match': (rgw_count.stdout | trim) == (groups['ceph_nodes'] | length | string) }] }} when: inventory_hostname in groups['ceph_bootstrap'] - name: Check cluster health ansible.builtin.command: ceph health --format json register: health_check when: inventory_hostname in groups['ceph_bootstrap'] changed_when: false - name: Record health drift ansible.builtin.set_fact: drift_results: >- {{ drift_results + [{ 'category': 'ceph', 'item': 'cluster health', 'expected': 'HEALTH_OK', 'actual': (health_check.stdout | from_json).status, 'match': (health_check.stdout | from_json).status == 'HEALTH_OK' }] }} when: inventory_hostname in groups['ceph_bootstrap'] # Last, to match the previous report order (see the ordering note above). - name: Ceph tuning checks ansible.builtin.include_role: name: ceph_tuning tasks_from: drift # --- Report --- - name: Build drift report ansible.builtin.set_fact: drift_report: | === Drift Report: {{ inventory_hostname }} @ {{ lookup('pipe', 'date -u +%Y-%m-%dT%H:%M:%SZ') }} === {% for r in drift_results %} {% if r.match %} {{ '%-14s' | format(r.category) }} {{ '%-30s' | format(r.item) }} expected: {{ '%-12s' | format(r.expected) }} actual: {{ '%-12s' | format(r.actual) }} OK {% else %} {{ '%-14s' | format(r.category) }} {{ '%-30s' | format(r.item) }} expected: {{ '%-12s' | format(r.expected) }} actual: {{ '%-12s' | format(r.actual) }} !! DRIFT {% endif %} {% endfor %} {% set drifts = drift_results | selectattr('match', 'equalto', false) | list %} {% if drifts | length == 0 %} No drift detected. Cluster matches expected state. {% else %} {{ drifts | length }} drift(s) detected. Run 'mise run deploy' to converge. {% endif %} - name: Display drift report ansible.builtin.debug: msg: "{{ drift_report.split('\n') }}"