Files

196 lines
6.9 KiB
YAML

---
# Drift detection: compares expected Ansible state against live cluster.
# Read-only. No changes. Reports mismatches.
#
# Each role owns its own checks in roles/<role>/tasks/drift.yml and appends to
# the drift_results accumulator; this play sequences them and renders the
# report. Adding a tunable means touching one role directory, not two files.
#
# Usage:
# scripts/ansible-play.sh drift.yml
# mise run drift
- name: Drift detection
hosts: ceph_nodes
become: true
gather_facts: false
vars:
drift_results: []
tasks:
# --- Per-role checks ---
#
# include_role, NOT vars_files. vars_files loads a role's defaults at
# precedence 14, above inventory group_vars (4) and host_vars (9), so every
# check compared the live host against the role default and a cluster that
# legitimately overrode a value would be reported as drifting. include_role
# loads defaults at precedence 2, where normal layering applies.
#
# Order is deliberate: it reproduces the report sequence this play emitted
# when the checks were inline, so the two can be diffed. security before
# baseline, and ceph_tuning after the cluster-state block below, for the
# same reason. Reorder freely once that comparison is no longer useful.
- name: OS tuning checks
ansible.builtin.include_role:
name: os_tuning
tasks_from: drift
- name: Hardware tuning checks
ansible.builtin.include_role:
name: hardware_tuning
tasks_from: drift
- name: Security checks
ansible.builtin.include_role:
name: security
tasks_from: drift
- name: Baseline checks
ansible.builtin.include_role:
name: baseline
tasks_from: drift
# --- Cluster state (bootstrap only) ---
#
# These assert live cluster shape rather than a declared config value, so
# they read no role defaults and stay at play level.
- name: Check OSD status
ansible.builtin.shell: |
set -o pipefail
ceph osd stat --format json 2>/dev/null | python3 -c "
import sys, json
d = json.load(sys.stdin)
print(d.get('num_osds', 0), d.get('num_up_osds', 0))
"
args:
executable: /bin/bash
register: osd_stat
when: inventory_hostname in groups['ceph_bootstrap']
changed_when: false
- name: Record OSD drift
ansible.builtin.set_fact:
drift_results: >-
{{ drift_results + [{
'category': 'ceph',
'item': 'OSDs up/total',
'expected': (osd_stat.stdout.split()[0]) ~ '/'
~ (osd_stat.stdout.split()[0]),
'actual': (osd_stat.stdout.split()[1]) ~ '/'
~ (osd_stat.stdout.split()[0]),
'match': osd_stat.stdout.split()[0]
== osd_stat.stdout.split()[1]
}] }}
when: inventory_hostname in groups['ceph_bootstrap']
- name: Check MON quorum
ansible.builtin.shell: |
set -o pipefail
ceph mon stat --format json 2>/dev/null | python3 -c "
import sys, json
print(json.load(sys.stdin).get('num_mons', 0))
"
args:
executable: /bin/bash
register: mon_count
when: inventory_hostname in groups['ceph_bootstrap']
changed_when: false
# Expect the declared quorum, not one MON per node. ceph_mon holds the hosts
# whose `roles` include "mon": every node on a small cluster, a pinned subset
# on a large one (spice runs 5 across 48). Comparing against ceph_nodes read
# as permanent drift on any cluster that pins its quorum.
- name: Record MON drift
ansible.builtin.set_fact:
drift_results: >-
{{ drift_results + [{
'category': 'ceph',
'item': 'MON count',
'expected': groups['ceph_mon'] | length | string,
'actual': mon_count.stdout | trim,
'match': (mon_count.stdout | trim)
== (groups['ceph_mon'] | length | string)
}] }}
when: inventory_hostname in groups['ceph_bootstrap']
- name: Check RGW daemons
ansible.builtin.shell: |
set -o pipefail
ceph orch ls --service-type rgw --format json 2>/dev/null \
| python3 -c "
import sys, json
svcs = json.load(sys.stdin)
print(sum(s.get('status',{}).get('running',0) for s in svcs))
"
args:
executable: /bin/bash
register: rgw_count
when: inventory_hostname in groups['ceph_bootstrap']
changed_when: false
- name: Record RGW drift
ansible.builtin.set_fact:
drift_results: >-
{{ drift_results + [{
'category': 'ceph',
'item': 'RGW daemons',
'expected': groups['ceph_nodes'] | length | string,
'actual': rgw_count.stdout | trim,
'match': (rgw_count.stdout | trim)
== (groups['ceph_nodes'] | length | string)
}] }}
when: inventory_hostname in groups['ceph_bootstrap']
- name: Check cluster health
ansible.builtin.command: ceph health --format json
register: health_check
when: inventory_hostname in groups['ceph_bootstrap']
changed_when: false
- name: Record health drift
ansible.builtin.set_fact:
drift_results: >-
{{ drift_results + [{
'category': 'ceph',
'item': 'cluster health',
'expected': 'HEALTH_OK',
'actual': (health_check.stdout | from_json).status,
'match': (health_check.stdout | from_json).status
== 'HEALTH_OK'
}] }}
when: inventory_hostname in groups['ceph_bootstrap']
# Last, to match the previous report order (see the ordering note above).
- name: Ceph tuning checks
ansible.builtin.include_role:
name: ceph_tuning
tasks_from: drift
# --- Report ---
- name: Build drift report
ansible.builtin.set_fact:
drift_report: |
=== Drift Report: {{ inventory_hostname }} @ {{ lookup('pipe', 'date -u +%Y-%m-%dT%H:%M:%SZ') }} ===
{% for r in drift_results %}
{% if r.match %}
{{ '%-14s' | format(r.category) }} {{ '%-30s' | format(r.item) }} expected: {{ '%-12s' | format(r.expected) }} actual: {{ '%-12s' | format(r.actual) }} OK
{% else %}
{{ '%-14s' | format(r.category) }} {{ '%-30s' | format(r.item) }} expected: {{ '%-12s' | format(r.expected) }} actual: {{ '%-12s' | format(r.actual) }} !! DRIFT
{% endif %}
{% endfor %}
{% set drifts = drift_results | selectattr('match', 'equalto', false) | list %}
{% if drifts | length == 0 %}
No drift detected. Cluster matches expected state.
{% else %}
{{ drifts | length }} drift(s) detected. Run 'mise run deploy' to converge.
{% endif %}
- name: Display drift report
ansible.builtin.debug:
msg: "{{ drift_report.split('\n') }}"