mirror of
https://github.com/immich-app/yucca.git
synced 2026-09-30 13:33:00 +08:00
* feat(ceph): log firewall drops over nflog and ship them to o11y * docs(ceph): note the drop-event source in the log-shipping headers
33 lines
1.5 KiB
YAML
33 lines
1.5 KiB
YAML
---
|
|
# Ship each ceph node's logs to o11y's VictoriaLogs over the NetBird overlay.
|
|
# Two sources: journald, where cephadm daemons write via libsystemd
|
|
# (log_to_journald true, log_to_file false) along with the mon cluster channel,
|
|
# and the nftables drop events ulogd2 writes (roles/security).
|
|
#
|
|
# Needs the node enrolled in NetBird and the ceph-to-o11y-gateway policy applied;
|
|
# without the policy there is no route to the gateway VIP.
|
|
#
|
|
# Runs LAST in site.yml, after harden.yml: log shipping is not on the Ceph
|
|
# critical path, so it must never stand between a node and convergence, and
|
|
# running it last means it is exercised against the final nftables ruleset.
|
|
#
|
|
# Canary: mise run logs -- --limit spice-ceph-adelia
|
|
# Fleet: mise run logs
|
|
- name: Host log shipping
|
|
hosts: ceph_nodes
|
|
become: true
|
|
serial: "{{ fluentbit_serial | default('25%') }}"
|
|
# No blast-radius control here, unlike baseline.yml: a node that will not ship
|
|
# logs cannot degrade the cluster. Isolated failures ride along and the rest of
|
|
# the fleet still converges.
|
|
# Ansible evaluates this per batch. At the 25% default, 12 nodes on spice, it
|
|
# rides out 2 failures and halts at 3. One dead node is a node problem. A third
|
|
# of a batch is the gateway, the policy, or the package.
|
|
# A wholly failed batch halts regardless, and failed hosts still fail the run.
|
|
max_fail_percentage: 20
|
|
tasks:
|
|
- name: Ship journald via the fluentbit role
|
|
ansible.builtin.import_role:
|
|
name: fluentbit
|
|
when: ceph_fluentbit_enabled | default(false) | bool
|