Files
yucca/ansible/ceph/roles/security/defaults/main.yml
T
Andy Molenda 6eca8f094b fix(ceph): converge-safe pulls, live-ruleset gate, and container hardening for monk (#593)
* fix(ceph): converge-safe pulls, live-ruleset gate, caps reconcile, container hardening

* docs(ceph): monk rotation, upgrade, and retirement coverage

* fix(ceph): let the caps task own caps, split cgroups, guard check mode

* fix(ceph): pair Delegate with cgroups=split and probe collect_success
2026-08-31 06:25:54 -07:00

97 lines
4.7 KiB
YAML

---
# Security hardening defaults for Ceph nodes
# --- Firewall (nftables) ---
ceph_firewall_enabled: true
# Trusted source networks allowed to reach Ceph services.
# RFC1918 + Tailscale CGNAT. Loopback is always allowed via iifname "lo".
ceph_firewall_trusted_networks:
- 10.0.0.0/8
- 172.16.0.0/12
- 192.168.0.0/16
- 100.64.0.0/10
# Interfaces accepted WHOLESALE (no port filtering), e.g. the closed
# replication VLAN (spice: bond0.122). Port rules only match daemon dports,
# so return traffic of connections established BEFORE the firewall loaded
# targets ephemeral ports and hits the drop policy (conntrack cannot adopt
# mid-stream flows it did not see open). On a closed cluster L2, filtering
# per-port adds nothing; accepting the interface keeps live replication
# flowing through a firewall (re)activation. Default empty: flat-network
# clusters (sietch) have no dedicated replication interface to trust.
ceph_firewall_trusted_ifaces: []
# Ports to open (derived from Ceph service layout)
ceph_firewall_mon_ports: [3300, 6789]
# Full modern bind range (ms_bind_port_max default 7568; the historic docs
# range 6800-7300 under-covers a restart-churned 15-OSD host).
ceph_firewall_osd_port_range: "6800-7568"
ceph_firewall_rgw_port: "{{ ceph_rgw_port | default(443) }}"
# Allow RGW/S3 from any source (true) or restrict to trusted networks (false).
# Mirrors ceph_firewall_ssh_any_source. Default true so a public S3 endpoint
# keeps working out of the box. Set false in production once the client path is
# confirmed to be inside ceph_firewall_trusted_networks (RFC1918 + NetBird
# 100.64.0.0/10); on spice the fabric (10.40.20.0/23) and NetBird overlay both
# already fall in those ranges, so restricting drops only public-internet reach.
ceph_firewall_rgw_any_source: true
ceph_firewall_dashboard_port: 8443
ceph_firewall_prometheus_port: "{{ ceph_prometheus_port | default(9095) }}"
ceph_firewall_grafana_port: "{{ ceph_grafana_port | default(3000) }}"
ceph_firewall_alertmanager_port: "{{ ceph_alertmanager_port | default(9093) }}"
ceph_firewall_node_exporter_port: 9100
ceph_firewall_mgr_exporter_port: 9283
ceph_firewall_ceph_exporter_port: 9926
# monk, the scrub-backlog exporter (roles/scrub_exporter). Mon hosts only, but
# opened everywhere like fluentbit: where the unit is absent nothing listens.
# roles/scrub_exporter asserts this rule is in the LIVE ruleset before it
# deploys; the roles run in different playbooks, so neither references the
# other beyond that check.
ceph_firewall_scrub_exporter_port: "{{ ceph_scrub_exporter_port | default(9284) }}"
# Fluent Bit's own metrics (roles/fluentbit). Bound to the fabric public address
# so vmagent reaches it the same way as the exporters above. Opening the port
# where the agent is not enabled is inert: nothing listens.
ceph_firewall_fluentbit_port: 2020
# cephadm service discovery (mgr). Prometheus fetches its scrape target list
# from this endpoint on the active mgr via http_sd. If it is not reachable from
# the Prometheus host, Prometheus discovers zero targets and Grafana shows no
# data. The active mgr can be any node, so this must be open on all of them.
ceph_firewall_service_discovery_port: 8765
# --- Optional service gateways (disabled by default) ---
# iSCSI gateway - block storage for non-native clients
ceph_firewall_iscsi_enabled: false
ceph_firewall_iscsi_port: 3260
ceph_firewall_iscsi_api_port: 5000
# NFS-Ganesha - CephFS export for NFS clients
ceph_firewall_nfs_enabled: false
ceph_firewall_nfs_port: 2049
ceph_firewall_nfs_mgmt_port: 12049
# --- Firewall drop logging (NFLOG -> ulogd2) ---
# The drop rule logs over NFLOG, not printk, so the journal and the kmsg ring
# buffer never see scanner noise. ulogd2 listens on the group and writes one
# JSON event per drop; roles/fluentbit tails that file to o11y.
ceph_firewall_droplog_group: 0
# Headers only: 64 bytes covers IP + TCP/UDP, which is all ulogd2 decodes.
ceph_firewall_droplog_snaplen: 64
# Packets per netlink datagram. The kernel still flushes partial batches on
# its ~1s nfnetlink_log timer, so batching costs at most ~1s of latency.
ceph_firewall_droplog_threshold: 64
# roles/fluentbit's ceph_fluentbit_droplog_file default must match this path;
# the roles run in different playbooks, so neither can reference the other.
ceph_firewall_droplog_file: /var/log/ulog/nft-drop.json
# Allow SSH from any source (true) or restrict to trusted networks (false).
# Default true for dev/bootstrapping. Set false in production once
# management networks are confirmed.
ceph_firewall_ssh_any_source: true
# --- SSH hardening ---
ceph_ssh_max_auth_tries: 3
ceph_ssh_permit_root_login: prohibit-password
ceph_ssh_allowed_users: "ansible-iac ops root"
# root is needed for cephadm orchestrator SSH between nodes