mirror of
https://github.com/immich-app/yucca.git
synced 2026-09-30 13:33:00 +08:00
fix(ceph): converge-safe pulls, live-ruleset gate, and container hardening for monk (#593)
* fix(ceph): converge-safe pulls, live-ruleset gate, caps reconcile, container hardening * docs(ceph): monk rotation, upgrade, and retirement coverage * fix(ceph): let the caps task own caps, split cgroups, guard check mode * fix(ceph): pair Delegate with cgroups=split and probe collect_success
This commit is contained in:
@@ -194,3 +194,14 @@ terragrunt apply
|
||||
|
||||
This regenerates the password in 1Password; the apply step to the live
|
||||
cluster (dashboard / grafana commands above) is still required.
|
||||
|
||||
## Cluster-minted identities (not in 1Password)
|
||||
|
||||
### monk scrub-exporter key (`client.scrub-exporter`)
|
||||
|
||||
A cephx identity minted on the cluster itself (caps `mon allow r, mgr allow
|
||||
r`), staged onto each mon host by `roles/scrub_exporter`. To rotate: `ceph
|
||||
auth del client.scrub-exporter` on the bootstrap node, then re-run `mise run
|
||||
monk` -- `get-or-create` mints a fresh key and the role restages the keyring
|
||||
and restarts every mon's unit. Read-only: compromise exposes cluster
|
||||
metadata reads, not object data (see `docs/security-model.md`).
|
||||
|
||||
@@ -127,3 +127,15 @@ not merely the latest. `baseline_held_packages` controls which packages are held
|
||||
procedure inline.
|
||||
- `docs/runbooks/replace-host.md`, `docs/runbooks/add-node.md` -- related day-2
|
||||
ops that also drive `ceph` from the bootstrap node.
|
||||
|
||||
## Post-upgrade: monk
|
||||
|
||||
monk (`roles/scrub_exporter`) runs on the mon hosts from a container based
|
||||
on the cluster ceph image. After an upgrade: bump `ARG CEPH_IMAGE` in
|
||||
`packages/monk/Dockerfile` to the new release (the mons' layer dedup depends
|
||||
on it), re-pin `ceph_scrub_exporter_image` once CI publishes the rebuilt
|
||||
tag, and re-run `mise run monk`. Then watch two canaries for format drift in
|
||||
the new release: `ceph_scrub_schedule_pgs{state="other"}` staying 0, and the
|
||||
parse-error alert staying quiet. A flapping monk unit triages like any
|
||||
podman systemd unit: `journalctl -u monk` on the mon; the newest error or
|
||||
warn line is the current failure.
|
||||
|
||||
@@ -254,3 +254,4 @@ routable from the public internet.
|
||||
| `ops` password auth | Password-based SSH is not disabled. | Password is stored in 1Password, rotated via Ansible. Interactive use only -- automation uses key-only `ansible-iac`. |
|
||||
| Operator SSH keys distributed out-of-band | No automated key lifecycle for `ops` user. | Acceptable for non-production. Production should use centralized key management (e.g., Teleport, Vault SSH CA). |
|
||||
| RGW S3 credentials static | No automatic rotation of S3 access/secret keys. | Keys stored in 1Password with access control. `radosgw-admin key create/rm` available for manual rotation. |
|
||||
| Shared `client.scrub-exporter` keyring on every mon | Compromise of one mon host exposes a fleet-wide cluster-METADATA-read credential (pg stats, config; not object data). | Inherent to any scraping identity; caps are `mon allow r, mgr allow r`, verified non-escalatable. Rotation runbook in `rotate-secrets.md`. |
|
||||
|
||||
@@ -13,7 +13,7 @@
|
||||
# include "mon" on any cluster shape (all of sietch, 5 of spice's 48). A
|
||||
# per-host mon var only exists on clusters that pin their quorum.
|
||||
#
|
||||
# Canary: mise run monk -- --limit spice-ceph-adelia
|
||||
# Canary: mise run monk -- --limit <mon-host>
|
||||
# Fleet: mise run monk
|
||||
- name: Scrub-backlog exporter
|
||||
hosts: ceph_mon
|
||||
|
||||
@@ -1,15 +1,26 @@
|
||||
---
|
||||
# Retirement contract: monk serves upstream ceph PR #68925's metric names
|
||||
# with locally chosen MIN-stamp-per-pool semantics; upstream's draft
|
||||
# aggregated the MAX. If a ceph release's mgr exporter starts serving
|
||||
# ceph_pg_last_scrub_stamp, diff the semantics first, then flip this off;
|
||||
# never feed both into the same store.
|
||||
ceph_scrub_exporter_enabled: false
|
||||
|
||||
# Delivered by yucca CI (deploy.yml pushes ghcr.io/immich-app/yucca/monk on
|
||||
# merge to main). No tag exists until the first release after the monk package
|
||||
# lands, so there is no usable default: pin a v<version> tag per cluster and
|
||||
# bump it deliberately, like the cluster ceph image.
|
||||
# merge to main). No usable default: pin an immutable reference per cluster
|
||||
# (a v<version> or 0.0.<n> tag, ideally with @sha256 digest) and bump it
|
||||
# deliberately, like the cluster ceph image.
|
||||
ceph_scrub_exporter_image: ""
|
||||
|
||||
ceph_scrub_exporter_port: 9284
|
||||
ceph_scrub_exporter_refresh: 2m
|
||||
|
||||
# Colocated with mon quorum, and monk's working set scales with pg ls output
|
||||
# (a few MB per thousand PGs plus the ceph CLI); generous ceilings that a
|
||||
# runaway can never blow past.
|
||||
ceph_scrub_exporter_memory_max: 1G
|
||||
ceph_scrub_exporter_cpu_quota: 100%
|
||||
|
||||
# Read-only identity; the keyring lands only on mon-role hosts. The container
|
||||
# runs as the ceph image's uid 167, which must be able to read it.
|
||||
ceph_scrub_exporter_auth_entity: client.scrub-exporter
|
||||
|
||||
@@ -1,11 +1,12 @@
|
||||
---
|
||||
- name: Require an image pin when the exporter is enabled
|
||||
- name: Require an image pin and a bootstrap host
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- ceph_scrub_exporter_image | length > 0
|
||||
- groups['ceph_bootstrap'] | default([]) | length > 0
|
||||
fail_msg: >-
|
||||
ceph_scrub_exporter_enabled is true but ceph_scrub_exporter_image is
|
||||
empty; pin a released ghcr.io/immich-app/yucca/monk tag.
|
||||
scrub_exporter needs ceph_scrub_exporter_image pinned to an immutable
|
||||
ghcr.io/immich-app/yucca/monk tag and a non-empty ceph_bootstrap group.
|
||||
run_once: true
|
||||
|
||||
# Cluster-level reads run once on the bootstrap host (the only host holding
|
||||
@@ -17,13 +18,26 @@
|
||||
- auth
|
||||
- get-or-create
|
||||
- "{{ ceph_scrub_exporter_auth_entity }}"
|
||||
register: scrub_exporter_keyring
|
||||
changed_when: false
|
||||
no_log: true
|
||||
run_once: true
|
||||
delegate_to: "{{ groups['ceph_bootstrap'] | first }}"
|
||||
|
||||
# get-or-create with caps hard-fails (behind no_log) on any mismatch, so the
|
||||
# bare form above only mints the key and this task alone owns the caps.
|
||||
- name: Reconcile the identity caps
|
||||
ansible.builtin.command:
|
||||
argv:
|
||||
- ceph
|
||||
- auth
|
||||
- caps
|
||||
- "{{ ceph_scrub_exporter_auth_entity }}"
|
||||
- mon
|
||||
- allow r
|
||||
- mgr
|
||||
- allow r
|
||||
register: scrub_exporter_keyring
|
||||
changed_when: false
|
||||
no_log: true
|
||||
run_once: true
|
||||
delegate_to: "{{ groups['ceph_bootstrap'] | first }}"
|
||||
|
||||
@@ -34,6 +48,27 @@
|
||||
run_once: true
|
||||
delegate_to: "{{ groups['ceph_bootstrap'] | first }}"
|
||||
|
||||
# The 9284 accept rule ships in roles/security; converging monk onto a fleet
|
||||
# whose LIVE ruleset predates it leaves the exporter unreachable off-host. The
|
||||
# kernel ruleset is checked, not /etc/nftables.conf: a written-but-unloaded
|
||||
# file is exactly the divergence that bites.
|
||||
- name: Read the live nftables ruleset
|
||||
ansible.builtin.command: nft list ruleset
|
||||
register: scrub_exporter_nft
|
||||
changed_when: false
|
||||
check_mode: false
|
||||
when: ceph_firewall_enabled | default(true) | bool
|
||||
|
||||
- name: Require the scrub-exporter firewall rule
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- "('dport ' ~ ceph_scrub_exporter_port) in scrub_exporter_nft.stdout"
|
||||
fail_msg: >-
|
||||
the live nftables ruleset has no rule for port
|
||||
{{ ceph_scrub_exporter_port }}; converge site.yml, or run harden.yml
|
||||
first, so ceph_firewall_scrub_exporter_port is applied.
|
||||
when: ceph_firewall_enabled | default(true) | bool
|
||||
|
||||
- name: Create the config directory
|
||||
ansible.builtin.file:
|
||||
path: "{{ ceph_scrub_exporter_config_dir }}"
|
||||
@@ -65,10 +100,21 @@
|
||||
no_log: true
|
||||
notify: Restart monk
|
||||
|
||||
# Image IDs are compared instead of parsing pull progress: podman prints
|
||||
# "Copying blob" even for cached layers, which read every converge as changed
|
||||
# and bounced the exporter fleet-wide for nothing.
|
||||
- name: Read the present monk image id
|
||||
ansible.builtin.command: >-
|
||||
podman image inspect --format {% raw %}{{.Id}}{% endraw %}
|
||||
{{ ceph_scrub_exporter_image }}
|
||||
register: scrub_exporter_image_pre
|
||||
changed_when: false
|
||||
failed_when: false
|
||||
|
||||
- name: Pull the monk image
|
||||
ansible.builtin.command: podman pull {{ ceph_scrub_exporter_image }}
|
||||
ansible.builtin.command: podman pull -q {{ ceph_scrub_exporter_image }}
|
||||
register: scrub_exporter_pull
|
||||
changed_when: "'Copying blob' in scrub_exporter_pull.stderr"
|
||||
changed_when: scrub_exporter_pull.stdout != scrub_exporter_image_pre.stdout
|
||||
notify: Restart monk
|
||||
|
||||
- name: Install the monk unit
|
||||
@@ -80,9 +126,29 @@
|
||||
mode: "0644"
|
||||
notify: Restart monk
|
||||
|
||||
# Flushing here keeps an aborted later play from stranding a running monk on
|
||||
# old unit arguments until some future converge.
|
||||
- name: Apply any pending monk restart
|
||||
ansible.builtin.meta: flush_handlers
|
||||
|
||||
- name: Enable and start monk
|
||||
ansible.builtin.systemd:
|
||||
name: monk
|
||||
enabled: true
|
||||
state: started
|
||||
daemon_reload: true
|
||||
|
||||
# monk answers 200 while collection fails (async collect starts MarkFailed);
|
||||
# only a collect_success sample proves the keyring, caps, and mon path. The
|
||||
# window covers the 90s first-collect ceph timeout plus a restart.
|
||||
- name: Wait for monk to report a successful collection
|
||||
ansible.builtin.uri:
|
||||
url: "http://{{ ceph_service_ip }}:{{ ceph_scrub_exporter_port }}/metrics"
|
||||
return_content: true
|
||||
register: scrub_exporter_probe
|
||||
retries: 24
|
||||
delay: 5
|
||||
until: >-
|
||||
scrub_exporter_probe.status == 200 and
|
||||
'ceph_scrub_collect_success 1' in scrub_exporter_probe.content
|
||||
changed_when: false
|
||||
|
||||
@@ -6,6 +6,7 @@ After=network-online.target
|
||||
[Service]
|
||||
ExecStartPre=-/usr/bin/podman rm -f monk
|
||||
ExecStart=/usr/bin/podman run --rm --name monk --net host \
|
||||
--cgroups=split --cap-drop=ALL --security-opt=no-new-privileges \
|
||||
-v {{ ceph_scrub_exporter_config_dir }}:/etc/ceph:ro \
|
||||
{{ ceph_scrub_exporter_image }} \
|
||||
-listen {{ ceph_service_ip }}:{{ ceph_scrub_exporter_port }} \
|
||||
@@ -14,6 +15,10 @@ ExecStart=/usr/bin/podman run --rm --name monk --net host \
|
||||
ExecStop=/usr/bin/podman stop monk
|
||||
Restart=always
|
||||
RestartSec=10
|
||||
# --cgroups=split needs the unit to own its subtree; quadlet always pairs these.
|
||||
Delegate=yes
|
||||
MemoryMax={{ ceph_scrub_exporter_memory_max }}
|
||||
CPUQuota={{ ceph_scrub_exporter_cpu_quota }}
|
||||
|
||||
[Install]
|
||||
WantedBy=multi-user.target
|
||||
|
||||
@@ -44,7 +44,10 @@ ceph_firewall_mgr_exporter_port: 9283
|
||||
ceph_firewall_ceph_exporter_port: 9926
|
||||
# monk, the scrub-backlog exporter (roles/scrub_exporter). Mon hosts only, but
|
||||
# opened everywhere like fluentbit: where the unit is absent nothing listens.
|
||||
ceph_firewall_scrub_exporter_port: 9284
|
||||
# roles/scrub_exporter asserts this rule is in the LIVE ruleset before it
|
||||
# deploys; the roles run in different playbooks, so neither references the
|
||||
# other beyond that check.
|
||||
ceph_firewall_scrub_exporter_port: "{{ ceph_scrub_exporter_port | default(9284) }}"
|
||||
# Fluent Bit's own metrics (roles/fluentbit). Bound to the fabric public address
|
||||
# so vmagent reaches it the same way as the exporters above. Opening the port
|
||||
# where the agent is not enabled is inert: nothing listens.
|
||||
|
||||
Reference in New Issue
Block a user