Files
devtools/kubernetes/apps/monitoring/victoria-metrics/replay/job.yaml
T

80 lines
3.3 KiB
YAML

---
# One-off vmalert replay Job to backfill recording rule history, applied once
# by flux (ks.yaml: vmetrics-replay). Once it has completed and the rule series
# validate, remove this directory and the vmetrics-replay Kustomization in a
# follow-up PR — prune will delete the Job and ConfigMap.
#
# Ordering is required and enforced by the pod structure: the base group is
# replayed first, then the derived groups (which read the base rule series),
# then the vmsingle rollup result cache is reset so queries immediately see
# the backfilled samples.
#
# Jobs are immutable; the force annotation makes flux delete+recreate (and
# therefore re-run the replay) if this spec is ever changed. To re-run with a
# different window, edit -replay.timeFrom and let flux recreate the Job.
apiVersion: batch/v1
kind: Job
metadata:
name: version-worker-replay
namespace: monitoring
annotations:
kustomize.toolkit.fluxcd.io/force: "true"
spec:
backoffLimit: 0
template:
metadata:
labels:
app.kubernetes.io/name: version-worker-replay
spec:
restartPolicy: Never
initContainers:
- name: replay-base
# Pinned to chart 0.86.0's vmalert version (Chart.yaml appVersion v1.147.0).
image: victoriametrics/vmalert:v1.147.0
args:
- -rule=/rules/base.yaml
- -datasource.url=http://vmsingle-vmetrics:8428
- -remoteWrite.url=http://vmsingle-vmetrics:8428
# ~31d before the expected merge date; earlier raw data is outside
# retention anyway, so a stale value only yields empty early points.
- -replay.timeFrom=2026-06-20T00:00:00Z
- -replay.disableProgressBar=true
# The default (1000) batches 1000 evaluation points per query_range
# request; against ~475k raw series that asks vmsingle for ~8GiB in
# one query and gets a 422, aborting the replay. 100 keeps the
# worst request well inside the ~5GiB concurrent-query budget at
# the cost of a longer (multi-hour) replay.
- -replay.maxDatapointsPerQuery=100
# Survive transient 422s from concurrent dashboard/rule load; a
# rule that exhausts its retries aborts the whole replay.
- -replay.ruleRetryAttempts=10
volumeMounts:
- name: rules
mountPath: /rules
readOnly: true
- name: replay-derived
image: victoriametrics/vmalert:v1.147.0
args:
- -rule=/rules/derived.yaml
- -datasource.url=http://vmsingle-vmetrics:8428
- -remoteWrite.url=http://vmsingle-vmetrics:8428
# Keep in sync with replay-base.
- -replay.timeFrom=2026-06-20T00:00:00Z
- -replay.disableProgressBar=true
- -replay.maxDatapointsPerQuery=100
- -replay.ruleRetryAttempts=10
volumeMounts:
- name: rules
mountPath: /rules
readOnly: true
containers:
- name: reset-rollup-cache
image: curlimages/curl:8.11.1
args:
- -sf
- http://vmsingle-vmetrics:8428/internal/resetRollupResultCache
volumes:
- name: rules
configMap:
name: version-worker-replay-rules