From a0dced3d1799f4eadee642fd9c63569d697d168b Mon Sep 17 00:00:00 2001 From: Zack Pollard Date: Tue, 21 Jul 2026 13:17:15 +0200 Subject: [PATCH] fix: batch replay queries to fit vmsingle memory budget (#1838) --- .../apps/monitoring/victoria-metrics/replay/job.yaml | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/kubernetes/apps/monitoring/victoria-metrics/replay/job.yaml b/kubernetes/apps/monitoring/victoria-metrics/replay/job.yaml index abcc6bc5..5721b0c5 100644 --- a/kubernetes/apps/monitoring/victoria-metrics/replay/job.yaml +++ b/kubernetes/apps/monitoring/victoria-metrics/replay/job.yaml @@ -39,6 +39,15 @@ spec: # retention anyway, so a stale value only yields empty early points. - -replay.timeFrom=2026-06-20T00:00:00Z - -replay.disableProgressBar=true + # The default (1000) batches 1000 evaluation points per query_range + # request; against ~475k raw series that asks vmsingle for ~8GiB in + # one query and gets a 422, aborting the replay. 100 keeps the + # worst request well inside the ~5GiB concurrent-query budget at + # the cost of a longer (multi-hour) replay. + - -replay.maxDatapointsPerQuery=100 + # Survive transient 422s from concurrent dashboard/rule load; a + # rule that exhausts its retries aborts the whole replay. + - -replay.ruleRetryAttempts=10 volumeMounts: - name: rules mountPath: /rules @@ -52,6 +61,8 @@ spec: # Keep in sync with replay-base. - -replay.timeFrom=2026-06-20T00:00:00Z - -replay.disableProgressBar=true + - -replay.maxDatapointsPerQuery=100 + - -replay.ruleRetryAttempts=10 volumeMounts: - name: rules mountPath: /rules