feat(monitoring): backup-grafana-restic CronJob → Hetzner S3 (DEV-486)
Add a daily 03:15 UTC restic backup of the grafana-storage PVC to
s3:${S3_ENDPOINT}/${S3_BUCKET}/restic/grafana. Pinned to
k3s-worker-2 via nodeSelector because grafana-storage is a
local-path PV anchored there — no podAffinity needed. Same restic
retention as the loki sibling (7d/4w/6m + prune + 5% check) with
metrics backup_grafana_{success,timestamp_seconds,check_status}
written to the emptyDir textfile path. Deployed in parallel with
the legacy backup-volumes CronJob (Option 4 rollout, DEV-482).
Co-Authored-By: Paperclip <noreply@paperclip.ing>
This commit is contained in:
parent
e2fd90a22b
commit
d495a5936f
2 changed files with 146 additions and 0 deletions
|
|
@ -5,6 +5,7 @@ Manifests recording the cluster-side monitoring backup CronJobs that were previo
|
|||
- `backup-k8s-resources-cronjob.yaml` — daily dump of Kubernetes resources into `backup-storage` PVC.
|
||||
- `backup-volumes-cronjob.yaml` — daily rsync/tar of Grafana + Loki PVCs into `backup-storage`. Prometheus data is NOT included here — it lives on a different node (see below). Being retired by the restic pipeline in [DEV-482](/DEV/issues/DEV-482) — keep running until step 6 (restore drill passed).
|
||||
- `backup-loki-restic-cronjob.yaml` — daily restic backup of `loki-storage-encrypted` to `hetzner-s3:${BUCKET}/restic/loki`. Co-schedules with the Loki pod via `podAffinity` (RWO permits additional read-only mounts on the same node). Deployed in parallel with `backup-volumes` per the [DEV-482](/DEV/issues/DEV-482) Option 4 rollout ([DEV-485](/DEV/issues/DEV-485)).
|
||||
- `backup-grafana-restic-cronjob.yaml` — daily restic backup of `grafana-storage` to `hetzner-s3:${BUCKET}/restic/grafana`. Pinned to `k3s-worker-2` via `nodeSelector` (the local-path PV anchors the grafana pod there already, no `podAffinity` needed). Schedule `15 3 * * *` — offset from the loki run at `03:00`. Deployed in parallel with `backup-volumes` per the [DEV-482](/DEV/issues/DEV-482) Option 4 rollout ([DEV-486](/DEV/issues/DEV-486)).
|
||||
- `prometheus-backup-cronjob.yaml` + `prometheus-backup-sealed.yaml` — dedicated Prometheus data backup that streams `prometheus-data-encrypted` to Hetzner S3 via rclone. Co-schedules with the Prometheus pod via `podAffinity` so the RWO PVC attaches on the same node ([DEV-465](/DEV/issues/DEV-465)).
|
||||
|
||||
The `backup-storage` PVC (100Gi, local-path, bound to k3s-worker-2) is the shared destination for `backup-k8s-resources` and `backup-volumes`.
|
||||
|
|
|
|||
145
apps/monitoring/backup-grafana-restic-cronjob.yaml
Normal file
145
apps/monitoring/backup-grafana-restic-cronjob.yaml
Normal file
|
|
@ -0,0 +1,145 @@
|
|||
---
|
||||
# Grafana data backup via restic to Hetzner Object Storage (DEV-486,
|
||||
# DEV-482 Option 4). Step 3 of the Option 4 rollout.
|
||||
#
|
||||
# Streams the RWO PVC `grafana-storage` (mounted read-only) into
|
||||
# `s3:${S3_ENDPOINT}/${S3_BUCKET}/restic/grafana`, a client-side
|
||||
# encrypted restic repository. Deployed in parallel with the legacy
|
||||
# `backup-volumes` CronJob — do not retire that job until DEV-482
|
||||
# step 6 (restore drill passed).
|
||||
#
|
||||
# `grafana-storage` is a local-path PV anchored on k3s-worker-2, so
|
||||
# the grafana pod is already pinned there; a plain nodeSelector on
|
||||
# the same host is enough (no podAffinity like the loki job needed).
|
||||
apiVersion: batch/v1
|
||||
kind: CronJob
|
||||
metadata:
|
||||
name: backup-grafana-restic
|
||||
namespace: monitoring
|
||||
labels:
|
||||
app: backup
|
||||
type: grafana
|
||||
backend: restic
|
||||
spec:
|
||||
schedule: "15 3 * * *"
|
||||
concurrencyPolicy: Forbid
|
||||
successfulJobsHistoryLimit: 3
|
||||
failedJobsHistoryLimit: 3
|
||||
jobTemplate:
|
||||
metadata:
|
||||
labels:
|
||||
app: backup
|
||||
type: grafana
|
||||
backend: restic
|
||||
spec:
|
||||
backoffLimit: 2
|
||||
activeDeadlineSeconds: 3600
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
app: backup
|
||||
type: grafana
|
||||
backend: restic
|
||||
spec:
|
||||
restartPolicy: OnFailure
|
||||
nodeSelector:
|
||||
kubernetes.io/hostname: k3s-worker-2
|
||||
containers:
|
||||
- name: restic
|
||||
image: restic/restic:0.17.3
|
||||
env:
|
||||
- name: AWS_ACCESS_KEY_ID
|
||||
valueFrom:
|
||||
secretKeyRef:
|
||||
name: monitoring-s3-backup
|
||||
key: access-key
|
||||
- name: AWS_SECRET_ACCESS_KEY
|
||||
valueFrom:
|
||||
secretKeyRef:
|
||||
name: monitoring-s3-backup
|
||||
key: secret-key
|
||||
- name: RESTIC_PASSWORD
|
||||
valueFrom:
|
||||
secretKeyRef:
|
||||
name: monitoring-s3-backup
|
||||
key: restic-password
|
||||
- name: S3_ENDPOINT
|
||||
valueFrom:
|
||||
secretKeyRef:
|
||||
name: monitoring-s3-backup
|
||||
key: endpoint
|
||||
- name: S3_BUCKET
|
||||
valueFrom:
|
||||
secretKeyRef:
|
||||
name: monitoring-s3-backup
|
||||
key: bucket
|
||||
- name: RESTIC_REPOSITORY
|
||||
value: "s3:$(S3_ENDPOINT)/$(S3_BUCKET)/restic/grafana"
|
||||
command:
|
||||
- /bin/sh
|
||||
- -c
|
||||
- |
|
||||
set -eu
|
||||
echo "=== backup-grafana-restic started at $(date -u +%FT%TZ) ==="
|
||||
echo "Repository: ${RESTIC_REPOSITORY}"
|
||||
|
||||
# First-run tolerance: init if the repo isn't there yet.
|
||||
if restic snapshots >/dev/null 2>&1; then
|
||||
echo "Repo exists, skipping init."
|
||||
else
|
||||
echo "Repo missing, initialising..."
|
||||
restic init
|
||||
fi
|
||||
|
||||
echo "--- restic backup /source ---"
|
||||
restic backup /source \
|
||||
--tag grafana \
|
||||
--host k3s \
|
||||
--exclude '*.tmp'
|
||||
|
||||
echo "--- restic forget/prune ---"
|
||||
restic forget --tag grafana \
|
||||
--keep-daily 7 \
|
||||
--keep-weekly 4 \
|
||||
--keep-monthly 6 \
|
||||
--prune
|
||||
|
||||
echo "--- restic check --read-data-subset=5% ---"
|
||||
CHECK_STATUS=0
|
||||
restic check --read-data-subset=5% || CHECK_STATUS=$?
|
||||
echo "restic check exit: ${CHECK_STATUS}"
|
||||
|
||||
# Textfile-collector metrics; identical wiring to the loki
|
||||
# sibling. Scrapeable once node-exporter's textfile
|
||||
# collector path is enabled — tracked in DEV-482.
|
||||
{
|
||||
echo "backup_grafana_success $([ ${CHECK_STATUS} -eq 0 ] && echo 1 || echo 0)"
|
||||
echo "backup_grafana_timestamp_seconds $(date +%s)"
|
||||
echo "backup_grafana_check_status ${CHECK_STATUS}"
|
||||
} > /metrics/backup_grafana.prom
|
||||
|
||||
echo "=== backup-grafana-restic finished at $(date -u +%FT%TZ) ==="
|
||||
exit ${CHECK_STATUS}
|
||||
volumeMounts:
|
||||
- name: grafana-data
|
||||
mountPath: /source
|
||||
readOnly: true
|
||||
- name: metrics
|
||||
mountPath: /metrics
|
||||
- name: cache
|
||||
mountPath: /root/.cache/restic
|
||||
resources:
|
||||
requests:
|
||||
cpu: 100m
|
||||
memory: 128Mi
|
||||
limits:
|
||||
cpu: 1500m
|
||||
memory: 1Gi
|
||||
volumes:
|
||||
- name: grafana-data
|
||||
persistentVolumeClaim:
|
||||
claimName: grafana-storage
|
||||
- name: metrics
|
||||
emptyDir: {}
|
||||
- name: cache
|
||||
emptyDir: {}
|
||||
Loading…
Add table
Reference in a new issue