diff --git a/apps/monitoring/README.md b/apps/monitoring/README.md index 5750b07..85efc21 100644 --- a/apps/monitoring/README.md +++ b/apps/monitoring/README.md @@ -5,6 +5,7 @@ Manifests recording the cluster-side monitoring backup CronJobs that were previo - `backup-k8s-resources-cronjob.yaml` — daily dump of Kubernetes resources into `backup-storage` PVC. - `backup-volumes-cronjob.yaml` — daily rsync/tar of Grafana + Loki PVCs into `backup-storage`. Prometheus data is NOT included here — it lives on a different node (see below). Being retired by the restic pipeline in [DEV-482](/DEV/issues/DEV-482) — keep running until step 6 (restore drill passed). - `backup-loki-restic-cronjob.yaml` — daily restic backup of `loki-storage-encrypted` to `hetzner-s3:${BUCKET}/restic/loki`. Co-schedules with the Loki pod via `podAffinity` (RWO permits additional read-only mounts on the same node). Deployed in parallel with `backup-volumes` per the [DEV-482](/DEV/issues/DEV-482) Option 4 rollout ([DEV-485](/DEV/issues/DEV-485)). +- `backup-grafana-restic-cronjob.yaml` — daily restic backup of `grafana-storage` to `hetzner-s3:${BUCKET}/restic/grafana`. Pinned to `k3s-worker-2` via `nodeSelector` (the local-path PV anchors the grafana pod there already, no `podAffinity` needed). Schedule `15 3 * * *` — offset from the loki run at `03:00`. Deployed in parallel with `backup-volumes` per the [DEV-482](/DEV/issues/DEV-482) Option 4 rollout ([DEV-486](/DEV/issues/DEV-486)). - `prometheus-backup-cronjob.yaml` + `prometheus-backup-sealed.yaml` — dedicated Prometheus data backup that streams `prometheus-data-encrypted` to Hetzner S3 via rclone. Co-schedules with the Prometheus pod via `podAffinity` so the RWO PVC attaches on the same node ([DEV-465](/DEV/issues/DEV-465)). The `backup-storage` PVC (100Gi, local-path, bound to k3s-worker-2) is the shared destination for `backup-k8s-resources` and `backup-volumes`. diff --git a/apps/monitoring/backup-grafana-restic-cronjob.yaml b/apps/monitoring/backup-grafana-restic-cronjob.yaml new file mode 100644 index 0000000..2d2bdac --- /dev/null +++ b/apps/monitoring/backup-grafana-restic-cronjob.yaml @@ -0,0 +1,145 @@ +--- +# Grafana data backup via restic to Hetzner Object Storage (DEV-486, +# DEV-482 Option 4). Step 3 of the Option 4 rollout. +# +# Streams the RWO PVC `grafana-storage` (mounted read-only) into +# `s3:${S3_ENDPOINT}/${S3_BUCKET}/restic/grafana`, a client-side +# encrypted restic repository. Deployed in parallel with the legacy +# `backup-volumes` CronJob — do not retire that job until DEV-482 +# step 6 (restore drill passed). +# +# `grafana-storage` is a local-path PV anchored on k3s-worker-2, so +# the grafana pod is already pinned there; a plain nodeSelector on +# the same host is enough (no podAffinity like the loki job needed). +apiVersion: batch/v1 +kind: CronJob +metadata: + name: backup-grafana-restic + namespace: monitoring + labels: + app: backup + type: grafana + backend: restic +spec: + schedule: "15 3 * * *" + concurrencyPolicy: Forbid + successfulJobsHistoryLimit: 3 + failedJobsHistoryLimit: 3 + jobTemplate: + metadata: + labels: + app: backup + type: grafana + backend: restic + spec: + backoffLimit: 2 + activeDeadlineSeconds: 3600 + template: + metadata: + labels: + app: backup + type: grafana + backend: restic + spec: + restartPolicy: OnFailure + nodeSelector: + kubernetes.io/hostname: k3s-worker-2 + containers: + - name: restic + image: restic/restic:0.17.3 + env: + - name: AWS_ACCESS_KEY_ID + valueFrom: + secretKeyRef: + name: monitoring-s3-backup + key: access-key + - name: AWS_SECRET_ACCESS_KEY + valueFrom: + secretKeyRef: + name: monitoring-s3-backup + key: secret-key + - name: RESTIC_PASSWORD + valueFrom: + secretKeyRef: + name: monitoring-s3-backup + key: restic-password + - name: S3_ENDPOINT + valueFrom: + secretKeyRef: + name: monitoring-s3-backup + key: endpoint + - name: S3_BUCKET + valueFrom: + secretKeyRef: + name: monitoring-s3-backup + key: bucket + - name: RESTIC_REPOSITORY + value: "s3:$(S3_ENDPOINT)/$(S3_BUCKET)/restic/grafana" + command: + - /bin/sh + - -c + - | + set -eu + echo "=== backup-grafana-restic started at $(date -u +%FT%TZ) ===" + echo "Repository: ${RESTIC_REPOSITORY}" + + # First-run tolerance: init if the repo isn't there yet. + if restic snapshots >/dev/null 2>&1; then + echo "Repo exists, skipping init." + else + echo "Repo missing, initialising..." + restic init + fi + + echo "--- restic backup /source ---" + restic backup /source \ + --tag grafana \ + --host k3s \ + --exclude '*.tmp' + + echo "--- restic forget/prune ---" + restic forget --tag grafana \ + --keep-daily 7 \ + --keep-weekly 4 \ + --keep-monthly 6 \ + --prune + + echo "--- restic check --read-data-subset=5% ---" + CHECK_STATUS=0 + restic check --read-data-subset=5% || CHECK_STATUS=$? + echo "restic check exit: ${CHECK_STATUS}" + + # Textfile-collector metrics; identical wiring to the loki + # sibling. Scrapeable once node-exporter's textfile + # collector path is enabled — tracked in DEV-482. + { + echo "backup_grafana_success $([ ${CHECK_STATUS} -eq 0 ] && echo 1 || echo 0)" + echo "backup_grafana_timestamp_seconds $(date +%s)" + echo "backup_grafana_check_status ${CHECK_STATUS}" + } > /metrics/backup_grafana.prom + + echo "=== backup-grafana-restic finished at $(date -u +%FT%TZ) ===" + exit ${CHECK_STATUS} + volumeMounts: + - name: grafana-data + mountPath: /source + readOnly: true + - name: metrics + mountPath: /metrics + - name: cache + mountPath: /root/.cache/restic + resources: + requests: + cpu: 100m + memory: 128Mi + limits: + cpu: 1500m + memory: 1Gi + volumes: + - name: grafana-data + persistentVolumeClaim: + claimName: grafana-storage + - name: metrics + emptyDir: {} + - name: cache + emptyDir: {}