The restic pipeline (loki/grafana/k8s-resources) is proven end-to-end after the DEV-488 restore drill, so the legacy rsync/tar pipeline into the 100 Gi local-path `backup-storage` PVC is removed. - Delete `apps/monitoring/backup-volumes-cronjob.yaml`. - Remove the DEV-483 bridge `nodeSelector: k3s-worker-2` from `apps/monitoring/loki-deployment.yaml`. Loki's data protection now runs via `backup-loki-restic`, which follows the pod via podAffinity regardless of which node the RWO CSI volume attaches on. The `Recreate` rollout strategy stays — it is unrelated (avoids the attach-deadlock during a rollout). Resolves the RWO/nodeSelector attach race that was blocking DEV-478 weekly OS updates. - Update `apps/monitoring/README.md` to drop the `backup-volumes` section, link the restic restore runbook, and record the pin removal. - Clean stale coexistence comments in the restic/prometheus CronJob manifests now that the legacy job is gone. Cluster-side (already applied out-of-band, since these manifests are `kubectl apply`-based, not Argo-managed): - `kubectl -n monitoring delete cronjob backup-volumes` -> NotFound. - `kubectl -n monitoring delete pvc backup-storage` -> gone; local-path PV `pvc-d0db0ba9-8f89-4f66-9e65-d573ebe1085a` reclaimed automatically (Delete policy). Two stale pre-DEV-487 `backup-k8s-resources` job pods that still referenced the PVC were deleted to release the `pvc-protection` finalizer. - `kubectl -n monitoring apply -f loki-deployment.yaml` -> Recreate rollout, new pod Ready in ~60s, no nodeSelector on the new spec. - No `VolumeAttachment` for the retired PV. - Restic CronJobs (`backup-loki-restic`, `backup-grafana-restic`, `backup-k8s-resources`, `prometheus-backup`) intact. Pre-delete snapshots retained on k3s-cp-1 under `/root/dev489-snapshots-20260816T155411Z/` for post-mortem. Co-Authored-By: Paperclip <noreply@paperclip.ing>
188 lines
6.9 KiB
YAML
188 lines
6.9 KiB
YAML
---
|
|
# Kubernetes-resource backup via restic to Hetzner Object Storage
|
|
# (DEV-487, DEV-482 Option 4). Step 4 of the Option 4 rollout.
|
|
#
|
|
# Streams a concatenated YAML dump of cluster-scoped and per-namespace
|
|
# resources through `restic backup --stdin` into
|
|
# `s3:${S3_ENDPOINT}/${S3_BUCKET}/restic/k8s-resources`. No PVC mount
|
|
# (drops the local-path `backup-storage` dependency), no node pin.
|
|
#
|
|
# Two-container pattern:
|
|
# 1. `kubectl-dump` init container (`alpine/k8s:1.29.4`) writes
|
|
# /dump/cluster.yaml into an emptyDir. Uses `serviceAccountName:
|
|
# backup-sa` (unchanged from the legacy job).
|
|
# 2. `restic` main container (`restic/restic:0.17.3`, matches the
|
|
# loki/grafana siblings) reads that file on stdin and streams it
|
|
# into the restic repo with `--stdin-filename cluster.yaml`.
|
|
apiVersion: batch/v1
|
|
kind: CronJob
|
|
metadata:
|
|
name: backup-k8s-resources
|
|
namespace: monitoring
|
|
labels:
|
|
app: backup
|
|
type: k8s-resources
|
|
backend: restic
|
|
spec:
|
|
schedule: "0 2 * * *"
|
|
concurrencyPolicy: Forbid
|
|
successfulJobsHistoryLimit: 3
|
|
failedJobsHistoryLimit: 3
|
|
jobTemplate:
|
|
metadata:
|
|
labels:
|
|
app: backup
|
|
type: k8s-resources
|
|
backend: restic
|
|
spec:
|
|
backoffLimit: 2
|
|
activeDeadlineSeconds: 3600
|
|
template:
|
|
metadata:
|
|
labels:
|
|
app: backup
|
|
type: k8s-resources
|
|
backend: restic
|
|
spec:
|
|
restartPolicy: OnFailure
|
|
serviceAccountName: backup-sa
|
|
initContainers:
|
|
- name: kubectl-dump
|
|
image: alpine/k8s:1.29.4
|
|
command:
|
|
- /bin/sh
|
|
- -c
|
|
- |
|
|
set -eu
|
|
echo "=== kubectl-dump started at $(date -u +%FT%TZ) ==="
|
|
DUMP=/dump/cluster.yaml
|
|
: > "${DUMP}"
|
|
|
|
echo "--- namespaces ---"
|
|
kubectl get namespaces -o yaml >> "${DUMP}"
|
|
echo "---" >> "${DUMP}"
|
|
|
|
echo "--- cluster-scoped resources ---"
|
|
kubectl get persistentvolumes,storageclasses,clusterroles,clusterrolebindings \
|
|
-o yaml >> "${DUMP}"
|
|
echo "---" >> "${DUMP}"
|
|
|
|
echo "--- namespaced resources ---"
|
|
for ns in $(kubectl get namespaces -o jsonpath='{.items[*].metadata.name}'); do
|
|
echo " ns=${ns}"
|
|
kubectl get \
|
|
configmaps,secrets,services,deployments,statefulsets,daemonsets,jobs,cronjobs,ingresses,persistentvolumeclaims \
|
|
-n "${ns}" -o yaml >> "${DUMP}" 2>/dev/null || true
|
|
echo "---" >> "${DUMP}"
|
|
done
|
|
|
|
echo "dump size: $(wc -c < ${DUMP}) bytes"
|
|
echo "=== kubectl-dump finished at $(date -u +%FT%TZ) ==="
|
|
resources:
|
|
requests:
|
|
cpu: 50m
|
|
memory: 128Mi
|
|
limits:
|
|
cpu: 500m
|
|
memory: 512Mi
|
|
volumeMounts:
|
|
- name: dump
|
|
mountPath: /dump
|
|
containers:
|
|
- name: restic
|
|
image: restic/restic:0.17.3
|
|
env:
|
|
- name: AWS_ACCESS_KEY_ID
|
|
valueFrom:
|
|
secretKeyRef:
|
|
name: monitoring-s3-backup
|
|
key: access-key
|
|
- name: AWS_SECRET_ACCESS_KEY
|
|
valueFrom:
|
|
secretKeyRef:
|
|
name: monitoring-s3-backup
|
|
key: secret-key
|
|
- name: RESTIC_PASSWORD
|
|
valueFrom:
|
|
secretKeyRef:
|
|
name: monitoring-s3-backup
|
|
key: restic-password
|
|
- name: S3_ENDPOINT
|
|
valueFrom:
|
|
secretKeyRef:
|
|
name: monitoring-s3-backup
|
|
key: endpoint
|
|
- name: S3_BUCKET
|
|
valueFrom:
|
|
secretKeyRef:
|
|
name: monitoring-s3-backup
|
|
key: bucket
|
|
- name: RESTIC_REPOSITORY
|
|
value: "s3:$(S3_ENDPOINT)/$(S3_BUCKET)/restic/k8s-resources"
|
|
command:
|
|
- /bin/sh
|
|
- -c
|
|
- |
|
|
set -eu
|
|
echo "=== backup-k8s-resources-restic started at $(date -u +%FT%TZ) ==="
|
|
echo "Repository: ${RESTIC_REPOSITORY}"
|
|
|
|
# First-run tolerance: init if the repo isn't there yet.
|
|
if restic snapshots >/dev/null 2>&1; then
|
|
echo "Repo exists, skipping init."
|
|
else
|
|
echo "Repo missing, initialising..."
|
|
restic init
|
|
fi
|
|
|
|
echo "--- restic backup --stdin cluster.yaml ---"
|
|
restic backup --stdin \
|
|
--stdin-filename cluster.yaml \
|
|
--tag k8s-resources \
|
|
--host k3s < /dump/cluster.yaml
|
|
|
|
echo "--- restic forget/prune ---"
|
|
restic forget --tag k8s-resources \
|
|
--keep-daily 7 \
|
|
--keep-weekly 4 \
|
|
--keep-monthly 6 \
|
|
--prune
|
|
|
|
echo "--- restic check --read-data-subset=5% ---"
|
|
CHECK_STATUS=0
|
|
restic check --read-data-subset=5% || CHECK_STATUS=$?
|
|
echo "restic check exit: ${CHECK_STATUS}"
|
|
|
|
# Textfile-collector metrics; identical wiring to the
|
|
# loki/grafana siblings. Scrapeable once node-exporter's
|
|
# textfile collector path is enabled — tracked in DEV-482.
|
|
{
|
|
echo "backup_k8s_resources_success $([ ${CHECK_STATUS} -eq 0 ] && echo 1 || echo 0)"
|
|
echo "backup_k8s_resources_timestamp_seconds $(date +%s)"
|
|
echo "backup_k8s_resources_check_status ${CHECK_STATUS}"
|
|
} > /metrics/backup_k8s_resources.prom
|
|
|
|
echo "=== backup-k8s-resources-restic finished at $(date -u +%FT%TZ) ==="
|
|
exit ${CHECK_STATUS}
|
|
volumeMounts:
|
|
- name: dump
|
|
mountPath: /dump
|
|
readOnly: true
|
|
- name: metrics
|
|
mountPath: /metrics
|
|
- name: cache
|
|
mountPath: /root/.cache/restic
|
|
resources:
|
|
requests:
|
|
cpu: 100m
|
|
memory: 128Mi
|
|
limits:
|
|
cpu: 1500m
|
|
memory: 1Gi
|
|
volumes:
|
|
- name: dump
|
|
emptyDir: {}
|
|
- name: metrics
|
|
emptyDir: {}
|
|
- name: cache
|
|
emptyDir: {}
|