- infrastructure/OS_UPDATE_PROCEDURE.md: agent-facing rolling update procedure (drain -> apt -> reboot -> verify -> uncordon -> health -> next). Explicit MUST NOT list around k3s config, PVs, and manifests. - infrastructure/OS_UPDATE_ROUTINE.md: describes the weekly Paperclip routine (Sun 03:00 Europe/Berlin) that fires this procedure. - infrastructure/scripts/os-update/: cluster-health.sh, update-node.sh, os-update.sh, README. Enforces the same guardrails in code: workers-first-then-CP, one node at a time, no --force drains, halts on reboot/kubelet/health failure, never touches k3s config. Co-Authored-By: Paperclip <noreply@paperclip.ing>
130 lines
3.3 KiB
Bash
Executable file
130 lines
3.3 KiB
Bash
Executable file
#!/bin/bash
|
|
# cluster-health.sh — verify k3s cluster is fully healthy.
|
|
# Exits 0 on green, 1 on any failure. Used as preflight and post-node check by os-update.sh.
|
|
#
|
|
# Runs kubectl commands against whatever the current KUBECONFIG resolves to;
|
|
# invoke via `ssh root@$CONTROL_PLANE_HOST bash -s < cluster-health.sh` to
|
|
# check the cluster from an operator machine without local kubeconfig.
|
|
#
|
|
# Environment overrides:
|
|
# RETRY_ON_TRANSIENT=1 — one retry after 30s for non-Ready-but-Running pods
|
|
# VERBOSE=1 — dump full failing rows on non-zero exit
|
|
|
|
set -euo pipefail
|
|
|
|
fail=0
|
|
|
|
echo "=== nodes ==="
|
|
kubectl get nodes -o wide
|
|
|
|
not_ready=$(kubectl get nodes -o json | jq -r '
|
|
.items[]
|
|
| select(.status.conditions[] | select(.type=="Ready" and .status!="True"))
|
|
| .metadata.name
|
|
' || true)
|
|
if [ -n "$not_ready" ]; then
|
|
echo "FAIL: nodes not Ready: $not_ready"
|
|
fail=1
|
|
fi
|
|
|
|
echo
|
|
echo "=== node pressure conditions ==="
|
|
pressure=$(kubectl get nodes -o json | jq -r '
|
|
.items[]
|
|
| . as $n
|
|
| .status.conditions[]
|
|
| select(.type=="DiskPressure" or .type=="MemoryPressure" or .type=="PIDPressure")
|
|
| select(.status=="True")
|
|
| "\($n.metadata.name) \(.type)=True"
|
|
' || true)
|
|
if [ -n "$pressure" ]; then
|
|
echo "FAIL: node pressure: $pressure"
|
|
fail=1
|
|
else
|
|
echo " (none)"
|
|
fi
|
|
|
|
check_bad_pods() {
|
|
kubectl get pods -A --field-selector=status.phase!=Running,status.phase!=Succeeded --no-headers 2>/dev/null || true
|
|
}
|
|
check_notready_pods() {
|
|
kubectl get pods -A -o json | jq -r '
|
|
.items[]
|
|
| select(.status.phase=="Running")
|
|
| select([.status.conditions[]?|select(.type=="Ready")|.status] | contains(["False"]))
|
|
| "\(.metadata.namespace)/\(.metadata.name)"
|
|
' 2>/dev/null || true
|
|
}
|
|
|
|
echo
|
|
echo "=== pods not Running/Succeeded ==="
|
|
bad_pods=$(check_bad_pods)
|
|
if [ -n "$bad_pods" ]; then
|
|
if [ "${RETRY_ON_TRANSIENT:-0}" = "1" ]; then
|
|
echo " transient? re-checking in 30s..."
|
|
sleep 30
|
|
bad_pods=$(check_bad_pods)
|
|
fi
|
|
fi
|
|
if [ -n "$bad_pods" ]; then
|
|
echo "FAIL: pods not Running/Succeeded:"
|
|
echo "$bad_pods"
|
|
fail=1
|
|
else
|
|
echo " (none)"
|
|
fi
|
|
|
|
echo
|
|
echo "=== pods Running but not Ready ==="
|
|
notready=$(check_notready_pods)
|
|
if [ -n "$notready" ] && [ "${RETRY_ON_TRANSIENT:-0}" = "1" ]; then
|
|
echo " transient? re-checking in 30s..."
|
|
sleep 30
|
|
notready=$(check_notready_pods)
|
|
fi
|
|
if [ -n "$notready" ]; then
|
|
echo "FAIL: pods Running-but-not-Ready:"
|
|
echo "$notready"
|
|
fail=1
|
|
else
|
|
echo " (none)"
|
|
fi
|
|
|
|
echo
|
|
echo "=== deployments below desired replicas ==="
|
|
deploy_bad=$(kubectl get deploy -A -o json | jq -r '
|
|
.items[]
|
|
| select((.status.readyReplicas // 0) < (.spec.replicas // 1))
|
|
| "\(.metadata.namespace)/\(.metadata.name) \(.status.readyReplicas // 0)/\(.spec.replicas)"
|
|
' || true)
|
|
if [ -n "$deploy_bad" ]; then
|
|
echo "FAIL: deployments not fully ready:"
|
|
echo "$deploy_bad"
|
|
fail=1
|
|
else
|
|
echo " (none)"
|
|
fi
|
|
|
|
echo
|
|
echo "=== statefulsets below desired replicas ==="
|
|
sts_bad=$(kubectl get sts -A -o json | jq -r '
|
|
.items[]
|
|
| select((.status.readyReplicas // 0) < (.spec.replicas // 1))
|
|
| "\(.metadata.namespace)/\(.metadata.name) \(.status.readyReplicas // 0)/\(.spec.replicas)"
|
|
' || true)
|
|
if [ -n "$sts_bad" ]; then
|
|
echo "FAIL: statefulsets not fully ready:"
|
|
echo "$sts_bad"
|
|
fail=1
|
|
else
|
|
echo " (none)"
|
|
fi
|
|
|
|
if [ "$fail" -ne 0 ]; then
|
|
echo
|
|
echo "CLUSTER-HEALTH: FAIL"
|
|
exit 1
|
|
fi
|
|
|
|
echo
|
|
echo "CLUSTER-HEALTH: OK"
|