Files
Homelabv4/infrastructure/vault/manifests/raft-snapshot-cronjob.yaml
T
Scooby HuskyandClaude Sonnet 5 07a59d3b34 Fix vault-raft-snapshot: detect active leader, suspend until VPS egress exists
vault operator raft snapshot save fails against a standby Vault node with
'incomplete snapshot, unable to read SHA256SUMS.sealed file' - it must run
directly against the active HA leader. The job was hardcoded to vault-0,
which has been a standby since boot, so it had been failing on every run.

Fixed to detect the actual leader at runtime via vault status is_self.
Also suspended the CronJob: the upload step targets
vps-minio.netbird.internal, unreachable from any pod in this cluster (same
missing egress-route gap as tonight's CrowdSec/CNPG-backup fixes). The
leader-detection fix is kept since it's correct and independent - just
unsuspend once real egress routing exists.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
2026-08-17 19:56:12 -05:00

154 lines
5.7 KiB
YAML

---
# Periodic Vault raft snapshot, shipped to the VPS MinIO backup receiver.
# This is the DR path for Vault's data independent of the unseal-key custody
# story - a Ceph/cluster-loss disaster is recovered by standing up a fresh
# Vault and `vault operator raft snapshot restore`ing the latest one of these,
# not by anything to do with the unseal key itself.
#
# Requires a one-time manual step after VPS bootstrap: store the VPS MinIO
# root credentials (see infrastructure/vps-standby/minio/values.yaml) into
# this cluster's Vault so ESO can hand them to the CronJob:
# vault kv put secret/vps-minio-credentials \
# access-key=<minio-root-user> secret-key=<minio-root-password>
apiVersion: external-secrets.io/v1
kind: ExternalSecret
metadata:
name: vps-minio-credentials
namespace: vault
spec:
refreshInterval: 1h
secretStoreRef:
kind: ClusterSecretStore
name: vault-backend
target:
name: vps-minio-credentials
creationPolicy: Owner
data:
- secretKey: access-key
remoteRef:
key: vps-minio-credentials
property: access-key
- secretKey: secret-key
remoteRef:
key: vps-minio-credentials
property: secret-key
---
apiVersion: v1
kind: ConfigMap
metadata:
name: vault-raft-snapshot-script
namespace: vault
data:
snapshot.sh: |
#!/bin/bash
set -euo pipefail
# VPS's Netbird address - replace with the actual peer IP/hostname once
# the VPS is bootstrapped and joined to the mesh (scripts/vps-bootstrap.sh).
VPS_MINIO_ENDPOINT="${VPS_MINIO_ENDPOINT:-vps-minio.netbird.internal:30900}"
BUCKET="vault-raft-snapshots"
SNAP_NAME="vault-raft-$(date -u +%Y%m%dT%H%M%SZ).snap"
ROOT_TOKEN="$(kubectl -n vault get secret vault-init-keys -o jsonpath='{.data.VAULT_ROOT_TOKEN}' | base64 -d)"
# `vault operator raft snapshot save` fails against a standby node with
# "incomplete snapshot, unable to read SHA256SUMS.sealed file" - it must
# run directly against the active HA leader, not a proxied standby.
# Found 2026-08-17: this job had been failing every run since it was
# added, hardcoded to vault-0 (frequently a standby). Detect the actual
# leader at runtime instead of hardcoding a pod name.
echo "==> Finding the active Vault leader..."
LEADER_POD=""
for pod in vault-0 vault-1 vault-2; do
is_leader="$(kubectl -n vault exec "$pod" -- env VAULT_ADDR=http://127.0.0.1:8200 VAULT_TOKEN="$ROOT_TOKEN" \
vault status -format=json 2>/dev/null | jq -r '.is_self // false')"
if [ "$is_leader" = "true" ]; then
LEADER_POD="$pod"
break
fi
done
if [ -z "$LEADER_POD" ]; then
echo "ERROR: could not find an active Vault leader among vault-0/1/2" >&2
exit 1
fi
echo "==> Taking raft snapshot from ${LEADER_POD} (active leader)..."
kubectl -n vault exec "$LEADER_POD" -- env VAULT_TOKEN="$ROOT_TOKEN" \
vault operator raft snapshot save "/tmp/${SNAP_NAME}"
echo "==> Copying snapshot out of ${LEADER_POD}..."
kubectl -n vault cp "${LEADER_POD}:/tmp/${SNAP_NAME}" "/tmp/${SNAP_NAME}"
kubectl -n vault exec "$LEADER_POD" -- rm -f "/tmp/${SNAP_NAME}"
echo "==> Installing mc (MinIO client)..."
curl -sf https://dl.min.io/client/mc/release/linux-amd64/mc -o /tmp/mc
chmod +x /tmp/mc
export MC_CONFIG_DIR=/tmp/.mc
/tmp/mc alias set vps-minio "http://${VPS_MINIO_ENDPOINT}" \
"${MINIO_ACCESS_KEY}" "${MINIO_SECRET_KEY}" >/dev/null
echo "==> Uploading ${SNAP_NAME} to vps-minio/${BUCKET}..."
/tmp/mc cp "/tmp/${SNAP_NAME}" "vps-minio/${BUCKET}/${SNAP_NAME}"
rm -f "/tmp/${SNAP_NAME}"
echo "==> Pruning snapshots older than 30 days..."
/tmp/mc find "vps-minio/${BUCKET}" --older-than 30d --exec "/tmp/mc rm {}" || true
echo "==> Done: ${SNAP_NAME}"
---
apiVersion: batch/v1
kind: CronJob
metadata:
name: vault-raft-snapshot
namespace: vault
spec:
schedule: "0 */6 * * *" # every 6 hours
# Suspended 2026-08-17: the upload step targets vps-minio.netbird.internal,
# which no pod in this cluster can currently reach - no node has an
# egress route into the Netbird mesh (see commit history around this
# date for the full investigation). The leader-detection fix above is
# real and independent of this, so it's kept fixed rather than reverted.
# Unsuspend once real pod-egress routing to the VPS exists.
suspend: true
concurrencyPolicy: Forbid
successfulJobsHistoryLimit: 3
failedJobsHistoryLimit: 3
jobTemplate:
spec:
backoffLimit: 2
template:
spec:
serviceAccountName: argocd-hook-sa # already has kubectl exec rights in this namespace (see vault-init-job.yaml)
restartPolicy: Never
securityContext:
runAsNonRoot: true
runAsUser: 65534
seccompProfile:
type: RuntimeDefault
containers:
- name: raft-snapshot
image: alpine/k8s:1.32.13
command: ["/bin/bash", "/scripts/snapshot.sh"]
securityContext:
allowPrivilegeEscalation: false
capabilities:
drop: ["ALL"]
env:
- name: MINIO_ACCESS_KEY
valueFrom:
secretKeyRef:
name: vps-minio-credentials
key: access-key
- name: MINIO_SECRET_KEY
valueFrom:
secretKeyRef:
name: vps-minio-credentials
key: secret-key
volumeMounts:
- name: scripts
mountPath: /scripts
volumes:
- name: scripts
configMap:
name: vault-raft-snapshot-script
defaultMode: 0755