Add VPS warm-standby/backup site (Phase 0-1b)

Foundation for a DR/backup path using an always-on VPS as a second
ArgoCD-managed cluster, plus DB/backup standardization work that fell
out of it:

- vps-standby ArgoCD cluster destination + AppProject, MinIO backup
  receiver, VPS bootstrap script (k3s, Netbird, cert-manager)
- Dual-site DNS failover watcher + home-IP DDNS CronJob, Cloudflare
  token moved out of git into Vault+ExternalSecret
- Nextcloud migrated from ad-hoc MariaDB to CNPG + redis-operator
  (matches n8n/Authentik/GitLab's backup-native pattern)
- Authentik's CNPG manifests moved into the actual ArgoCD-synced
  manifests/ path (were present but never wired into the sync path)
- Vault raft-snapshot CronJob, CNPG barmanObjectStore backups
  (Authentik/n8n/Nextcloud), Nextcloud file-PVC restic sync - all
  targeting the new VPS MinIO receiver

See VPS Warm-Standby plan doc for full design rationale.
This commit is contained in:
Scooby Husky
2026-08-17 14:59:26 -05:00
parent 5163403e24
commit 7990f1fa47
25 changed files with 1161 additions and 139 deletions
@@ -0,0 +1,10 @@
[Unit]
Description=VPS dual-site DNS failover check (Homelabv4 vps-standby)
After=network-online.target
Wants=network-online.target
[Service]
Type=oneshot
ExecStart=/usr/local/bin/vps-dns-failover.sh
# Deliberately no dependency on k3s/docker being up - this must keep working
# even if the VPS's own cluster is unhealthy.
+132
View File
@@ -0,0 +1,132 @@
#!/usr/bin/env bash
# vps-dns-failover.sh - Phase 0.5 dual-site DNS failover watcher
#
# Runs ON THE VPS as a systemd timer (see vps-dns-failover.timer/.service in this
# directory) - deliberately plain systemd, not a k8s CronJob, so it works even if
# the VPS's own k3s is unhealthy. It must never depend on anything inside the home
# cluster (that's the thing it's watching) or the VPS's own k3s (that's a separate
# failure domain from this box's basic OS-level networking).
#
# Health-checks home.kube.huskypup.net (kept current by an in-cluster CronJob while
# home is up - see infrastructure/cert-manager/manifests/home-ip-ddns-cronjob.yaml)
# and flips Cloudflare A records for the standby-service hostnames between home's
# public IP and this VPS's own public IP, with a consecutive-check threshold so a
# single blip doesn't cause a flap.
#
# State (current active site + streak counters) persists in $STATE_DIR between
# runs since each systemd timer firing is a fresh process.
#
# Install:
# sudo mkdir -p /etc/vps-dns-failover
# sudo sh -c 'echo "<dns-edit-scoped-cloudflare-token>" > /etc/vps-dns-failover/cloudflare-token'
# sudo chmod 600 /etc/vps-dns-failover/cloudflare-token
# sudo cp vps-dns-failover.sh /usr/local/bin/vps-dns-failover.sh
# sudo chmod +x /usr/local/bin/vps-dns-failover.sh
# sudo cp vps-dns-failover.service vps-dns-failover.timer /etc/systemd/system/
# sudo systemctl daemon-reload
# sudo systemctl enable --now vps-dns-failover.timer
set -euo pipefail
TOKEN_FILE="/etc/vps-dns-failover/cloudflare-token"
STATE_DIR="/var/lib/vps-dns-failover"
ZONE_NAME="kube.huskypup.net"
HOME_CHECK_HOST="home.kube.huskypup.net"
HOME_CHECK_PORT=443
FAILURE_THRESHOLD=3 # consecutive failed checks before flipping to the VPS
SUCCESS_THRESHOLD=3 # consecutive successful checks before flipping back to home
STANDBY_HOSTNAMES=(
vault.kube.huskypup.net
auth.kube.huskypup.net
gitea.kube.huskypup.net
n8n.kube.huskypup.net
nextcloud.kube.huskypup.net
)
mkdir -p "$STATE_DIR"
TOKEN="$(cat "$TOKEN_FILE")"
STATE_FILE="${STATE_DIR}/state" # format: "<active-site> <fail-streak> <success-streak>"
if [ -f "$STATE_FILE" ]; then
read -r ACTIVE FAIL_STREAK SUCCESS_STREAK < "$STATE_FILE"
else
ACTIVE="home"
FAIL_STREAK=0
SUCCESS_STREAK=0
fi
# --- health check ------------------------------------------------------------
HOME_IP="$(dig +short A "$HOME_CHECK_HOST" @1.1.1.1 | tail -n1)"
if [ -n "$HOME_IP" ] && timeout 5 bash -c "cat < /dev/null > /dev/tcp/${HOME_IP}/${HOME_CHECK_PORT}" 2>/dev/null; then
HEALTHY=1
else
HEALTHY=0
fi
if [ "$HEALTHY" = 1 ]; then
FAIL_STREAK=0
SUCCESS_STREAK=$((SUCCESS_STREAK + 1))
else
SUCCESS_STREAK=0
FAIL_STREAK=$((FAIL_STREAK + 1))
fi
echo "$(date -u +%FT%TZ) active=${ACTIVE} healthy=${HEALTHY} fail_streak=${FAIL_STREAK} success_streak=${SUCCESS_STREAK} home_ip=${HOME_IP:-none}"
# --- Cloudflare helpers --------------------------------------------------------
cf_zone_id() {
curl -sf -H "Authorization: Bearer ${TOKEN}" \
"https://api.cloudflare.com/client/v4/zones?name=${ZONE_NAME}" | jq -r '.result[0].id'
}
cf_set_record() {
local zone_id="$1" hostname="$2" target_ip="$3"
local record_json record_id
record_json="$(curl -sf -H "Authorization: Bearer ${TOKEN}" \
"https://api.cloudflare.com/client/v4/zones/${zone_id}/dns_records?name=${hostname}&type=A")"
record_id="$(echo "$record_json" | jq -r '.result[0].id // empty')"
local body="{\"type\":\"A\",\"name\":\"${hostname}\",\"content\":\"${target_ip}\",\"ttl\":60,\"proxied\":false}"
if [ -n "$record_id" ]; then
curl -sf -X PATCH -H "Authorization: Bearer ${TOKEN}" -H "Content-Type: application/json" \
-d "$body" "https://api.cloudflare.com/client/v4/zones/${zone_id}/dns_records/${record_id}" >/dev/null
else
curl -sf -X POST -H "Authorization: Bearer ${TOKEN}" -H "Content-Type: application/json" \
-d "$body" "https://api.cloudflare.com/client/v4/zones/${zone_id}/dns_records" >/dev/null
fi
echo " ${hostname} -> ${target_ip}"
}
flip_to() {
local target="$1"
local target_ip
if [ "$target" = "vps" ]; then
target_ip="$(curl -sf https://cloudflare.com/cdn-cgi/trace | grep -o '^ip=.*' | cut -d= -f2)"
else
target_ip="$HOME_IP"
fi
if [ -z "$target_ip" ]; then
echo "ERROR: could not determine target IP for '${target}', not flipping"
return 1
fi
echo "Flipping standby hostnames to ${target} (${target_ip})..."
local zone_id
zone_id="$(cf_zone_id)"
for h in "${STANDBY_HOSTNAMES[@]}"; do
cf_set_record "$zone_id" "$h" "$target_ip"
done
}
# --- decide ------------------------------------------------------------------
if [ "$ACTIVE" = "home" ] && [ "$FAIL_STREAK" -ge "$FAILURE_THRESHOLD" ]; then
flip_to "vps"
ACTIVE="vps"
FAIL_STREAK=0
SUCCESS_STREAK=0
elif [ "$ACTIVE" = "vps" ] && [ "$SUCCESS_STREAK" -ge "$SUCCESS_THRESHOLD" ]; then
flip_to "home"
ACTIVE="home"
FAIL_STREAK=0
SUCCESS_STREAK=0
fi
echo "${ACTIVE} ${FAIL_STREAK} ${SUCCESS_STREAK}" > "$STATE_FILE"
@@ -0,0 +1,10 @@
[Unit]
Description=Run vps-dns-failover check every 2 minutes
[Timer]
OnBootSec=1min
OnUnitActiveSec=2min
AccuracySec=10s
[Install]
WantedBy=timers.target