restore-k8s-apps.sh's own header explicitly scoped out fresh-cluster
provisioning ("assumes the target cluster/namespaces/PVCs already
exist"). This adds exactly that missing piece as an Ansible playbook
rather than another bash script — a better fit for "provision a target
into a working state" than for the actual data-restore logic, which stays
in restore-k8s-apps.sh unchanged and is just invoked as the final step.
dr-bootstrap.yml, given an app name + a backup archive:
1. Checks the target's environment (disk/mem/cpu) against k3s minimums
before touching anything
2. Installs k3s only if not already present (checked, not assumed)
3. Creates the namespace, PVC(s), Secret, and Deployments/Services/
Ingress from the backup's captured manifests — sanitized first
(sanitize_k8s_manifest.py strips resourceVersion/uid/status/
volumeName, the same class of bug as the earlier Secret-apply fix,
generalized) since a raw `kubectl get -o yaml` dump can't be
reapplied to a different cluster as-is
4. Hands off to the existing, already-validated restore-k8s-apps.sh for
the actual data population
backup-k8s-apps.sh gets a small but necessary addition: PVC specs were
never captured at all before (only Deployment/Service/ConfigMap/Ingress),
so there was nothing for a fresh-cluster bootstrap to provision the PVC
from. Captured with the same label selector as the existing manifests
capture, so it stays in lockstep automatically.
Verified for real, not just syntax-checked: ran the full playbook against
the standby VM (freshly installed k3s, empty namespace) using a live n8n
backup. Namespace/PVCs/Secret/Deployments provisioned from nothing, data
restored, pod healthy, and the workflow count in the restored Postgres DB
(11) matched the real source server exactly.
646 lines
28 KiB
Bash
Executable File
646 lines
28 KiB
Bash
Executable File
#!/bin/bash
|
|
# =============================================
|
|
# backup-k8s-apps.sh — k8s-native replacement for backup-myapps.sh
|
|
# Backs up: n8n, Odoo, Mautic, Nextcloud, Frappe/ERPNext (all live in k3s)
|
|
#
|
|
# Per app, captures:
|
|
# - k8s manifests (Deployment/Service/ConfigMap/Ingress, via `kubectl get -o yaml`)
|
|
# - Secret manifest (plaintext for now — encryption-at-rest explicitly deferred)
|
|
# - DB dump (pg_dump / mysqldump via `kubectl exec`, never a raw PGDATA/datadir copy)
|
|
# - App-data PVC contents (via `kubectl exec ... tar`, never raw filesystem access —
|
|
# see migration-status.md §5, sandbox-filesystem-divergence rule)
|
|
#
|
|
# Storage tiers (logic reused verbatim from backup-myapps.sh, just repointed at the
|
|
# new archive name/contents):
|
|
# 1. Local → /root/backups/ (CRITICAL — always runs)
|
|
# 2. VM → SSH tunnel → /backups/cloudproject/
|
|
# 3. Cloud → Cloudflare R2 (S3-compatible, optional)
|
|
#
|
|
# NOTE: R2 and VM failures do NOT stop the backup. Local backup always completes first.
|
|
#
|
|
# Archive name is deliberately NOT "myapps-backup-*" (the legacy Docker-era pattern
|
|
# still produced nightly by /root/backup-myapps.sh via cron) — using a distinct
|
|
# "myapps-k8s-backup-*" prefix keeps the two backup lineages from mixing in the same
|
|
# retention pool / R2 prefix until the legacy cron job is explicitly retired.
|
|
# =============================================
|
|
|
|
set -uo pipefail
|
|
|
|
# When run inside the management-platform pod, only the static k3s binary is
|
|
# hostPath-mounted (no separate kubectl binary) — fall back to `k3s kubectl`.
|
|
# IMPORTANT: this pod also has a pre-existing rw hostPath mount of /root (for
|
|
# unrelated file-browsing features), which means the host's own root-owned
|
|
# ~/.kube/config (full cluster-admin) is silently visible inside the pod too.
|
|
# `--kubeconfig=/dev/null` is REQUIRED here to stop k3s kubectl from picking
|
|
# that up as a base config — without it, every command below silently runs
|
|
# as cluster-admin instead of the scoped management-platform-sa token,
|
|
# defeating the RBAC in management-platform-backup-rbac.yaml entirely.
|
|
if ! command -v kubectl &>/dev/null && command -v k3s &>/dev/null; then
|
|
kubectl() {
|
|
k3s kubectl \
|
|
--kubeconfig=/dev/null \
|
|
--server=https://kubernetes.default.svc \
|
|
--certificate-authority=/var/run/secrets/kubernetes.io/serviceaccount/ca.crt \
|
|
--token="$(cat /var/run/secrets/kubernetes.io/serviceaccount/token)" \
|
|
"$@"
|
|
}
|
|
fi
|
|
|
|
# --------------------------------------------------
|
|
# Parse --apps
|
|
# --------------------------------------------------
|
|
APPS_FILTER="all" # "all" or comma-separated: frappe,odoo,nextcloud,mautic,n8n
|
|
VALID_APPS="frappe odoo nextcloud mautic n8n"
|
|
|
|
while [[ $# -gt 0 ]]; do
|
|
case "$1" in
|
|
--apps)
|
|
APPS_FILTER="$2"
|
|
shift 2
|
|
;;
|
|
*)
|
|
shift
|
|
;;
|
|
esac
|
|
done
|
|
|
|
if [ "$APPS_FILTER" != "all" ]; then
|
|
APPS_FILTER=$(echo "$APPS_FILTER" | tr -d '[:space:]' | tr '[:upper:]' '[:lower:]')
|
|
IFS=',' read -ra _APP_CHECK <<< "$APPS_FILTER"
|
|
for _a in "${_APP_CHECK[@]}"; do
|
|
if [[ " $VALID_APPS " != *" $_a "* ]]; then
|
|
echo "❌ Unknown app '$_a' in --apps. Valid options: $VALID_APPS"
|
|
exit 1
|
|
fi
|
|
done
|
|
fi
|
|
|
|
app_selected() {
|
|
local app="$1"
|
|
[ "$APPS_FILTER" = "all" ] && return 0
|
|
[[ ",$APPS_FILTER," == *",${app},"* ]]
|
|
}
|
|
|
|
# --------------------------------------------------
|
|
# Per-app metadata table
|
|
# app-key : namespace : app-deployment : db-deployment : db-engine : db-name(or DYNAMIC) : pvc-mount-path : container-name(for exec) : secret-name : db-user-key : db-pass-key
|
|
# --------------------------------------------------
|
|
declare -A NS=( [n8n]=n8n [odoo]=odoo [mautic]=mautic [nextcloud]=nextcloud [frappe]=erpnext )
|
|
declare -A APP_DEPLOY=( [n8n]=n8n [odoo]=odoo [mautic]=mautic [nextcloud]=nextcloud [frappe]=erpnext-web )
|
|
declare -A APP_CTR=( [n8n]=n8n [odoo]=odoo [mautic]=mautic [nextcloud]=nextcloud [frappe]=erpnext-web )
|
|
declare -A DB_DEPLOY=( [n8n]=n8n-postgres [odoo]=odoo-postgres [mautic]=mautic-mariadb [nextcloud]=nextcloud-postgres [frappe]=erpnext-mariadb )
|
|
declare -A DB_ENGINE=( [n8n]=postgres [odoo]=postgres [mautic]=mariadb [nextcloud]=postgres [frappe]=mariadb )
|
|
declare -A DB_NAME=( [n8n]=n8n [odoo]=odoo [mautic]=mautic [nextcloud]=nextcloud [frappe]=DYNAMIC )
|
|
declare -A PVC_PATH=( [n8n]="/home/node/.n8n" [odoo]="/var/lib/odoo" [mautic]="/var/www/html" [nextcloud]="/var/www/html" [frappe]="/home/frappe/frappe-bench/sites" )
|
|
declare -A SECRET_NAME=( [n8n]=n8n-secrets [odoo]=odoo-secrets [mautic]=mautic-secrets [nextcloud]=nextcloud-secrets [frappe]=erpnext-secrets )
|
|
declare -A DB_USER_KEY=( [n8n]=DB_POSTGRESDB_USER [odoo]=POSTGRES_USER [mautic]=MYSQL_USER [nextcloud]=POSTGRES_USER [frappe]="" )
|
|
declare -A DB_PASS_KEY=( [n8n]=DB_POSTGRESDB_PASSWORD [odoo]=POSTGRES_PASSWORD [mautic]=MYSQL_ROOT_PASSWORD [nextcloud]=POSTGRES_PASSWORD [frappe]=MARIADB_ROOT_PASSWORD )
|
|
|
|
# Reproducible framework/vendor code that lives on the same PVC as real user
|
|
# data for these two apps (single shared mount, no separate volume). Both
|
|
# images bake these dirs into their own filesystem layer independent of any
|
|
# volume — Mautic has no logic to reconstruct them, so restore-k8s-apps.sh
|
|
# re-seeds them straight from the app's own image instead of from backup.
|
|
# Never pull request/exclude changes here without updating the matching
|
|
# RECONSTRUCT_DIRS table in restore-k8s-apps.sh.
|
|
declare -A TAR_EXCLUDES=( [mautic]="node_modules vendor" [nextcloud]="apps core dist 3rdparty lib" )
|
|
|
|
ALL_APPS="n8n odoo mautic nextcloud frappe"
|
|
|
|
BACKUP_DATE=$(date +%Y%m%d_%H%M%S)
|
|
BACKUP_NAME="myapps-k8s-backup-${BACKUP_DATE}"
|
|
BACKUP_DIR="/root/backups/${BACKUP_NAME}"
|
|
BACKUP_ARCHIVE="/root/backups/${BACKUP_NAME}.tar.gz"
|
|
|
|
VM_USER="root"
|
|
VM_HOST="178.18.243.51"
|
|
VM_PORT="22"
|
|
VM_KEY="/root/.ssh/id_rsa"
|
|
VM_DEST="/backups/cloudproject/"
|
|
|
|
# ── Load R2 credentials (identical to backup-myapps.sh) ────────────────────────
|
|
CREDENTIALS_FILE="/root/.r2-credentials"
|
|
if [ -f "$CREDENTIALS_FILE" ]; then
|
|
set +u
|
|
source "$CREDENTIALS_FILE"
|
|
set -u
|
|
fi
|
|
|
|
if [ -z "${AWS_ACCESS_KEY_ID:-}" ]; then
|
|
CONFIG_PY="/root/management-platform/config.py"
|
|
if [ -f "$CONFIG_PY" ]; then
|
|
_KEY=$(grep -oP '(?<=R2_ACCESS_KEY_ID",\s{5}")[^"]+' "$CONFIG_PY" 2>/dev/null || true)
|
|
_SEC=$(grep -oP '(?<=R2_SECRET_ACCESS_KEY", ")[^"]+' "$CONFIG_PY" 2>/dev/null || true)
|
|
_BKT=$(grep -oP '(?<=R2_BUCKET_NAME",\s{7}")[^"]+' "$CONFIG_PY" 2>/dev/null || true)
|
|
[ -n "$_KEY" ] && export AWS_ACCESS_KEY_ID="$_KEY"
|
|
[ -n "$_SEC" ] && export AWS_SECRET_ACCESS_KEY="$_SEC"
|
|
[ -n "$_BKT" ] && export R2_BUCKET_NAME="$_BKT"
|
|
fi
|
|
fi
|
|
|
|
R2_ACCOUNT_ID="${R2_ACCOUNT_ID:-35e00c230cc8066252a2d9890b69aea2}"
|
|
R2_BUCKET="${R2_BUCKET_NAME:-navitrends-backups}"
|
|
R2_ENDPOINT="https://${R2_ACCOUNT_ID}.r2.cloudflarestorage.com"
|
|
|
|
BACKUP_LOG_FILE="/root/backups/k8s-backup-status.log"
|
|
MAX_BACKUPS=10
|
|
MAX_R2_BACKUPS=5
|
|
|
|
TIER1_STATUS="pending"
|
|
TIER2_STATUS="pending"
|
|
TIER3_STATUS="pending"
|
|
|
|
log_status() {
|
|
local status="$1" name="$2" msg="${3:-}"
|
|
echo "$(date '+%Y-%m-%d %H:%M:%S') | ${status} | ${name} | ${msg}" >> "$BACKUP_LOG_FILE"
|
|
}
|
|
|
|
# --------------------------------------------------
|
|
# Resource-safety gate (migration-status.md §5): abort before starting
|
|
# any heavy per-app work if we're already below threshold.
|
|
# --------------------------------------------------
|
|
check_resources() {
|
|
local avail_mb avail_gb
|
|
# /proc/meminfo instead of `free` — `free` isn't present in the
|
|
# management-platform pod's minimal image, /proc/meminfo always is.
|
|
avail_mb=$(awk '/^MemAvailable:/{printf "%d", $2/1024}' /proc/meminfo)
|
|
avail_gb=$(df -BG / | awk 'NR==2{gsub("G","",$4); print $4}')
|
|
# Threshold lowered 1536->1024MB 2026-08-21: this host's steady-state
|
|
# available RAM sits ~1.3-1.5GB even when idle (k3s + Jenkins + Docker
|
|
# monitoring/wazuh stack + dev tooling baseline) — investigated, no
|
|
# single runaway process, just genuine multi-service baseline. 1536MB
|
|
# was tripping on ordinary single-app backups. 1024MB still leaves
|
|
# real headroom above what a pg_dump/mysqldump/tar op actually needs.
|
|
if [ "$avail_mb" -lt 1024 ]; then
|
|
echo " ❌ RAM available ${avail_mb}MB < 1024MB threshold — aborting."
|
|
log_status "FAILED" "$BACKUP_NAME" "resource_safety_ram_${avail_mb}mb"
|
|
exit 1
|
|
fi
|
|
if [ "$avail_gb" -lt 10 ]; then
|
|
echo " ❌ Disk available ${avail_gb}GB < 10GB threshold — aborting."
|
|
log_status "FAILED" "$BACKUP_NAME" "resource_safety_disk_${avail_gb}gb"
|
|
exit 1
|
|
fi
|
|
}
|
|
|
|
# --------------------------------------------------
|
|
# Per-backup status metadata sidecar (.meta.json) — powers the UI's
|
|
# green/yellow/red status dot without ever needing to decompress the
|
|
# (possibly 600MB+) archive itself. Written only after all 3 tiers are
|
|
# known, then best-effort mirrored to VM/R2 alongside the archive, same
|
|
# as the .sha256 sidecar already is — so status is readable from
|
|
# whichever tier the UI is listing, not just local.
|
|
# --------------------------------------------------
|
|
write_meta_sidecar() {
|
|
local size_bytes meta_path
|
|
size_bytes=$(stat -c%s "$BACKUP_ARCHIVE" 2>/dev/null || echo 0)
|
|
meta_path="/root/backups/${BACKUP_NAME}.tar.gz.meta.json"
|
|
|
|
python3 - "$BACKUP_NAME" "$size_bytes" "$TIER1_STATUS" "$TIER2_STATUS" "$TIER3_STATUS" \
|
|
"${#INCLUDED_APPS[@]}" "${INCLUDED_APPS[@]}" "${ERRORS[@]}" > "$meta_path" <<'PYEOF'
|
|
import sys, json, datetime
|
|
|
|
backup_name = sys.argv[1]
|
|
size_bytes = int(sys.argv[2])
|
|
tier1, tier2, tier3 = sys.argv[3], sys.argv[4], sys.argv[5]
|
|
n_apps = int(sys.argv[6])
|
|
apps = sys.argv[7:7 + n_apps]
|
|
errors_raw = sys.argv[7 + n_apps:]
|
|
|
|
def norm_tier(s):
|
|
# "ok:572M" -> "ok"; "failed:rc=2" / "skipped:no_credentials" pass through
|
|
# unchanged so the UI can tell a real failure apart from a deliberate skip.
|
|
return 'ok' if s.startswith('ok') else s
|
|
|
|
errors = []
|
|
for e in errors_raw:
|
|
parts = e.split(':', 2)
|
|
errors.append({
|
|
'app': parts[0] if len(parts) > 0 else '',
|
|
'step': parts[1] if len(parts) > 1 else '',
|
|
'detail': parts[2] if len(parts) > 2 else '',
|
|
})
|
|
|
|
meta = {
|
|
'backup_name': backup_name,
|
|
'created_at': datetime.datetime.now(datetime.timezone.utc).isoformat(timespec='seconds'),
|
|
'apps': apps,
|
|
'apps_key': ','.join(sorted(apps)),
|
|
'size_bytes': size_bytes,
|
|
'errors': errors,
|
|
'tiers': {'local': norm_tier(tier1), 'vm': norm_tier(tier2), 'r2': norm_tier(tier3)},
|
|
}
|
|
json.dump(meta, sys.stdout, indent=2)
|
|
PYEOF
|
|
|
|
if [ -s "$meta_path" ]; then
|
|
echo " ✅ Metadata sidecar written: $(basename "$meta_path")"
|
|
else
|
|
echo " ⚠️ Metadata sidecar write failed"
|
|
return 1
|
|
fi
|
|
|
|
# Best-effort mirror to whichever tiers actually succeeded — never fatal,
|
|
# the local sidecar (already written above) is what matters most.
|
|
if [ "$TIER2_STATUS" = "ok" ]; then
|
|
scp -i "$VM_KEY" -P "$VM_PORT" -o StrictHostKeyChecking=no -o ConnectTimeout=15 \
|
|
"$meta_path" "${VM_USER}@${VM_HOST}:${VM_DEST}" &>/dev/null || true
|
|
fi
|
|
if [ "$TIER3_STATUS" = "ok" ] && [ -n "${AWS_ACCESS_KEY_ID:-}" ]; then
|
|
META_SCRIPT="/tmp/r2_meta_$$.py"
|
|
cat > "$META_SCRIPT" << PYEOF2
|
|
import boto3
|
|
client = boto3.client('s3', endpoint_url="${R2_ENDPOINT}",
|
|
aws_access_key_id="${AWS_ACCESS_KEY_ID}", aws_secret_access_key="${AWS_SECRET_ACCESS_KEY}",
|
|
region_name='auto')
|
|
client.upload_file("$meta_path", "${R2_BUCKET}", "backups/${BACKUP_NAME}.tar.gz.meta.json")
|
|
PYEOF2
|
|
"$PYTHON_BIN" "$META_SCRIPT" &>/dev/null || true
|
|
rm -f "$META_SCRIPT"
|
|
fi
|
|
}
|
|
|
|
echo "========================================="
|
|
echo "📦 Starting k8s Backup: $BACKUP_NAME"
|
|
if [ "$APPS_FILTER" = "all" ]; then
|
|
echo " Apps: ALL ($ALL_APPS)"
|
|
else
|
|
echo " Apps: $APPS_FILTER"
|
|
fi
|
|
echo " Tiers: Local → VM → ☁ Cloudflare R2"
|
|
echo "========================================="
|
|
|
|
check_resources
|
|
mkdir -p "$BACKUP_DIR" "/root/backups"
|
|
cd "$BACKUP_DIR"
|
|
|
|
APP_COUNT=0
|
|
INCLUDED_APPS=() # apps actually attempted this run — feeds the .meta.json sidecar's apps_key
|
|
ERRORS=() # "app:step:detail" entries — any entry here makes the sidecar's status red
|
|
for app in $ALL_APPS; do
|
|
app_selected "$app" || continue
|
|
INCLUDED_APPS+=("$app")
|
|
ns="${NS[$app]}"
|
|
app_deploy="${APP_DEPLOY[$app]}"
|
|
app_ctr="${APP_CTR[$app]}"
|
|
db_deploy="${DB_DEPLOY[$app]}"
|
|
db_engine="${DB_ENGINE[$app]}"
|
|
db_name="${DB_NAME[$app]}"
|
|
pvc_path="${PVC_PATH[$app]}"
|
|
secret_name="${SECRET_NAME[$app]}"
|
|
db_user_key="${DB_USER_KEY[$app]}"
|
|
db_pass_key="${DB_PASS_KEY[$app]}"
|
|
|
|
echo ""
|
|
echo "-----------------------------------------"
|
|
echo "📁 [$app] namespace=$ns"
|
|
echo "-----------------------------------------"
|
|
|
|
APP_DIR="$BACKUP_DIR/$app"
|
|
mkdir -p "$APP_DIR"
|
|
|
|
# ---- 1. Manifests ----
|
|
echo -n " 📄 Manifests (deployment/service/configmap/ingress) ... "
|
|
kubectl get deployment,service,configmap,ingress -n "$ns" -l owner=ameni-boukattaya -o yaml \
|
|
> "$APP_DIR/manifests.yaml" 2>"$APP_DIR/manifests.err" \
|
|
&& { echo "✅"; rm -f "$APP_DIR/manifests.err"; } \
|
|
|| { echo "⚠️ FAILED (see manifests.err)"; ERRORS+=("$app:manifests:$(tr '\n' ' ' < "$APP_DIR/manifests.err" | cut -c1-200)"); }
|
|
|
|
# ---- 1b. PVC specs — NOT covered by the manifests capture above, and
|
|
# needed to provision a namespace from scratch on a fresh/DR
|
|
# cluster (restore-k8s-apps.sh assumes the PVC already exists;
|
|
# this is what actually creates it there). Same label selector
|
|
# as the manifests capture, so this stays in lockstep with
|
|
# whatever PVCs are actually labeled as ours — no separate
|
|
# per-app name table to keep in sync. ----
|
|
echo -n " 💾 PVC specs ... "
|
|
kubectl get pvc -n "$ns" -l owner=ameni-boukattaya -o yaml \
|
|
> "$APP_DIR/pvc.yaml" 2>"$APP_DIR/pvc.err" \
|
|
&& { echo "✅"; rm -f "$APP_DIR/pvc.err"; } \
|
|
|| { echo "⚠️ FAILED (see pvc.err)"; ERRORS+=("$app:pvc_spec:$(tr '\n' ' ' < "$APP_DIR/pvc.err" | cut -c1-200)"); }
|
|
|
|
# ---- 2. Secret (plaintext, encryption-at-rest deferred) ----
|
|
echo -n " 🔑 Secret ($secret_name) ... "
|
|
kubectl get secret "$secret_name" -n "$ns" -o yaml \
|
|
> "$APP_DIR/secret.yaml" 2>"$APP_DIR/secret.err" \
|
|
&& { echo "✅"; rm -f "$APP_DIR/secret.err"; } \
|
|
|| { echo "⚠️ FAILED (see secret.err)"; ERRORS+=("$app:secret:$(tr '\n' ' ' < "$APP_DIR/secret.err" | cut -c1-200)"); }
|
|
|
|
# ---- 3. DB dump via kubectl exec (never raw datadir copy) ----
|
|
echo -n " 💾 DB dump ($db_engine) ... "
|
|
case "$db_engine" in
|
|
postgres)
|
|
db_user=$(kubectl get secret "$secret_name" -n "$ns" -o jsonpath="{.data.${db_user_key}}" 2>/dev/null | base64 -d)
|
|
db_pass=$(kubectl get secret "$secret_name" -n "$ns" -o jsonpath="{.data.${db_pass_key}}" 2>/dev/null | base64 -d)
|
|
if [ -n "$db_user" ] && [ -n "$db_pass" ]; then
|
|
kubectl exec -n "$ns" "deploy/${db_deploy}" -- env PGPASSWORD="$db_pass" \
|
|
pg_dump -U "$db_user" -d "$db_name" 2>"$APP_DIR/db-dump.err" \
|
|
| gzip > "$APP_DIR/db-dump.sql.gz"
|
|
pg_dump_rc=${PIPESTATUS[0]}
|
|
if [ "$pg_dump_rc" -eq 0 ] && [ -s "$APP_DIR/db-dump.sql.gz" ]; then
|
|
echo "✅ ($(du -h "$APP_DIR/db-dump.sql.gz" | cut -f1))"
|
|
rm -f "$APP_DIR/db-dump.err"
|
|
else
|
|
echo "⚠️ FAILED (rc=$pg_dump_rc, see db-dump.err)"
|
|
ERRORS+=("$app:db_dump:rc=$pg_dump_rc $(tr '\n' ' ' < "$APP_DIR/db-dump.err" 2>/dev/null | cut -c1-160)")
|
|
fi
|
|
else
|
|
echo "⚠️ could not read DB credentials from secret"
|
|
ERRORS+=("$app:db_dump:could not read DB credentials from secret")
|
|
fi
|
|
;;
|
|
mariadb)
|
|
if [ "$db_name" = "DYNAMIC" ]; then
|
|
# Frappe: real db_name/db_password live in site_config.json on the
|
|
# sites PVC, not in the k8s Secret. Dump with root instead, but still
|
|
# need the site's real db_name to target the right database.
|
|
site_dir=$(kubectl exec -n "$ns" "deploy/${app_deploy}" -c "$app_ctr" -- \
|
|
sh -c 'for d in sites/*/; do [ -f "$d/site_config.json" ] && [ ! -L "${d%/}" ] && basename "$d" && break; done' 2>/dev/null)
|
|
db_name=$(kubectl exec -n "$ns" "deploy/${app_deploy}" -c "$app_ctr" -- \
|
|
cat "sites/${site_dir}/site_config.json" 2>/dev/null \
|
|
| python3 -c 'import json,sys; print(json.load(sys.stdin)["db_name"])' 2>/dev/null)
|
|
fi
|
|
db_pass=$(kubectl get secret "$secret_name" -n "$ns" -o jsonpath="{.data.${db_pass_key}}" 2>/dev/null | base64 -d)
|
|
if [ -n "$db_name" ] && [ -n "$db_pass" ]; then
|
|
kubectl exec -n "$ns" "deploy/${db_deploy}" -- env MYSQL_PWD="$db_pass" \
|
|
mysqldump -uroot --databases "$db_name" 2>"$APP_DIR/db-dump.err" \
|
|
| gzip > "$APP_DIR/db-dump.sql.gz"
|
|
mysqldump_rc=${PIPESTATUS[0]}
|
|
if [ "$mysqldump_rc" -eq 0 ] && [ -s "$APP_DIR/db-dump.sql.gz" ]; then
|
|
echo "✅ ($(du -h "$APP_DIR/db-dump.sql.gz" | cut -f1), db=$db_name)"
|
|
echo "$db_name" > "$APP_DIR/db-name.txt"
|
|
rm -f "$APP_DIR/db-dump.err"
|
|
else
|
|
echo "⚠️ FAILED (rc=$mysqldump_rc, see db-dump.err)"
|
|
ERRORS+=("$app:db_dump:rc=$mysqldump_rc $(tr '\n' ' ' < "$APP_DIR/db-dump.err" 2>/dev/null | cut -c1-160)")
|
|
fi
|
|
else
|
|
echo "⚠️ could not determine DB name/credentials"
|
|
ERRORS+=("$app:db_dump:could not determine DB name/credentials")
|
|
fi
|
|
;;
|
|
esac
|
|
|
|
# ---- 4. App-data PVC contents via kubectl exec (live tar, app-data only,
|
|
# never raw filesystem/PVC-path access — §5 sandbox-divergence rule) ----
|
|
exclude_args=()
|
|
if [ -n "${TAR_EXCLUDES[$app]:-}" ]; then
|
|
for _excl_dir in ${TAR_EXCLUDES[$app]}; do
|
|
exclude_args+=(--exclude="./$_excl_dir")
|
|
done
|
|
fi
|
|
echo -n " 📦 PVC data ($pvc_path${exclude_args:+, excl: ${TAR_EXCLUDES[$app]}}) ... "
|
|
kubectl exec -n "$ns" "deploy/${app_deploy}" -c "$app_ctr" -- \
|
|
tar czf - "${exclude_args[@]}" -C "$pvc_path" . 2>"$APP_DIR/pvc-data.err" \
|
|
> "$APP_DIR/pvc-data.tar.gz"
|
|
if [ -s "$APP_DIR/pvc-data.tar.gz" ]; then
|
|
echo "✅ ($(du -h "$APP_DIR/pvc-data.tar.gz" | cut -f1))"
|
|
rm -f "$APP_DIR/pvc-data.err"
|
|
else
|
|
echo "⚠️ FAILED (see pvc-data.err)"
|
|
ERRORS+=("$app:pvc_data:$(tr '\n' ' ' < "$APP_DIR/pvc-data.err" 2>/dev/null | cut -c1-160)")
|
|
fi
|
|
|
|
APP_COUNT=$((APP_COUNT + 1))
|
|
|
|
# Resource check between apps, not just at the start (§5: check every
|
|
# 1-2 min during long-running/heavy ops; PVC tars can be sizeable).
|
|
check_resources
|
|
done
|
|
|
|
# --------------------------------------------------
|
|
# Metadata + checksum
|
|
# --------------------------------------------------
|
|
echo ""
|
|
echo "📝 Writing backup metadata..."
|
|
cat > "$BACKUP_DIR/backup-info.txt" << EOF
|
|
Backup Name: $BACKUP_NAME
|
|
Backup Date: $(date)
|
|
Hostname: $(hostname)
|
|
Apps: $([ "$APPS_FILTER" = "all" ] && echo "$ALL_APPS" || echo "$APPS_FILTER")
|
|
Apps backed up: $APP_COUNT
|
|
kubectl: $(kubectl version --client -o yaml 2>/dev/null | grep gitVersion | head -1 || echo 'N/A')
|
|
Storage Tiers:
|
|
- Local: /root/backups/
|
|
- VM: ${VM_HOST}:${VM_PORT} → ${VM_DEST}
|
|
- Cloud: Cloudflare R2 → s3://${R2_BUCKET}/backups/
|
|
EOF
|
|
|
|
echo "" >> "$BACKUP_DIR/backup-info.txt"
|
|
echo "Per-app checksums:" >> "$BACKUP_DIR/backup-info.txt"
|
|
find "$BACKUP_DIR" -type f \( -name "*.tar.gz" -o -name "*.sql.gz" \) -print0 \
|
|
| xargs -0 -r sha256sum >> "$BACKUP_DIR/backup-info.txt"
|
|
echo " ✅ Done"
|
|
|
|
# --------------------------------------------------
|
|
# Compress (identical logic to backup-myapps.sh)
|
|
# --------------------------------------------------
|
|
echo ""
|
|
echo "🗜️ Compressing backup..."
|
|
cd /root/backups
|
|
|
|
tar -czf "${BACKUP_NAME}.tar.gz" "${BACKUP_NAME}/" 2>/dev/null
|
|
if [ $? -ne 0 ]; then
|
|
echo " ❌ CRITICAL: Compression failed!"
|
|
log_status "FAILED" "$BACKUP_NAME" "compression_failed"
|
|
exit 1
|
|
fi
|
|
|
|
COMPRESSED_SIZE=$(du -h "${BACKUP_NAME}.tar.gz" | cut -f1)
|
|
echo " ✅ Compressed size: $COMPRESSED_SIZE → $BACKUP_ARCHIVE"
|
|
|
|
sha256sum "${BACKUP_NAME}.tar.gz" > "${BACKUP_NAME}.tar.gz.sha256"
|
|
echo " ✅ Checksum written"
|
|
|
|
rm -rf "$BACKUP_DIR"
|
|
TIER1_STATUS="ok:${COMPRESSED_SIZE}"
|
|
echo ""
|
|
echo "✅ [TIER 1] Local backup complete: $BACKUP_ARCHIVE ($COMPRESSED_SIZE)"
|
|
|
|
# --------------------------------------------------
|
|
# Retention — keep only MAX_BACKUPS locally (own glob, own lineage)
|
|
# --------------------------------------------------
|
|
echo ""
|
|
echo "🧹 [Retention] Keeping latest ${MAX_BACKUPS} local k8s backups..."
|
|
ARCHIVE_LIST=$(ls -t /root/backups/myapps-k8s-backup-*.tar.gz 2>/dev/null || true)
|
|
ARCHIVE_COUNT=$(echo "$ARCHIVE_LIST" | grep -c '.tar.gz' 2>/dev/null || echo "0")
|
|
|
|
if [ "$ARCHIVE_COUNT" -gt "$MAX_BACKUPS" ]; then
|
|
TO_DELETE=$(echo "$ARCHIVE_LIST" | tail -n +$((MAX_BACKUPS + 1)))
|
|
while IFS= read -r old_file; do
|
|
[ -z "$old_file" ] && continue
|
|
rm -f "$old_file" "${old_file}.sha256" "${old_file}.meta.json"
|
|
echo " 🗑️ Deleted: $(basename "$old_file")"
|
|
done <<< "$TO_DELETE"
|
|
else
|
|
echo " ✅ ${ARCHIVE_COUNT}/${MAX_BACKUPS} backups — nothing to prune"
|
|
fi
|
|
|
|
# --------------------------------------------------
|
|
# VM transfer [TIER 2] — non-fatal (identical to backup-myapps.sh)
|
|
# --------------------------------------------------
|
|
echo ""
|
|
echo "📤 [TIER 2] Sending backup to VM (${VM_HOST}:${VM_PORT})..."
|
|
|
|
scp -i "$VM_KEY" -P "$VM_PORT" \
|
|
-o StrictHostKeyChecking=no -o ConnectTimeout=30 \
|
|
"${BACKUP_NAME}.tar.gz" \
|
|
"${VM_USER}@${VM_HOST}:${VM_DEST}" 2>/dev/null
|
|
VM_SCP_RC=$?
|
|
|
|
if [ $VM_SCP_RC -eq 0 ]; then
|
|
echo " ✅ Backup sent to VM successfully!"
|
|
scp -i "$VM_KEY" -P "$VM_PORT" \
|
|
-o StrictHostKeyChecking=no -o ConnectTimeout=15 \
|
|
"${BACKUP_NAME}.tar.gz.sha256" \
|
|
"${VM_USER}@${VM_HOST}:${VM_DEST}" 2>/dev/null || true
|
|
TIER2_STATUS="ok"
|
|
|
|
# VM-side retention — this tier never had any pruning at all until
|
|
# found accumulating 85 legacy pre-k8s archives (131GB) untouched
|
|
# since May. Mirrors the local [TIER 1] policy above: same
|
|
# MAX_BACKUPS, own glob/lineage only (never touches other
|
|
# unrelated backup dirs also living on this VM).
|
|
ssh -i "$VM_KEY" -p "$VM_PORT" -o StrictHostKeyChecking=no -o ConnectTimeout=15 \
|
|
"${VM_USER}@${VM_HOST}" \
|
|
"cd ${VM_DEST} && ls -t myapps-k8s-backup-*.tar.gz 2>/dev/null | tail -n +$((MAX_BACKUPS + 1)) | xargs -r -I{} rm -f {} {}.sha256 {}.meta.json" \
|
|
&>/dev/null || true
|
|
else
|
|
echo " ⚠️ VM transfer failed (rc=$VM_SCP_RC) — local backup is safe, continuing..."
|
|
TIER2_STATUS="failed:rc=${VM_SCP_RC}"
|
|
fi
|
|
|
|
# --------------------------------------------------
|
|
# Cloudflare R2 [TIER 3] — non-fatal (identical to backup-myapps.sh,
|
|
# key prefix stays "backups/" — shared with the legacy lineage in the
|
|
# same bucket, distinguished by filename prefix)
|
|
# --------------------------------------------------
|
|
echo ""
|
|
echo "☁️ [TIER 3] Uploading to Cloudflare R2..."
|
|
|
|
if [ -z "${AWS_ACCESS_KEY_ID:-}" ] || [ -z "${AWS_SECRET_ACCESS_KEY:-}" ]; then
|
|
echo " ⚠️ R2 credentials not found — skipping cloud upload"
|
|
TIER3_STATUS="skipped:no_credentials"
|
|
else
|
|
echo " Credentials: found (key=${AWS_ACCESS_KEY_ID:0:8}...)"
|
|
echo " Bucket: ${R2_BUCKET}"
|
|
echo " Endpoint: ${R2_ENDPOINT}"
|
|
|
|
PYTHON_BIN="/root/management-platform/venv/bin/python3"
|
|
if [ ! -f "$PYTHON_BIN" ]; then
|
|
PYTHON_BIN="python3"
|
|
fi
|
|
|
|
R2_SCRIPT="/tmp/r2_upload_$$.py"
|
|
cat > "$R2_SCRIPT" << PYEOF
|
|
import sys, os, boto3
|
|
from datetime import datetime, timezone
|
|
|
|
account_id = "${R2_ACCOUNT_ID}"
|
|
bucket = "${R2_BUCKET}"
|
|
endpoint = "https://${R2_ACCOUNT_ID}.r2.cloudflarestorage.com"
|
|
access_key = "${AWS_ACCESS_KEY_ID}"
|
|
secret_key = "${AWS_SECRET_ACCESS_KEY}"
|
|
archive = "/root/backups/${BACKUP_NAME}.tar.gz"
|
|
sha_file = archive + '.sha256'
|
|
key = "backups/${BACKUP_NAME}.tar.gz"
|
|
max_r2 = ${MAX_R2_BACKUPS}
|
|
name_prefix = "backups/myapps-k8s-backup-"
|
|
|
|
try:
|
|
client = boto3.client(
|
|
's3',
|
|
endpoint_url=endpoint,
|
|
aws_access_key_id=access_key,
|
|
aws_secret_access_key=secret_key,
|
|
region_name='auto'
|
|
)
|
|
|
|
try:
|
|
client.head_bucket(Bucket=bucket)
|
|
except Exception:
|
|
client.create_bucket(Bucket=bucket)
|
|
print(f' Created bucket: {bucket}')
|
|
|
|
size_mb = os.path.getsize(archive) / 1024 / 1024
|
|
timestamp = datetime.now(timezone.utc).isoformat()
|
|
print(f' Uploading {size_mb:.1f} MB to r2://{bucket}/{key}')
|
|
client.upload_file(
|
|
archive, bucket, key,
|
|
ExtraArgs={"Metadata": {
|
|
"uploaded-by": "backup-k8s-apps-script",
|
|
"uploaded-at": timestamp,
|
|
"original-file": archive.split("/")[-1],
|
|
}}
|
|
)
|
|
print(' ✅ R2 archive uploaded successfully')
|
|
|
|
if os.path.exists(sha_file):
|
|
client.upload_file(sha_file, bucket, key + '.sha256')
|
|
print(' ✅ R2 SHA256 checksum uploaded')
|
|
|
|
# Retention scoped to this script's own lineage (name_prefix) only —
|
|
# never touches the legacy myapps-backup-* objects in the same bucket.
|
|
print(f' 🧹 Enforcing R2 retention (max {max_r2}, prefix={name_prefix})...')
|
|
resp = client.list_objects_v2(Bucket=bucket, Prefix=name_prefix)
|
|
objects = sorted(
|
|
[o for o in resp.get('Contents', []) if not o['Key'].endswith('.sha256')],
|
|
key=lambda x: x['LastModified'],
|
|
reverse=True
|
|
)
|
|
to_delete = objects[max_r2:]
|
|
if not to_delete:
|
|
print(f' ✅ {len(objects)}/{max_r2} R2 backups — nothing to prune')
|
|
for obj in to_delete:
|
|
old_key = obj['Key']
|
|
client.delete_object(Bucket=bucket, Key=old_key)
|
|
for suffix in ('.sha256', '.meta.json'):
|
|
try:
|
|
client.delete_object(Bucket=bucket, Key=old_key + suffix)
|
|
except Exception:
|
|
pass
|
|
print(f' 🗑️ Deleted from R2: {old_key.replace("backups/", "")}')
|
|
|
|
sys.exit(0)
|
|
|
|
except Exception as e:
|
|
print(f' ERROR: {e}')
|
|
sys.exit(2)
|
|
PYEOF
|
|
|
|
"$PYTHON_BIN" "$R2_SCRIPT"
|
|
R2_RC=$?
|
|
rm -f "$R2_SCRIPT"
|
|
|
|
if [ $R2_RC -eq 0 ]; then
|
|
echo " ✅ R2 upload + retention complete"
|
|
TIER3_STATUS="ok"
|
|
else
|
|
echo " ⚠️ R2 upload failed (rc=$R2_RC) — local backup is safe at: $BACKUP_ARCHIVE"
|
|
TIER3_STATUS="failed:rc=${R2_RC}"
|
|
fi
|
|
fi
|
|
|
|
# --------------------------------------------------
|
|
# Status metadata sidecar — after all 3 tiers are known
|
|
# --------------------------------------------------
|
|
echo ""
|
|
echo "📝 Writing status metadata sidecar..."
|
|
write_meta_sidecar
|
|
|
|
# --------------------------------------------------
|
|
# Final summary + log
|
|
# --------------------------------------------------
|
|
echo ""
|
|
echo "========================================="
|
|
echo "✅ K8S BACKUP COMPLETE"
|
|
echo ""
|
|
echo " 📦 Name: $BACKUP_NAME"
|
|
echo " 💾 Local: $BACKUP_ARCHIVE ($COMPRESSED_SIZE)"
|
|
echo " 🖥️ VM: ${TIER2_STATUS}"
|
|
echo " ☁️ R2: ${TIER3_STATUS}"
|
|
echo "========================================="
|
|
|
|
log_status "SUCCESS" "$BACKUP_NAME" \
|
|
"apps=${APP_COUNT} | size=${COMPRESSED_SIZE} | tier1=${TIER1_STATUS} | tier2=${TIER2_STATUS} | tier3=${TIER3_STATUS}"
|