Two gaps found testing this live from the standby: 1. restore.html's "Restore on This Server" was checked by default regardless of RUNNING_ON_MAIN_SERVER — on the standby that option is nonsensical (no local cluster/kubectl at all) and restore_start() would have just tried and failed confusingly. Now: that radio is disabled with an explanatory note when not on the main server, "External Machine" is checked instead and pre-filled with the tunnel details (localhost:2224, contabo-key) so restoring from the standby just targets the real main server without the user having to know any of that. Added a matching server-side guard in restore_start() for target=='local' + not RUNNING_ON_MAIN_SERVER (defense in depth — the UI already prevents it, this catches a direct API call too). Also fixed refreshSystemMetrics() in platform.js, which would have overwritten the disabled option's label with the (now-correct, see previous commit) main-server hostname — looking like a working local target when it isn't. 2. sync-standby-platform.sh only ever mirrored platform/ — but restore_start() references /root/CloudOps/backup/restore-k8s-apps.sh as a fixed absolute path to scp to the remote target, and that directory never existed on the VM at all. Every restore attempt from the standby failed immediately with "restore-k8s-apps.sh not found", regardless of target. Now mirrors /root/CloudOps/backup/ too. Verified end-to-end for real: triggered a restore of frappe/erpnext from the standby's actual web UI (target=remote, localhost:2224) — connected over the tunnel, copied the backup archive + script to the main server, ran restore-k8s-apps.sh there, scaled the deployment down/up, restored the DB. Confirmed after: 740 tables in the DB, /api/method/ping responding on the live pod. Not a dry run — a real restore, actually initiated from the standby machine.
128 lines
5.2 KiB
Bash
Executable File
128 lines
5.2 KiB
Bash
Executable File
#!/bin/bash
|
|
# ─────────────────────────────────────────────────────────────
|
|
# sync-standby-platform.sh
|
|
# Keeps the warm-standby copy of management-platform on the VM
|
|
# (178.18.243.51) in sync with the real source of truth on this
|
|
# server (/root/CloudOps/platform), pushed directly over SSH.
|
|
#
|
|
# Replaces the old pull-platform-backup.sh + deploy-platform.sh
|
|
# pair on the VM, which pulled/redeployed a platform-backup-*.tar.gz
|
|
# tarball of /root/management-platform (the pre-migration Docker-era
|
|
# deploy dir). That tarball generation was disabled on 2026-08-14
|
|
# when the platform's source of truth moved to git-tracked
|
|
# /root/CloudOps + Jenkins — nothing replaced it on the VM side, so
|
|
# the standby silently kept re-deploying its last cached snapshot
|
|
# (found stale at 2+ months old during PFE failover investigation).
|
|
#
|
|
# config.py is intentionally NOT sourced from /root/CloudOps/platform
|
|
# (gitignored there, and can drift — see that file's own history).
|
|
# The live /root/management-platform/config.py on THIS server is the
|
|
# real one actually mounted into the running pod; it already contains
|
|
# the RUNNING_ON_MAIN_SERVER auto-detect logic that makes the exact
|
|
# same file behave correctly on both this server and the VM, so it's
|
|
# copied byte-for-byte rather than templated.
|
|
#
|
|
# Run hourly via cron on this server (NOT on the VM — the VM has no
|
|
# reason to pull, this server already pushes everywhere else it
|
|
# manages, e.g. backup-k8s-apps.sh's VM tier).
|
|
# ─────────────────────────────────────────────────────────────
|
|
|
|
VM_HOST="178.18.243.51"
|
|
VM_PORT="22"
|
|
VM_USER="root"
|
|
VM_KEY="/root/.ssh/id_rsa"
|
|
VM_DEPLOY_DIR="/root/management-platform"
|
|
|
|
LOCAL_SRC="/root/CloudOps/platform/"
|
|
LIVE_CONFIG="/root/management-platform/config.py"
|
|
|
|
# restore_start()/api_backup_run() in app.py reference this as a fixed
|
|
# absolute path (/root/CloudOps/backup/restore-k8s-apps.sh) — it's a
|
|
# sibling of platform/, never baked into the image, reached via the same
|
|
# rw hostPath mount as everything else. Restoring FROM the standby (target
|
|
# = remote, pointed at the main server over the tunnel) needs a local copy
|
|
# here to scp across, so mirror it too, not just platform/.
|
|
LOCAL_BACKUP_SRC="/root/CloudOps/backup/"
|
|
VM_BACKUP_DIR="/root/CloudOps/backup"
|
|
LOG="/root/CloudOps/scripts/sync-standby-platform.log"
|
|
|
|
SSH_OPTS="-i $VM_KEY -p $VM_PORT -o StrictHostKeyChecking=no -o ConnectTimeout=15"
|
|
SCP_OPTS="-i $VM_KEY -P $VM_PORT -o StrictHostKeyChecking=no -o ConnectTimeout=15"
|
|
|
|
log() { echo "$(date '+%Y-%m-%d %H:%M:%S') $1" | tee -a "$LOG"; }
|
|
|
|
log "🔄 Syncing platform code → $VM_USER@$VM_HOST:$VM_DEPLOY_DIR"
|
|
|
|
ssh $SSH_OPTS "$VM_USER@$VM_HOST" "mkdir -p $VM_DEPLOY_DIR" || {
|
|
log "❌ Cannot reach VM — aborting"
|
|
exit 1
|
|
}
|
|
|
|
# --itemize-changes so we can tell below whether anything actually
|
|
# changed (drives the restart-only-if-needed decision).
|
|
CHANGES=$(rsync -a --delete -e "ssh $SSH_OPTS" \
|
|
--exclude venv --exclude __pycache__ --exclude '*.pyc' \
|
|
--exclude config.py --exclude '*.log' \
|
|
--itemize-changes \
|
|
"$LOCAL_SRC" "$VM_USER@$VM_HOST:$VM_DEPLOY_DIR/" 2>>"$LOG")
|
|
RSYNC_RC=$?
|
|
|
|
if [ $RSYNC_RC -ne 0 ]; then
|
|
log "❌ rsync failed (rc=$RSYNC_RC)"
|
|
exit 1
|
|
fi
|
|
|
|
ssh $SSH_OPTS "$VM_USER@$VM_HOST" "mkdir -p $VM_BACKUP_DIR"
|
|
BACKUP_CHANGES=$(rsync -a --delete -e "ssh $SSH_OPTS" \
|
|
--itemize-changes \
|
|
"$LOCAL_BACKUP_SRC" "$VM_USER@$VM_HOST:$VM_BACKUP_DIR/" 2>>"$LOG")
|
|
if [ $? -ne 0 ]; then
|
|
log "❌ rsync of backup/ failed"
|
|
exit 1
|
|
fi
|
|
CHANGES="${CHANGES}${BACKUP_CHANGES}"
|
|
|
|
CONFIG_CHANGED=""
|
|
if ! scp $SCP_OPTS "$LIVE_CONFIG" "$VM_USER@$VM_HOST:$VM_DEPLOY_DIR/config.py.new" &>>"$LOG"; then
|
|
log "❌ Could not copy live config.py to VM"
|
|
exit 1
|
|
fi
|
|
if ! ssh $SSH_OPTS "$VM_USER@$VM_HOST" \
|
|
"cmp -s $VM_DEPLOY_DIR/config.py $VM_DEPLOY_DIR/config.py.new 2>/dev/null"; then
|
|
CONFIG_CHANGED="yes"
|
|
fi
|
|
ssh $SSH_OPTS "$VM_USER@$VM_HOST" "mv $VM_DEPLOY_DIR/config.py.new $VM_DEPLOY_DIR/config.py"
|
|
|
|
if [ -z "$CHANGES" ] && [ -z "$CONFIG_CHANGED" ]; then
|
|
log "✅ Already up to date — nothing changed, no restart needed"
|
|
exit 0
|
|
fi
|
|
|
|
log "📝 Changes detected:"
|
|
echo "$CHANGES" | tee -a "$LOG"
|
|
[ -n "$CONFIG_CHANGED" ] && log " (config.py changed too)"
|
|
|
|
log "📦 Syncing venv deps (requirements.txt) ..."
|
|
ssh $SSH_OPTS "$VM_USER@$VM_HOST" "
|
|
cd $VM_DEPLOY_DIR
|
|
chmod +x *.sh 2>/dev/null
|
|
chmod +x *.py 2>/dev/null
|
|
if [ ! -d venv ]; then
|
|
python3 -m venv venv
|
|
venv/bin/python -m ensurepip --upgrade 2>/dev/null || true
|
|
fi
|
|
venv/bin/pip install --quiet -r requirements.txt
|
|
" >>"$LOG" 2>&1
|
|
|
|
log "🔁 Restarting management-platform service on VM ..."
|
|
ssh $SSH_OPTS "$VM_USER@$VM_HOST" "systemctl restart management-platform"
|
|
sleep 3
|
|
STATUS=$(ssh $SSH_OPTS "$VM_USER@$VM_HOST" "systemctl is-active management-platform 2>/dev/null")
|
|
|
|
if [ "$STATUS" = "active" ]; then
|
|
log "✅ Standby platform restarted and running latest code"
|
|
else
|
|
log "⚠️ Service not active after restart (status: $STATUS) — check journalctl -u management-platform on the VM"
|
|
exit 1
|
|
fi
|