From 5329799a5a0ed070e61c7b03c3bc4bce9e049ff9 Mon Sep 17 00:00:00 2001 From: root Date: Fri, 21 Aug 2026 13:04:05 +0200 Subject: [PATCH] Add sync-standby-platform.sh, replacing the dead standby-sync pipeline MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The VM standby's management-platform was found 2+ months stale during failover investigation (last deploy: June 10). Root cause: the old pull-platform-backup.sh + deploy-platform.sh pair on the VM pulled/ redeployed platform-backup-*.tar.gz tarballs of /root/management-platform, but tarball generation was disabled 2026-08-14 when the source of truth moved to git-tracked /root/CloudOps — nothing replaced it, so the VM kept re-deploying its last cached snapshot every hour indefinitely. New script runs hourly from the main server (cron) instead of pulling from the VM — pushes /root/CloudOps/platform directly via rsync, plus the live /root/management-platform/config.py (gitignored, holds real secrets/DB creds — copied byte-for-byte since it already contains the RUNNING_ON_MAIN_SERVER auto-detect logic that makes it work correctly unmodified on both servers). Only restarts the systemd service when rsync or config actually changed. Old pull/deploy cron entries commented out on the VM (superseded, not deleted, for history). Verified end-to-end: ran it live, standby now serves current code (status- dot markers present in synced templates/app.py), service restarted clean, config auto-detect correctly logs "running on VM / backup host" and takes the SSH-fallback code path as designed. --- scripts/sync-standby-platform.sh | 108 +++++++++++++++++++++++++++++++ 1 file changed, 108 insertions(+) create mode 100755 scripts/sync-standby-platform.sh diff --git a/scripts/sync-standby-platform.sh b/scripts/sync-standby-platform.sh new file mode 100755 index 0000000..c196609 --- /dev/null +++ b/scripts/sync-standby-platform.sh @@ -0,0 +1,108 @@ +#!/bin/bash +# ───────────────────────────────────────────────────────────── +# sync-standby-platform.sh +# Keeps the warm-standby copy of management-platform on the VM +# (178.18.243.51) in sync with the real source of truth on this +# server (/root/CloudOps/platform), pushed directly over SSH. +# +# Replaces the old pull-platform-backup.sh + deploy-platform.sh +# pair on the VM, which pulled/redeployed a platform-backup-*.tar.gz +# tarball of /root/management-platform (the pre-migration Docker-era +# deploy dir). That tarball generation was disabled on 2026-08-14 +# when the platform's source of truth moved to git-tracked +# /root/CloudOps + Jenkins — nothing replaced it on the VM side, so +# the standby silently kept re-deploying its last cached snapshot +# (found stale at 2+ months old during PFE failover investigation). +# +# config.py is intentionally NOT sourced from /root/CloudOps/platform +# (gitignored there, and can drift — see that file's own history). +# The live /root/management-platform/config.py on THIS server is the +# real one actually mounted into the running pod; it already contains +# the RUNNING_ON_MAIN_SERVER auto-detect logic that makes the exact +# same file behave correctly on both this server and the VM, so it's +# copied byte-for-byte rather than templated. +# +# Run hourly via cron on this server (NOT on the VM — the VM has no +# reason to pull, this server already pushes everywhere else it +# manages, e.g. backup-k8s-apps.sh's VM tier). +# ───────────────────────────────────────────────────────────── + +VM_HOST="178.18.243.51" +VM_PORT="22" +VM_USER="root" +VM_KEY="/root/.ssh/id_rsa" +VM_DEPLOY_DIR="/root/management-platform" + +LOCAL_SRC="/root/CloudOps/platform/" +LIVE_CONFIG="/root/management-platform/config.py" +LOG="/root/CloudOps/scripts/sync-standby-platform.log" + +SSH_OPTS="-i $VM_KEY -p $VM_PORT -o StrictHostKeyChecking=no -o ConnectTimeout=15" +SCP_OPTS="-i $VM_KEY -P $VM_PORT -o StrictHostKeyChecking=no -o ConnectTimeout=15" + +log() { echo "$(date '+%Y-%m-%d %H:%M:%S') $1" | tee -a "$LOG"; } + +log "🔄 Syncing platform code → $VM_USER@$VM_HOST:$VM_DEPLOY_DIR" + +ssh $SSH_OPTS "$VM_USER@$VM_HOST" "mkdir -p $VM_DEPLOY_DIR" || { + log "❌ Cannot reach VM — aborting" + exit 1 +} + +# --itemize-changes so we can tell below whether anything actually +# changed (drives the restart-only-if-needed decision). +CHANGES=$(rsync -a --delete -e "ssh $SSH_OPTS" \ + --exclude venv --exclude __pycache__ --exclude '*.pyc' \ + --exclude config.py --exclude '*.log' \ + --itemize-changes \ + "$LOCAL_SRC" "$VM_USER@$VM_HOST:$VM_DEPLOY_DIR/" 2>>"$LOG") +RSYNC_RC=$? + +if [ $RSYNC_RC -ne 0 ]; then + log "❌ rsync failed (rc=$RSYNC_RC)" + exit 1 +fi + +CONFIG_CHANGED="" +if ! scp $SCP_OPTS "$LIVE_CONFIG" "$VM_USER@$VM_HOST:$VM_DEPLOY_DIR/config.py.new" &>>"$LOG"; then + log "❌ Could not copy live config.py to VM" + exit 1 +fi +if ! ssh $SSH_OPTS "$VM_USER@$VM_HOST" \ + "cmp -s $VM_DEPLOY_DIR/config.py $VM_DEPLOY_DIR/config.py.new 2>/dev/null"; then + CONFIG_CHANGED="yes" +fi +ssh $SSH_OPTS "$VM_USER@$VM_HOST" "mv $VM_DEPLOY_DIR/config.py.new $VM_DEPLOY_DIR/config.py" + +if [ -z "$CHANGES" ] && [ -z "$CONFIG_CHANGED" ]; then + log "✅ Already up to date — nothing changed, no restart needed" + exit 0 +fi + +log "📝 Changes detected:" +echo "$CHANGES" | tee -a "$LOG" +[ -n "$CONFIG_CHANGED" ] && log " (config.py changed too)" + +log "📦 Syncing venv deps (requirements.txt) ..." +ssh $SSH_OPTS "$VM_USER@$VM_HOST" " + cd $VM_DEPLOY_DIR + chmod +x *.sh 2>/dev/null + chmod +x *.py 2>/dev/null + if [ ! -d venv ]; then + python3 -m venv venv + venv/bin/python -m ensurepip --upgrade 2>/dev/null || true + fi + venv/bin/pip install --quiet -r requirements.txt +" >>"$LOG" 2>&1 + +log "🔁 Restarting management-platform service on VM ..." +ssh $SSH_OPTS "$VM_USER@$VM_HOST" "systemctl restart management-platform" +sleep 3 +STATUS=$(ssh $SSH_OPTS "$VM_USER@$VM_HOST" "systemctl is-active management-platform 2>/dev/null") + +if [ "$STATUS" = "active" ]; then + log "✅ Standby platform restarted and running latest code" +else + log "⚠️ Service not active after restart (status: $STATUS) — check journalctl -u management-platform on the VM" + exit 1 +fi