#!/usr/bin/env bash # failover.sh — graceful HA cluster failover # # Detects which node is active and moves all resources to the other node by # putting the active node into Pacemaker standby. Waits for the XFS mount to # appear on the target before returning. # # Usage: # scripts/ha/failover.sh [--to node1|node2] [--force] [--timeout ] [--dry-run] # # --to node1|node2 target node (default: the node that is NOT currently active) # --force skip the interactive confirmation prompt # --timeout seconds to wait for resources to move (default: 120) # --dry-run show what would be done without changing anything set -euo pipefail # ── Configuration ───────────────────────────────────────────────────────── NODE1="${NODE1:-ha-server-1}" NODE2="${NODE2:-ha-server-2}" NODE1_IP="${NODE1_IP:-192.168.2.228}" NODE2_IP="${NODE2_IP:-192.168.2.227}" VIP="${VIP:-192.168.2.229}" XFS_MOUNT="${XFS_MOUNT:-/srv/ha-data}" HA_USER="${HA_USER:-nixos}" # ────────────────────────────────────────────────────────────────────────── SSH_OPTS="-i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no -o ConnectTimeout=5" n1() { ssh $SSH_OPTS "${HA_USER}@${NODE1_IP}" sudo "$@" 2>/dev/null; } n2() { ssh $SSH_OPTS "${HA_USER}@${NODE2_IP}" sudo "$@" 2>/dev/null; } # ── Argument parsing ─────────────────────────────────────────────────────── TARGET_NODE="" FORCE=false DRY_RUN=false TIMEOUT=120 while [[ $# -gt 0 ]]; do case "$1" in --to) shift case "${1:-}" in node1|ha-server-1) TARGET_NODE="$NODE1" ;; node2|ha-server-2) TARGET_NODE="$NODE2" ;; *) echo "ERROR: --to must be node1, node2, ha-server-1, or ha-server-2"; exit 1 ;; esac ;; --force) FORCE=true ;; --dry-run) DRY_RUN=true ;; --timeout) shift; TIMEOUT="${1:?--timeout requires a value}" ;; *) echo "Unknown argument: $1"; echo "Usage: $0 [--to node1|node2] [--force] [--timeout ] [--dry-run]"; exit 1 ;; esac shift done DRY_PREFIX="" $DRY_RUN && DRY_PREFIX="[dry-run] " echo "════════════════════════════════════════════════════" echo " HA Cluster Failover — $(date '+%Y-%m-%d %H:%M:%S')" $DRY_RUN && echo " MODE: dry-run — no changes will be made" echo "════════════════════════════════════════════════════" # ── Detect active node ───────────────────────────────────────────────────── echo "" echo "Detecting active node..." CRM_OUT="" if ssh $SSH_OPTS "${HA_USER}@${NODE1_IP}" true 2>/dev/null; then CRM_OUT=$(n1 "crm_mon -1 2>/dev/null" 2>/dev/null || true) fi if [[ -z "$CRM_OUT" ]]; then if ssh $SSH_OPTS "${HA_USER}@${NODE2_IP}" true 2>/dev/null; then CRM_OUT=$(n2 "crm_mon -1 2>/dev/null" 2>/dev/null || true) fi fi ACTIVE_NODE=$(echo "$CRM_OUT" | grep -E '^\s*(Promoted|Masters):' | \ grep -oE '\b(ha-server-[0-9]+)\b' | head -1 || true) if [[ -z "$ACTIVE_NODE" ]]; then echo "" echo "ERROR: could not determine active node from crm_mon." echo " Is Pacemaker still settling? Try running scripts/ha/health.sh first." echo " If Pacemaker is down on both nodes, manual recovery is required." exit 1 fi if [[ "$ACTIVE_NODE" == "$NODE1" ]]; then ACTIVE_IP="$NODE1_IP" STANDBY_NODE="$NODE2" STANDBY_IP="$NODE2_IP" na() { n1 "$@"; } ns() { n2 "$@"; } else ACTIVE_IP="$NODE2_IP" STANDBY_NODE="$NODE1" STANDBY_IP="$NODE1_IP" na() { n2 "$@"; } ns() { n1 "$@"; } fi echo " Active: $ACTIVE_NODE ($ACTIVE_IP)" echo " Standby: $STANDBY_NODE ($STANDBY_IP)" # ── Validate target ──────────────────────────────────────────────────────── if [[ -n "$TARGET_NODE" ]]; then if [[ "$TARGET_NODE" == "$ACTIVE_NODE" ]]; then echo "" echo "ERROR: $TARGET_NODE is already the active node — nothing to do." exit 1 fi echo " Target: $TARGET_NODE (as requested)" else echo " Target: $STANDBY_NODE (auto — the other node)" fi # ── Pre-checks ───────────────────────────────────────────────────────────── echo "" echo "Pre-checks..." DRBD_DSTATE=$(na "drbdadm dstate ha-data 2>/dev/null" 2>/dev/null || echo "unknown") if ! echo "$DRBD_DSTATE" | grep -q "UpToDate/UpToDate"; then echo "" echo " WARNING: DRBD dstate is '$DRBD_DSTATE' (not UpToDate/UpToDate)." echo " Failing over with a partially-synced disk risks split-brain." if ! $FORCE; then echo " Use --force to proceed anyway (not recommended)." exit 1 fi echo " --force specified — proceeding despite non-ideal DRBD state." else echo " DRBD dstate: $DRBD_DSTATE — OK" fi QUORUM_OK=$(na "corosync-quorumtool -s 2>/dev/null | grep -c 'Quorate:.*Yes'" 2>/dev/null || echo "0") if [[ "$QUORUM_OK" -lt 1 ]]; then echo " ERROR: cluster does not have quorum — failover would be unsafe." exit 1 fi echo " Quorum: OK" # ── Confirm ──────────────────────────────────────────────────────────────── if ! $FORCE && ! $DRY_RUN; then echo "" echo " This will move all resources from $ACTIVE_NODE → $STANDBY_NODE." echo " VIP and services will be unreachable for ~10–30 seconds." printf " Proceed? [y/N] " read -r ANSWER [[ "${ANSWER,,}" == "y" || "${ANSWER,,}" == "yes" ]] || { echo "Aborted."; exit 0; } fi # ── Capture active node's crm_node name ─────────────────────────────────── # crm_node -n returns the node name as registered in Pacemaker (may differ # from hostname if Pacemaker was configured with explicit node names). ACTIVE_CRMD_NAME=$(na "crm_node -n 2>/dev/null" 2>/dev/null || echo "$ACTIVE_NODE") # ── Perform failover ─────────────────────────────────────────────────────── echo "" echo "${DRY_PREFIX}Putting $ACTIVE_NODE into standby (resources will migrate to $STANDBY_NODE)..." if ! $DRY_RUN; then na "crm_standby -N '${ACTIVE_CRMD_NAME}' -v on" 2>/dev/null || true fi # ── Wait for resources to move ───────────────────────────────────────────── echo "${DRY_PREFIX}Waiting up to ${TIMEOUT}s for XFS to mount on $STANDBY_NODE..." MOVED=false SPIN_CHARS=('|' '/' '-' '\') SPIN_I=0 if $DRY_RUN; then echo " [dry-run] would wait for mountpoint $XFS_MOUNT on $STANDBY_NODE" MOVED=true else for i in $(seq 1 "$TIMEOUT"); do if ns "mountpoint -q '${XFS_MOUNT}' 2>/dev/null" 2>/dev/null; then printf "\r%-80s\r" "" echo " Resources moved in ${i}s" MOVED=true break fi SPIN_I=$(( SPIN_I + 1 )) SC="${SPIN_CHARS[$((SPIN_I % 4))]}" printf "\r [%s] waiting... (%ds) " "$SC" "$i" sleep 1 done fi if ! $MOVED; then echo "" echo "ERROR: XFS did not mount on $STANDBY_NODE within ${TIMEOUT}s." echo "" echo " Current resource state:" na "crm_mon -1 2>/dev/null" 2>/dev/null | grep -E 'Started|Stopped|Promoted|Unpromoted|FAILED' | sed 's/^/ /' || true echo "" echo " Clearing standby to restore $ACTIVE_NODE (undo the failover attempt)..." na "crm_standby -N '${ACTIVE_CRMD_NAME}' -v off" 2>/dev/null || true na "crm_resource --cleanup" 2>/dev/null || true exit 1 fi # ── Clear failure history ────────────────────────────────────────────────── echo "${DRY_PREFIX}Clearing Pacemaker failure history..." if ! $DRY_RUN; then ns "crm_resource --cleanup 2>/dev/null" 2>/dev/null || true fi # ── Re-enable original active node as standby ───────────────────────────── echo "${DRY_PREFIX}Re-enabling $ACTIVE_NODE (now standby — will not claim resources)..." if ! $DRY_RUN; then na "crm_standby -N '${ACTIVE_CRMD_NAME}' -v off" 2>/dev/null || true fi # ── Wait briefly for DRBD resync to begin ───────────────────────────────── if ! $DRY_RUN; then sleep 5 fi # ── Final state ──────────────────────────────────────────────────────────── echo "" echo "Failover complete. Final state:" echo "" CRM_OUT_AFTER="" if ! $DRY_RUN; then CRM_OUT_AFTER=$(ns "crm_mon -1 2>/dev/null" 2>/dev/null || na "crm_mon -1 2>/dev/null" 2>/dev/null || true) else CRM_OUT_AFTER="$CRM_OUT" fi NEW_ACTIVE=$(echo "$CRM_OUT_AFTER" | grep -E '^\s*(Promoted|Masters):' | \ grep -oE '\b(ha-server-[0-9]+)\b' | head -1 || true) if [[ -n "$NEW_ACTIVE" ]]; then if [[ "$NEW_ACTIVE" == "$ACTIVE_NODE" ]]; then echo " WARNING: $ACTIVE_NODE is still showing as active in crm_mon." echo " Pacemaker may still be settling — check again in a few seconds." else echo " Active: $NEW_ACTIVE" echo " Standby: $ACTIVE_NODE" fi fi echo "" RESOURCES_AFTER=$(echo "$CRM_OUT_AFTER" | awk '/Full List of Resources/,0' | tail -n +2 || true) [[ -z "$RESOURCES_AFTER" ]] && RESOURCES_AFTER=$(echo "$CRM_OUT_AFTER" | \ grep -E 'Started|Stopped|Promoted|Unpromoted|FAILED|Master|Slave' || true) [[ -n "$RESOURCES_AFTER" ]] && echo "$RESOURCES_AFTER" | sed 's/^/ /' echo "" echo " (DRBD resync of $ACTIVE_NODE may take a moment; monitor with:" echo " ssh nixos@${ACTIVE_IP} 'sudo watch -n3 cat /proc/drbd')" echo "" echo "════════════════════════════════════════════════════"