Archived
Check NixOS configurations / eval-hosts (push) Successful in 10m26s
drbdmeta lives in the same Nix store dir as drbdadm but sudo doesn't inherit the full PATH, so drbdmeta was not found (exit 127) even though drbdadm was. Resolve drbdmeta's directory from drbdadm's location and prepend it to PATH. Replace openssl rand for UUID generation with /proc/sys/kernel/random/uuid — openssl is not guaranteed to be on PATH in a minimal NixOS root environment, but /proc/sys/kernel/random/uuid is always present. Apply the same PATH fix on NODE2 inline in the bash -c invocations that call drbdmeta over SSH. Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
397 lines
19 KiB
Bash
Executable File
397 lines
19 KiB
Bash
Executable File
#!/usr/bin/env bash
|
|
# cluster-init.sh — one-time HA cluster initialisation script
|
|
#
|
|
# Run ONCE from ha-server-1 as root AFTER both VMs are booted and have SSH
|
|
# access. It:
|
|
# 1. Generates and distributes the corosync authkey
|
|
# 2. Waits for corosync quorum and pacemaker
|
|
# 3. Initialises DRBD metadata, promotes node1 to primary
|
|
# 4. Creates XFS on /dev/drbd0 and mounts it
|
|
# 5. Creates the directory tree and iSCSI LUN backing file
|
|
# 6. Configures LIO iSCSI target (file-backed LUN)
|
|
# 7. Configures Pacemaker resources: DRBD → XFS → iSCSI → NFS → VIP
|
|
#
|
|
# Prerequisites:
|
|
# - Both VMs booted with the ha-server config (nixos-rebuild done)
|
|
# - SSH key access from node1 to root@NODE2_IP
|
|
# - VMID_NODE1 / VMID_NODE2 filled in below (needed for STONITH setup;
|
|
# cluster starts without STONITH, which you enable separately via
|
|
# scripts/ha/cluster-enable-stonith.sh)
|
|
# - Run as root on ha-server-1
|
|
set -euo pipefail
|
|
|
|
# ── Configuration ─────────────────────────────────────────────────────────
|
|
# All values override-able via environment variables; defaults match variables.nix.
|
|
NODE1="${NODE1:-ha-server-1}"
|
|
NODE2="${NODE2:-ha-server-2}"
|
|
NODE1_IP="${NODE1_IP:-192.168.2.228}" # vars.haServer1Ip
|
|
NODE2_IP="${NODE2_IP:-192.168.2.227}" # vars.haServer2Ip
|
|
VIP="${VIP:-192.168.2.229}" # vars.haServerVip
|
|
XFS_MOUNT="${XFS_MOUNT:-/srv/ha-data}" # vars.haStorageRoot
|
|
ISCSI_IQN="${ISCSI_IQN:-iqn.2026-01.home.sweet:ha-storage}" # vars.haIscsiIqn
|
|
ISCSI_LUN_FILE="${XFS_MOUNT}/iscsi-lun.img"
|
|
ISCSI_LUN_SIZE="10G"
|
|
DRBD_DEVICE="/dev/drbd0"
|
|
VMID_NODE1="${VMID_NODE1:-}" # set by deploy.sh; needed for STONITH
|
|
VMID_NODE2="${VMID_NODE2:-}"
|
|
PVE_HOST="${PVE_HOST:-pve1.sweet.home}"
|
|
PVE_USER="${PVE_USER:-wayne}"
|
|
# Inter-node SSH: HA_USER is the user to SSH as on NODE2; HA_KEY is the private
|
|
# key to use. Default is root-to-root (no key arg). deploy.sh sets HA_USER=nixos
|
|
# and HA_KEY=/tmp/cluster-init-key so the script works even when root-to-root SSH
|
|
# is not available.
|
|
HA_USER="${HA_USER:-root}"
|
|
HA_KEY="${HA_KEY:-}"
|
|
|
|
# NFS dataset subdirectories to create under XFS_MOUNT.
|
|
# Must mirror vars.nfsShares subpath values in variables.nix.
|
|
NFS_SUBDIRS=(
|
|
"docker/config"
|
|
"docker/volumes"
|
|
"docker/databases"
|
|
"docker/nextcloud-data"
|
|
"raspi/volumes"
|
|
"proxmox/iso"
|
|
"proxmox/lxc"
|
|
"pxe-boot/images"
|
|
)
|
|
# ──────────────────────────────────────────────────────────────────────────
|
|
|
|
log() { echo "[cluster-init] $*"; }
|
|
die() { echo "[cluster-init] ERROR: $*" >&2; exit 1; }
|
|
warn() { echo "[cluster-init] WARNING: $*" >&2; }
|
|
|
|
[[ $(id -u) -eq 0 ]] || die "must run as root"
|
|
[[ "$(hostname)" == "$NODE1" ]] || die "must run on $NODE1"
|
|
|
|
# NixOS may not include xfsprogs in root's PATH even when it's in the store.
|
|
# If mkfs.xfs is missing, search the Nix store for it.
|
|
if ! command -v mkfs.xfs &>/dev/null; then
|
|
_xfs_bin=$(find /nix/store -maxdepth 3 -name mkfs.xfs 2>/dev/null | head -1 | xargs dirname 2>/dev/null || true)
|
|
[[ -n "$_xfs_bin" ]] && export PATH="$_xfs_bin:$PATH" \
|
|
|| die "mkfs.xfs not found — add xfsprogs to ha-server.nix environment.systemPackages and rebuild"
|
|
fi
|
|
|
|
# drbdmeta lives alongside drbdadm but may not be in PATH when run via sudo.
|
|
if ! command -v drbdmeta &>/dev/null; then
|
|
_drbd_bin=$(dirname "$(command -v drbdadm)" 2>/dev/null || true)
|
|
[[ -n "$_drbd_bin" ]] && export PATH="$_drbd_bin:$PATH" \
|
|
|| die "drbdmeta not found — is drbd-utils in ha-server environment.systemPackages?"
|
|
fi
|
|
|
|
# Portable 16-hex-char UUID generator (no openssl required).
|
|
_rand_uuid() {
|
|
cat /proc/sys/kernel/random/uuid 2>/dev/null | tr -d '-' | cut -c1-16 | tr '[:lower:]' '[:upper:]'
|
|
}
|
|
|
|
# Inter-node SSH/SCP helpers — abstract over root-to-root vs nixos+sudo.
|
|
_SSH_OPTS="-o StrictHostKeyChecking=no -o ConnectTimeout=10"
|
|
[[ -n "$HA_KEY" ]] && _SSH_OPTS="-i $HA_KEY $_SSH_OPTS"
|
|
if [[ "$HA_USER" == "root" ]]; then
|
|
n2_ssh() { ssh $_SSH_OPTS "root@${NODE2_IP}" "$@"; }
|
|
n2_scp() { scp $_SSH_OPTS "$1" "root@${NODE2_IP}:$2"; }
|
|
else
|
|
# Non-root user with passwordless sudo; wrap each command with sudo.
|
|
n2_ssh() { ssh $_SSH_OPTS "${HA_USER}@${NODE2_IP}" sudo "$@"; }
|
|
n2_scp() {
|
|
# SCP to a tmp path, then sudo-move to the real destination as the remote user.
|
|
local src="$1" dst="$2"
|
|
local tmp="/tmp/_cluster_init_scp_$$"
|
|
scp $_SSH_OPTS "$src" "${HA_USER}@${NODE2_IP}:${tmp}"
|
|
ssh $_SSH_OPTS "${HA_USER}@${NODE2_IP}" sudo mv "${tmp}" "${dst}"
|
|
}
|
|
fi
|
|
|
|
# ── 0. Corosync authkey ───────────────────────────────────────────────────
|
|
AUTHKEY="/etc/corosync/authkey"
|
|
mkdir -p /etc/corosync
|
|
if [[ ! -f "$AUTHKEY" ]]; then
|
|
log "Generating corosync authkey..."
|
|
corosync-keygen -k "$AUTHKEY"
|
|
chmod 0400 "$AUTHKEY"
|
|
fi
|
|
log "Distributing authkey to $NODE2..."
|
|
n2_ssh "mkdir -p /etc/corosync"
|
|
n2_scp "$AUTHKEY" "$AUTHKEY"
|
|
n2_ssh "chmod 0400 '${AUTHKEY}'"
|
|
|
|
log "Restarting corosync and pacemaker on both nodes..."
|
|
systemctl restart corosync
|
|
n2_ssh "systemctl restart corosync"
|
|
sleep 3
|
|
|
|
log "Starting pacemaker on both nodes (may have failed at boot before authkey was placed)..."
|
|
systemctl start pacemaker 2>/dev/null || systemctl restart pacemaker 2>/dev/null || true
|
|
n2_ssh "systemctl start pacemaker 2>/dev/null || systemctl restart pacemaker 2>/dev/null || true"
|
|
sleep 2
|
|
|
|
# ── 1. Corosync quorum ────────────────────────────────────────────────────
|
|
log "Waiting for corosync quorum..."
|
|
for i in $(seq 1 30); do
|
|
if corosync-quorumtool -s 2>/dev/null | grep -q 'Quorate:.*Yes'; then
|
|
log "Quorum established"
|
|
break
|
|
fi
|
|
[[ $i -eq 30 ]] && die "corosync quorum not established after 60 s"
|
|
sleep 2
|
|
done
|
|
|
|
log "Waiting for pacemaker..."
|
|
for i in $(seq 1 30); do
|
|
if crm_mon -1 &>/dev/null; then
|
|
log "Pacemaker running"
|
|
break
|
|
fi
|
|
[[ $i -eq 30 ]] && die "pacemaker not running after 60 s"
|
|
sleep 2
|
|
done
|
|
|
|
# ── 2. DRBD initialisation ────────────────────────────────────────────────
|
|
# Put both nodes in Pacemaker standby before touching DRBD metadata.
|
|
# Without this, the OCF DRBD agent races: it sees drbdadm-down as a failure
|
|
# and immediately calls drbdadm-up again, leaving /dev/sdb busy when
|
|
# create-md / write-dev-uuid runs. On a fresh cluster with no resources
|
|
# configured this is a no-op; on a re-run it stops the race.
|
|
log "Setting both nodes to Pacemaker standby for DRBD metadata init..."
|
|
crm_standby -N "$NODE1" -v on 2>/dev/null || true
|
|
crm_standby -N "$NODE2" -v on 2>/dev/null || true
|
|
|
|
# Wait for Pacemaker to actually stop DRBD (if it was managing it).
|
|
log "Waiting for DRBD to stop under Pacemaker control..."
|
|
for i in $(seq 1 30); do
|
|
n1_role=$(drbdadm role ha-data 2>/dev/null || echo "Unconfigured")
|
|
n2_role=$(n2_ssh "drbdadm role ha-data 2>/dev/null" 2>/dev/null || echo "Unconfigured")
|
|
if [[ "$n1_role" == "Unconfigured" ]] && [[ "$n2_role" == "Unconfigured" ]]; then
|
|
log "DRBD stopped on both nodes"
|
|
break
|
|
fi
|
|
[[ $i -eq 30 ]] && warn "DRBD still active after 60s standby — forcing down anyway"
|
|
sleep 2
|
|
done
|
|
|
|
log "Detaching DRBD on $NODE1 (belt-and-suspenders after standby)..."
|
|
drbdadm down ha-data 2>/dev/null || true
|
|
log "Detaching DRBD on $NODE2..."
|
|
n2_ssh "drbdadm down ha-data 2>/dev/null || true"
|
|
sleep 2
|
|
|
|
log "Initialising DRBD metadata on $NODE1..."
|
|
# Use drbdmeta --force directly for BOTH create-md and write-dev-uuid.
|
|
# drbdadm create-md --force passes --force to drbdmeta create-md but NOT to
|
|
# the write-dev-uuid sub-call it makes internally, so write-dev-uuid fails when
|
|
# /dev/sdb is still busy (udev auto-attach, stale DRBD state, etc.) and stdin
|
|
# is not a TTY: "stdin not a TTY, not waiting for confirmation" → exit 20.
|
|
# Calling drbdmeta --force directly bypasses the exclusive-open confirmation on
|
|
# both steps without needing a TTY, regardless of whether the device is busy.
|
|
if ! drbdadm dstate ha-data 2>/dev/null | grep -q "UpToDate"; then
|
|
UUID1=$(_rand_uuid)
|
|
drbdmeta --force 0 v08 /dev/sdb internal create-md
|
|
drbdmeta --force 0 v08 /dev/sdb internal write-dev-uuid "$UUID1"
|
|
fi
|
|
|
|
log "Initialising DRBD metadata on $NODE2..."
|
|
if ! n2_ssh "drbdadm dstate ha-data 2>/dev/null | grep -q UpToDate" 2>/dev/null; then
|
|
UUID2=$(n2_ssh "cat /proc/sys/kernel/random/uuid 2>/dev/null | tr -d '-' | cut -c1-16 | tr '[:lower:]' '[:upper:]'")
|
|
# PATH on NODE2 may not include drbdmeta when run via sudo; resolve via drbdadm's directory.
|
|
n2_ssh "bash -c 'export PATH=\"\$(dirname \"\$(command -v drbdadm)\"):\$PATH\"; drbdmeta --force 0 v08 /dev/sdb internal create-md'"
|
|
n2_ssh "bash -c 'export PATH=\"\$(dirname \"\$(command -v drbdadm)\"):\$PATH\"; drbdmeta --force 0 v08 /dev/sdb internal write-dev-uuid \"${UUID2}\"'"
|
|
fi
|
|
|
|
log "Bringing up DRBD on both nodes..."
|
|
drbdadm up ha-data 2>/dev/null || true
|
|
n2_ssh "drbdadm up ha-data" 2>/dev/null || true
|
|
|
|
log "Clearing Pacemaker standby — DRBD is up, letting Pacemaker resume..."
|
|
crm_standby -N "$NODE1" -v off 2>/dev/null || true
|
|
crm_standby -N "$NODE2" -v off 2>/dev/null || true
|
|
|
|
log "Forcing $NODE1 to DRBD Primary for initial sync..."
|
|
drbdadm primary ha-data --force
|
|
|
|
log "Waiting for DRBD to finish initial sync (this may take several minutes)..."
|
|
for i in $(seq 1 300); do
|
|
state=$(drbdadm dstate ha-data 2>/dev/null || echo "unknown")
|
|
if echo "$state" | grep -q "UpToDate/UpToDate"; then
|
|
log "DRBD sync complete: $state"
|
|
break
|
|
fi
|
|
[[ $i -eq 300 ]] && warn "DRBD not UpToDate after 300 s — continuing anyway (check drbdadm status)"
|
|
sleep 1
|
|
done
|
|
|
|
# ── 3. XFS filesystem ─────────────────────────────────────────────────────
|
|
log "Creating XFS on ${DRBD_DEVICE}..."
|
|
if ! xfs_info "${DRBD_DEVICE}" &>/dev/null; then
|
|
mkfs.xfs -f "${DRBD_DEVICE}"
|
|
fi
|
|
|
|
log "Mounting ${DRBD_DEVICE} at ${XFS_MOUNT}..."
|
|
mkdir -p "${XFS_MOUNT}"
|
|
mountpoint -q "${XFS_MOUNT}" || mount "${DRBD_DEVICE}" "${XFS_MOUNT}"
|
|
|
|
# ── 4. NFS dataset directories ────────────────────────────────────────────
|
|
log "Creating NFS dataset directories..."
|
|
for subdir in "${NFS_SUBDIRS[@]}"; do
|
|
mkdir -p "${XFS_MOUNT}/${subdir}"
|
|
done
|
|
|
|
# ── 5. iSCSI LUN backing file ─────────────────────────────────────────────
|
|
log "Creating iSCSI LUN backing file ${ISCSI_LUN_FILE} (${ISCSI_LUN_SIZE})..."
|
|
if [[ ! -f "${ISCSI_LUN_FILE}" ]]; then
|
|
fallocate -l "${ISCSI_LUN_SIZE}" "${ISCSI_LUN_FILE}"
|
|
fi
|
|
|
|
# ── 6. LIO iSCSI target ───────────────────────────────────────────────────
|
|
log "Configuring LIO iSCSI target via targetcli..."
|
|
# Note: do NOT bind portal to ${VIP} here — the VIP isn't assigned yet (Pacemaker
|
|
# creates it). The default portal (all IPs, port 3260) is correct; Pacemaker's
|
|
# VIP resource will make the target reachable at the VIP address.
|
|
#
|
|
# Clear any existing LIO state first (idempotent: re-run after a partial failure).
|
|
# Use specific delete commands — clearconfig does not reliably clear kernel state.
|
|
if ls /sys/kernel/config/target/iscsi/ 2>/dev/null | grep -q "${ISCSI_IQN}"; then
|
|
log "Clearing existing LIO target ${ISCSI_IQN} before reconfiguration..."
|
|
targetcli "/iscsi delete ${ISCSI_IQN}" 2>/dev/null || true
|
|
fi
|
|
if ls /sys/kernel/config/target/core/ 2>/dev/null | grep -q "fileio"; then
|
|
log "Clearing existing LIO backstore ha-lun0 before reconfiguration..."
|
|
targetcli "/backstores/fileio delete ha-lun0" 2>/dev/null || true
|
|
fi
|
|
targetcli <<EOF
|
|
/backstores/fileio create name=ha-lun0 file_or_dev=${ISCSI_LUN_FILE} size=0 write_back=false
|
|
/iscsi create ${ISCSI_IQN}
|
|
/iscsi/${ISCSI_IQN}/tpg1/luns create /backstores/fileio/ha-lun0
|
|
/iscsi/${ISCSI_IQN}/tpg1 set attribute authentication=0
|
|
/iscsi/${ISCSI_IQN}/tpg1 set attribute demo_mode_write_protect=0
|
|
saveconfig /etc/target/saveconfig.json
|
|
EOF
|
|
|
|
log "Tearing down LIO kernel objects — Pacemaker will restore via targetctl on the Active node..."
|
|
# LIO holds the backing file open; clear kernel state now so the XFS unmount succeeds.
|
|
# Use specific delete commands (clearconfig does not reliably clear kernel configfs state).
|
|
targetcli "/iscsi delete ${ISCSI_IQN}" 2>/dev/null || warn "LIO iscsi delete failed — umount may fail"
|
|
targetcli "/backstores/fileio delete ha-lun0" 2>/dev/null || warn "LIO backstore delete failed"
|
|
|
|
log "Distributing iSCSI saveconfig to $NODE2..."
|
|
n2_scp /etc/target/saveconfig.json /etc/target/saveconfig.json
|
|
|
|
|
|
log "Unmounting ${XFS_MOUNT} — Pacemaker manages it..."
|
|
umount "${XFS_MOUNT}" || { sync; umount -l "${XFS_MOUNT}"; }
|
|
|
|
log "Demoting DRBD to Secondary — Pacemaker manages primary role..."
|
|
drbdadm role ha-data 2>/dev/null | grep -q "^Primary" && drbdadm secondary ha-data || true
|
|
|
|
# ── 7. Pacemaker resources ────────────────────────────────────────────────
|
|
log "Configuring Pacemaker cluster properties..."
|
|
crm_attribute -t crm_config -n stonith-enabled -v false
|
|
crm_attribute -t crm_config -n no-quorum-policy -v ignore
|
|
|
|
log "Creating Pacemaker resources via cibadmin..."
|
|
# Use cibadmin --replace with pacemaker-4.0-compatible XML.
|
|
# Key schema rules for pacemaker-4.0:
|
|
# - globally-unique must be in <meta_attributes>, not a direct <clone> attribute
|
|
# - promoted-max / promoted-node-max (not master-max / master-node-max)
|
|
# - constraint with-rsc-role="Promoted" (not "Master")
|
|
cibadmin --replace --scope resources --xml-text '<resources>
|
|
<clone id="ms-drbd0">
|
|
<meta_attributes id="ms-drbd0-meta">
|
|
<nvpair id="ms-drbd0-globally-unique" name="globally-unique" value="false"/>
|
|
<nvpair id="ms-drbd0-promotable" name="promotable" value="true"/>
|
|
<nvpair id="ms-drbd0-promoted-max" name="promoted-max" value="1"/>
|
|
<nvpair id="ms-drbd0-promoted-node-max" name="promoted-node-max" value="1"/>
|
|
<nvpair id="ms-drbd0-clone-max" name="clone-max" value="2"/>
|
|
<nvpair id="ms-drbd0-clone-node-max" name="clone-node-max" value="1"/>
|
|
<nvpair id="ms-drbd0-notify" name="notify" value="true"/>
|
|
<nvpair id="ms-drbd0-interleave" name="interleave" value="true"/>
|
|
</meta_attributes>
|
|
<primitive id="drbd0" class="ocf" type="drbd" provider="linbit">
|
|
<instance_attributes id="drbd0-attrs">
|
|
<nvpair id="drbd0-resource" name="drbd_resource" value="ha-data"/>
|
|
</instance_attributes>
|
|
<operations>
|
|
<op id="drbd0-start" name="start" interval="0" timeout="240s"/>
|
|
<op id="drbd0-stop" name="stop" interval="0" timeout="120s"/>
|
|
<op id="drbd0-promote" name="promote" interval="0" timeout="240s"/>
|
|
<op id="drbd0-demote" name="demote" interval="0" timeout="90s"/>
|
|
<op id="drbd0-monitor-promoted" name="monitor" interval="20s" timeout="20s" role="Promoted"/>
|
|
<op id="drbd0-monitor-unpromoted" name="monitor" interval="30s" timeout="20s" role="Unpromoted"/>
|
|
</operations>
|
|
</primitive>
|
|
</clone>
|
|
<group id="ha-group">
|
|
<primitive id="xfs-data" class="ocf" type="Filesystem" provider="heartbeat">
|
|
<instance_attributes id="xfs-data-attrs">
|
|
<nvpair id="xfs-data-device" name="device" value="/dev/drbd0"/>
|
|
<nvpair id="xfs-data-directory" name="directory" value="/srv/ha-data"/>
|
|
<nvpair id="xfs-data-fstype" name="fstype" value="xfs"/>
|
|
<nvpair id="xfs-data-options" name="options" value="defaults"/>
|
|
<nvpair id="xfs-data-force_unmount" name="force_unmount" value="true"/>
|
|
</instance_attributes>
|
|
<operations>
|
|
<op id="xfs-data-start" name="start" interval="0" timeout="60s"/>
|
|
<op id="xfs-data-stop" name="stop" interval="0" timeout="60s"/>
|
|
<op id="xfs-data-monitor" name="monitor" interval="20s" timeout="40s"/>
|
|
</operations>
|
|
</primitive>
|
|
<primitive id="iscsi-target" class="systemd" type="targetctl">
|
|
<operations>
|
|
<op id="iscsi-start" name="start" interval="0" timeout="60s"/>
|
|
<op id="iscsi-stop" name="stop" interval="0" timeout="60s"/>
|
|
<op id="iscsi-monitor" name="monitor" interval="20s" timeout="40s"/>
|
|
</operations>
|
|
</primitive>
|
|
<primitive id="nfs-server" class="systemd" type="nfs-server">
|
|
<operations>
|
|
<op id="nfs-start" name="start" interval="0" timeout="60s"/>
|
|
<op id="nfs-stop" name="stop" interval="0" timeout="60s"/>
|
|
<op id="nfs-monitor" name="monitor" interval="30s" timeout="40s"/>
|
|
</operations>
|
|
</primitive>
|
|
<primitive id="vip" class="ocf" type="IPaddr2" provider="heartbeat">
|
|
<instance_attributes id="vip-attrs">
|
|
<nvpair id="vip-ip" name="ip" value="192.168.2.229"/>
|
|
<nvpair id="vip-cidr" name="cidr_netmask" value="24"/>
|
|
</instance_attributes>
|
|
<operations>
|
|
<op id="vip-start" name="start" interval="0" timeout="20s"/>
|
|
<op id="vip-stop" name="stop" interval="0" timeout="20s"/>
|
|
<op id="vip-monitor" name="monitor" interval="10s" timeout="20s"/>
|
|
</operations>
|
|
</primitive>
|
|
</group>
|
|
</resources>'
|
|
|
|
log "Adding Pacemaker ordering and colocation constraints..."
|
|
cibadmin --replace --scope constraints --xml-text '<constraints>
|
|
<rsc_order id="order-drbd-group" first="ms-drbd0" first-action="promote" then="ha-group" then-action="start" kind="Mandatory"/>
|
|
<rsc_colocation id="coloc-group-with-drbd" score="INFINITY" rsc="ha-group" with-rsc="ms-drbd0" with-rsc-role="Promoted"/>
|
|
</constraints>'
|
|
|
|
log "Waiting for resources to start..."
|
|
for i in $(seq 1 60); do
|
|
if crm_resource -r vip --locate 2>/dev/null | grep -q "running on"; then
|
|
log "VIP is up: $(crm_resource -r vip --locate)"
|
|
break
|
|
fi
|
|
[[ $i -eq 60 ]] && { warn "VIP not up after 120 s — check: crm_mon -1"; break; }
|
|
sleep 2
|
|
done
|
|
|
|
log ""
|
|
log "═══════════════════════════════════════════════════════════════"
|
|
log " HA cluster initialised."
|
|
log ""
|
|
log " crm_mon -1 — cluster status"
|
|
log " iscsiadm -m discovery -t st -p ${VIP} — verify iSCSI target"
|
|
log " showmount -e ${VIP} — verify NFS exports"
|
|
log ""
|
|
log " To enable STONITH (after deploying fence SSH key):"
|
|
log " 1. Fill in VMID_NODE1 / VMID_NODE2 in cluster-enable-stonith.sh"
|
|
log " 2. Copy scripts/ha/fence-pve-ssh.py to /etc/pacemaker/fence_pve_ssh"
|
|
log " on both nodes (chmod +x)"
|
|
log " 3. Generate and distribute the fence SSH key"
|
|
log " (see docs or cluster-enable-stonith.sh header)"
|
|
log " 4. bash scripts/ha/cluster-enable-stonith.sh"
|
|
log "═══════════════════════════════════════════════════════════════"
|