Archived
Root SSH was failing because only the RSA admin key was authorized but the local dev box only has an ed25519 key. Fix: - cluster-config.nix: add ed25519 keys to root (same set as nixos user) so future deployments work without the temp-key workaround - deploy.sh/acceptance-tests.sh: SSH as nixos user with sudo instead of root@ - cluster-init.sh: HA_USER/HA_KEY env vars + n2_ssh()/n2_scp() helpers so inter-node SSH works regardless of whether root-to-root is available - deploy.sh Phase 6: generate temp keypair, authorize on node2, place on node1 for root to use during cluster-init, clean up afterward Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01HaH1cSGvhogRP5ExoF6nD8
304 lines
14 KiB
Bash
Executable File
304 lines
14 KiB
Bash
Executable File
#!/usr/bin/env bash
|
|
# cluster-init.sh — one-time HA cluster initialisation script
|
|
#
|
|
# Run ONCE from ha-server-1 as root AFTER both VMs are booted and have SSH
|
|
# access. It:
|
|
# 1. Generates and distributes the corosync authkey
|
|
# 2. Waits for corosync quorum and pacemaker
|
|
# 3. Initialises DRBD metadata, promotes node1 to primary
|
|
# 4. Creates XFS on /dev/drbd0 and mounts it
|
|
# 5. Creates the directory tree and iSCSI LUN backing file
|
|
# 6. Configures LIO iSCSI target (file-backed LUN)
|
|
# 7. Configures Pacemaker resources: DRBD → XFS → iSCSI → NFS → VIP
|
|
#
|
|
# Prerequisites:
|
|
# - Both VMs booted with the ha-server config (nixos-rebuild done)
|
|
# - SSH key access from node1 to root@NODE2_IP
|
|
# - VMID_NODE1 / VMID_NODE2 filled in below (needed for STONITH setup;
|
|
# cluster starts without STONITH, which you enable separately via
|
|
# scripts/ha/cluster-enable-stonith.sh)
|
|
# - Run as root on ha-server-1
|
|
set -euo pipefail
|
|
|
|
# ── Configuration ─────────────────────────────────────────────────────────
|
|
# All values override-able via environment variables; defaults match variables.nix.
|
|
NODE1="${NODE1:-ha-server-1}"
|
|
NODE2="${NODE2:-ha-server-2}"
|
|
NODE1_IP="${NODE1_IP:-192.168.2.228}" # vars.haServer1Ip
|
|
NODE2_IP="${NODE2_IP:-192.168.2.227}" # vars.haServer2Ip
|
|
VIP="${VIP:-192.168.2.229}" # vars.haServerVip
|
|
XFS_MOUNT="${XFS_MOUNT:-/srv/ha-data}" # vars.haStorageRoot
|
|
ISCSI_IQN="${ISCSI_IQN:-iqn.2026-01.home.sweet:ha-storage}" # vars.haIscsiIqn
|
|
ISCSI_LUN_FILE="${XFS_MOUNT}/iscsi-lun.img"
|
|
ISCSI_LUN_SIZE="10G"
|
|
DRBD_DEVICE="/dev/drbd0"
|
|
VMID_NODE1="${VMID_NODE1:-}" # set by deploy.sh; needed for STONITH
|
|
VMID_NODE2="${VMID_NODE2:-}"
|
|
PVE_HOST="${PVE_HOST:-pve1.sweet.home}"
|
|
PVE_USER="${PVE_USER:-wayne}"
|
|
# Inter-node SSH: HA_USER is the user to SSH as on NODE2; HA_KEY is the private
|
|
# key to use. Default is root-to-root (no key arg). deploy.sh sets HA_USER=nixos
|
|
# and HA_KEY=/tmp/cluster-init-key so the script works even when root-to-root SSH
|
|
# is not available.
|
|
HA_USER="${HA_USER:-root}"
|
|
HA_KEY="${HA_KEY:-}"
|
|
|
|
# NFS dataset subdirectories to create under XFS_MOUNT.
|
|
# Must mirror vars.nfsShares subpath values in variables.nix.
|
|
NFS_SUBDIRS=(
|
|
"docker/config"
|
|
"docker/volumes"
|
|
"docker/databases"
|
|
"docker/nextcloud-data"
|
|
"raspi/volumes"
|
|
"proxmox/iso"
|
|
"proxmox/lxc"
|
|
"pxe-boot/images"
|
|
)
|
|
# ──────────────────────────────────────────────────────────────────────────
|
|
|
|
log() { echo "[cluster-init] $*"; }
|
|
die() { echo "[cluster-init] ERROR: $*" >&2; exit 1; }
|
|
warn() { echo "[cluster-init] WARNING: $*" >&2; }
|
|
|
|
[[ $(id -u) -eq 0 ]] || die "must run as root"
|
|
[[ "$(hostname)" == "$NODE1" ]] || die "must run on $NODE1"
|
|
|
|
# Inter-node SSH/SCP helpers — abstract over root-to-root vs nixos+sudo.
|
|
_SSH_OPTS="-o StrictHostKeyChecking=no -o ConnectTimeout=10"
|
|
[[ -n "$HA_KEY" ]] && _SSH_OPTS="-i $HA_KEY $_SSH_OPTS"
|
|
if [[ "$HA_USER" == "root" ]]; then
|
|
n2_ssh() { ssh $_SSH_OPTS "root@${NODE2_IP}" "$@"; }
|
|
n2_scp() { scp $_SSH_OPTS "$1" "root@${NODE2_IP}:$2"; }
|
|
else
|
|
# Non-root user with passwordless sudo; wrap each command with sudo.
|
|
n2_ssh() { ssh $_SSH_OPTS "${HA_USER}@${NODE2_IP}" sudo "$@"; }
|
|
n2_scp() {
|
|
# SCP to a tmp path, then sudo-move to the real destination as the remote user.
|
|
local src="$1" dst="$2"
|
|
local tmp="/tmp/_cluster_init_scp_$$"
|
|
scp $_SSH_OPTS "$src" "${HA_USER}@${NODE2_IP}:${tmp}"
|
|
ssh $_SSH_OPTS "${HA_USER}@${NODE2_IP}" sudo mv "${tmp}" "${dst}"
|
|
}
|
|
fi
|
|
|
|
# ── 0. Corosync authkey ───────────────────────────────────────────────────
|
|
AUTHKEY="/etc/corosync/authkey"
|
|
mkdir -p /etc/corosync
|
|
if [[ ! -f "$AUTHKEY" ]]; then
|
|
log "Generating corosync authkey..."
|
|
corosync-keygen -k "$AUTHKEY"
|
|
chmod 0400 "$AUTHKEY"
|
|
fi
|
|
log "Distributing authkey to $NODE2..."
|
|
n2_ssh "mkdir -p /etc/corosync"
|
|
n2_scp "$AUTHKEY" "$AUTHKEY"
|
|
n2_ssh "chmod 0400 '${AUTHKEY}'"
|
|
|
|
log "Restarting corosync on both nodes..."
|
|
systemctl restart corosync
|
|
n2_ssh "systemctl restart corosync"
|
|
sleep 3
|
|
|
|
# ── 1. Corosync quorum ────────────────────────────────────────────────────
|
|
log "Waiting for corosync quorum..."
|
|
for i in $(seq 1 30); do
|
|
if corosync-quorumtool -s 2>/dev/null | grep -q 'Quorate:.*Yes'; then
|
|
log "Quorum established"
|
|
break
|
|
fi
|
|
[[ $i -eq 30 ]] && die "corosync quorum not established after 60 s"
|
|
sleep 2
|
|
done
|
|
|
|
log "Waiting for pacemaker..."
|
|
for i in $(seq 1 30); do
|
|
if crm_mon -1 &>/dev/null; then
|
|
log "Pacemaker running"
|
|
break
|
|
fi
|
|
[[ $i -eq 30 ]] && die "pacemaker not running after 60 s"
|
|
sleep 2
|
|
done
|
|
|
|
# ── 2. DRBD initialisation ────────────────────────────────────────────────
|
|
log "Initialising DRBD metadata on $NODE1..."
|
|
if ! drbdadm dstate ha-data 2>/dev/null | grep -q "UpToDate\|Inconsistent\|Diskless"; then
|
|
drbdadm create-md ha-data --force
|
|
fi
|
|
|
|
log "Initialising DRBD metadata on $NODE2..."
|
|
n2_ssh "bash -c 'if ! drbdadm dstate ha-data 2>/dev/null | grep -q UpToDate.Inconsistent.Diskless; then drbdadm create-md ha-data --force; fi'"
|
|
|
|
log "Bringing up DRBD on both nodes..."
|
|
drbdadm up ha-data 2>/dev/null || true
|
|
n2_ssh "drbdadm up ha-data" 2>/dev/null || true
|
|
|
|
log "Forcing $NODE1 to DRBD Primary for initial sync..."
|
|
drbdadm primary ha-data --force
|
|
|
|
log "Waiting for DRBD to finish initial sync (this may take several minutes)..."
|
|
for i in $(seq 1 300); do
|
|
state=$(drbdadm dstate ha-data 2>/dev/null || echo "unknown")
|
|
if echo "$state" | grep -q "UpToDate/UpToDate"; then
|
|
log "DRBD sync complete: $state"
|
|
break
|
|
fi
|
|
[[ $i -eq 300 ]] && warn "DRBD not UpToDate after 300 s — continuing anyway (check drbdadm status)"
|
|
sleep 1
|
|
done
|
|
|
|
# ── 3. XFS filesystem ─────────────────────────────────────────────────────
|
|
log "Creating XFS on ${DRBD_DEVICE}..."
|
|
if ! xfs_info "${DRBD_DEVICE}" &>/dev/null; then
|
|
mkfs.xfs -f "${DRBD_DEVICE}"
|
|
fi
|
|
|
|
log "Mounting ${DRBD_DEVICE} at ${XFS_MOUNT}..."
|
|
mkdir -p "${XFS_MOUNT}"
|
|
mount "${DRBD_DEVICE}" "${XFS_MOUNT}"
|
|
|
|
# ── 4. NFS dataset directories ────────────────────────────────────────────
|
|
log "Creating NFS dataset directories..."
|
|
for subdir in "${NFS_SUBDIRS[@]}"; do
|
|
mkdir -p "${XFS_MOUNT}/${subdir}"
|
|
done
|
|
|
|
# ── 5. iSCSI LUN backing file ─────────────────────────────────────────────
|
|
log "Creating iSCSI LUN backing file ${ISCSI_LUN_FILE} (${ISCSI_LUN_SIZE})..."
|
|
if [[ ! -f "${ISCSI_LUN_FILE}" ]]; then
|
|
fallocate -l "${ISCSI_LUN_SIZE}" "${ISCSI_LUN_FILE}"
|
|
fi
|
|
|
|
# ── 6. LIO iSCSI target ───────────────────────────────────────────────────
|
|
log "Configuring LIO iSCSI target via targetcli..."
|
|
targetcli <<EOF
|
|
/backstores/fileio create name=ha-lun0 file_or_dev=${ISCSI_LUN_FILE} size=0 write_back=false
|
|
/iscsi create ${ISCSI_IQN}
|
|
/iscsi/${ISCSI_IQN}/tpg1/luns create /backstores/fileio/ha-lun0
|
|
/iscsi/${ISCSI_IQN}/tpg1/portals create ${VIP}
|
|
/iscsi/${ISCSI_IQN}/tpg1 set attribute authentication=0
|
|
/iscsi/${ISCSI_IQN}/tpg1 set attribute demo_mode_write_protect=0
|
|
saveconfig /etc/target/saveconfig.json
|
|
EOF
|
|
|
|
log "Distributing iSCSI saveconfig to $NODE2..."
|
|
n2_scp /etc/target/saveconfig.json /etc/target/saveconfig.json
|
|
|
|
log "Unmounting ${XFS_MOUNT} — Pacemaker manages it..."
|
|
umount "${XFS_MOUNT}"
|
|
|
|
log "Demoting DRBD to Secondary — Pacemaker manages primary role..."
|
|
drbdadm secondary ha-data
|
|
|
|
# ── 7. Pacemaker resources ────────────────────────────────────────────────
|
|
log "Configuring Pacemaker cluster properties..."
|
|
crm_attribute -t crm_config -n stonith-enabled -v false
|
|
crm_attribute -t crm_config -n no-quorum-policy -v ignore
|
|
|
|
log "Creating DRBD promotable clone resource..."
|
|
cibadmin --replace --scope resources --xml-text "
|
|
<resources>
|
|
<clone id=\"ms-drbd0\" globally-unique=\"false\">
|
|
<meta_attributes id=\"ms-drbd0-meta\">
|
|
<nvpair id=\"ms-drbd0-promotable\" name=\"promotable\" value=\"true\"/>
|
|
<nvpair id=\"ms-drbd0-master-max\" name=\"master-max\" value=\"1\"/>
|
|
<nvpair id=\"ms-drbd0-master-node-max\" name=\"master-node-max\" value=\"1\"/>
|
|
<nvpair id=\"ms-drbd0-clone-max\" name=\"clone-max\" value=\"2\"/>
|
|
<nvpair id=\"ms-drbd0-clone-node-max\" name=\"clone-node-max\" value=\"1\"/>
|
|
<nvpair id=\"ms-drbd0-notify\" name=\"notify\" value=\"true\"/>
|
|
<nvpair id=\"ms-drbd0-interleave\" name=\"interleave\" value=\"true\"/>
|
|
</meta_attributes>
|
|
<primitive id=\"drbd0\" class=\"ocf\" type=\"drbd\" provider=\"linbit\">
|
|
<instance_attributes id=\"drbd0-attrs\">
|
|
<nvpair id=\"drbd0-resource\" name=\"drbd_resource\" value=\"ha-data\"/>
|
|
</instance_attributes>
|
|
<operations>
|
|
<op id=\"drbd0-start\" name=\"start\" interval=\"0\" timeout=\"240s\"/>
|
|
<op id=\"drbd0-stop\" name=\"stop\" interval=\"0\" timeout=\"120s\"/>
|
|
<op id=\"drbd0-promote\" name=\"promote\" interval=\"0\" timeout=\"90s\"/>
|
|
<op id=\"drbd0-demote\" name=\"demote\" interval=\"0\" timeout=\"90s\"/>
|
|
<op id=\"drbd0-monitor-master\" name=\"monitor\" interval=\"20s\" timeout=\"20s\" role=\"Promoted\"/>
|
|
<op id=\"drbd0-monitor-slave\" name=\"monitor\" interval=\"30s\" timeout=\"20s\" role=\"Unpromoted\"/>
|
|
</operations>
|
|
</primitive>
|
|
</clone>
|
|
<group id=\"ha-group\">
|
|
<primitive id=\"xfs-data\" class=\"ocf\" type=\"Filesystem\" provider=\"heartbeat\">
|
|
<instance_attributes id=\"xfs-data-attrs\">
|
|
<nvpair id=\"xfs-data-device\" name=\"device\" value=\"${DRBD_DEVICE}\"/>
|
|
<nvpair id=\"xfs-data-directory\" name=\"directory\" value=\"${XFS_MOUNT}\"/>
|
|
<nvpair id=\"xfs-data-fstype\" name=\"fstype\" value=\"xfs\"/>
|
|
<nvpair id=\"xfs-data-options\" name=\"options\" value=\"defaults\"/>
|
|
<nvpair id=\"xfs-data-force_unmount\" name=\"force_unmount\" value=\"false\"/>
|
|
</instance_attributes>
|
|
<operations>
|
|
<op id=\"xfs-data-start\" name=\"start\" interval=\"0\" timeout=\"60s\"/>
|
|
<op id=\"xfs-data-stop\" name=\"stop\" interval=\"0\" timeout=\"60s\"/>
|
|
<op id=\"xfs-data-monitor\" name=\"monitor\" interval=\"20s\" timeout=\"40s\"/>
|
|
</operations>
|
|
</primitive>
|
|
<primitive id=\"iscsi-target\" class=\"systemd\" type=\"targetctl\">
|
|
<operations>
|
|
<op id=\"iscsi-start\" name=\"start\" interval=\"0\" timeout=\"60s\"/>
|
|
<op id=\"iscsi-stop\" name=\"stop\" interval=\"0\" timeout=\"60s\"/>
|
|
<op id=\"iscsi-monitor\" name=\"monitor\" interval=\"20s\" timeout=\"40s\"/>
|
|
</operations>
|
|
</primitive>
|
|
<primitive id=\"nfs-server\" class=\"systemd\" type=\"nfs-server\">
|
|
<operations>
|
|
<op id=\"nfs-start\" name=\"start\" interval=\"0\" timeout=\"60s\"/>
|
|
<op id=\"nfs-stop\" name=\"stop\" interval=\"0\" timeout=\"60s\"/>
|
|
<op id=\"nfs-monitor\" name=\"monitor\" interval=\"30s\" timeout=\"40s\"/>
|
|
</operations>
|
|
</primitive>
|
|
<primitive id=\"vip\" class=\"ocf\" type=\"IPaddr2\" provider=\"heartbeat\">
|
|
<instance_attributes id=\"vip-attrs\">
|
|
<nvpair id=\"vip-ip\" name=\"ip\" value=\"${VIP}\"/>
|
|
<nvpair id=\"vip-cidr\" name=\"cidr_netmask\" value=\"24\"/>
|
|
</instance_attributes>
|
|
<operations>
|
|
<op id=\"vip-start\" name=\"start\" interval=\"0\" timeout=\"20s\"/>
|
|
<op id=\"vip-stop\" name=\"stop\" interval=\"0\" timeout=\"20s\"/>
|
|
<op id=\"vip-monitor\" name=\"monitor\" interval=\"10s\" timeout=\"20s\"/>
|
|
</operations>
|
|
</primitive>
|
|
</group>
|
|
</resources>
|
|
"
|
|
|
|
log "Adding ordering and colocation constraints..."
|
|
cibadmin --create --scope constraints --xml-text "
|
|
<constraints>
|
|
<rsc_order id=\"order-drbd-group\" first=\"ms-drbd0\" first-action=\"promote\" then=\"ha-group\" then-action=\"start\"/>
|
|
<rsc_colocation id=\"coloc-group-with-drbd\" rsc=\"ha-group\" with-rsc=\"ms-drbd0\" with-rsc-role=\"Master\" score=\"INFINITY\"/>
|
|
</constraints>
|
|
"
|
|
|
|
log "Waiting for resources to start..."
|
|
for i in $(seq 1 60); do
|
|
if crm_resource -r vip --locate 2>/dev/null | grep -q "running on"; then
|
|
log "VIP is up: $(crm_resource -r vip --locate)"
|
|
break
|
|
fi
|
|
[[ $i -eq 60 ]] && { warn "VIP not up after 120 s — check: crm_mon -1"; break; }
|
|
sleep 2
|
|
done
|
|
|
|
log ""
|
|
log "═══════════════════════════════════════════════════════════════"
|
|
log " HA cluster initialised."
|
|
log ""
|
|
log " crm_mon -1 — cluster status"
|
|
log " iscsiadm -m discovery -t st -p ${VIP} — verify iSCSI target"
|
|
log " showmount -e ${VIP} — verify NFS exports"
|
|
log ""
|
|
log " To enable STONITH (after deploying fence SSH key):"
|
|
log " 1. Fill in VMID_NODE1 / VMID_NODE2 in cluster-enable-stonith.sh"
|
|
log " 2. Copy scripts/ha/fence-pve-ssh.py to /etc/pacemaker/fence_pve_ssh"
|
|
log " on both nodes (chmod +x)"
|
|
log " 3. Generate and distribute the fence SSH key"
|
|
log " (see docs or cluster-enable-stonith.sh header)"
|
|
log " 4. bash scripts/ha/cluster-enable-stonith.sh"
|
|
log "═══════════════════════════════════════════════════════════════"
|