Archived
Adds proxmox-ha-server-1 and proxmox-ha-server-2 as real mkTarget entries
alongside the existing proxmox-server, backed by a new ha-server build type.
New modules
modules/ha/cluster-config.nix — DRBD resource + corosync nodelist sourced
from vars (haServer1Host/Ip, haServer2Host/Ip); resource-only fencing for
production STONITH; HA port firewall rules for DRBD, iSCSI, Corosync, pcsd
modules/build-types/ha-server.nix — imports pacemaker-stack + iscsi-target
+ cluster-config + beszel; NFS exports from vars.haStorageRoot (XFS-over-DRBD
mount); nfs-server.service.wantedBy force-cleared so Pacemaker controls
start/stop on the Active node only
New hosts
hosts/ha-server-{1,2}/host.nix — static IP from vars, unique hostId; sops
secrets (beszel, corosync authkey) are TODOs pending sync-host-keys.sh
variables.nix
haServer1/2Host, haServer1/2Ip, haServerVip, haStorageRoot, haIscsiIqn
ports.haServerDrbd/Iscsi/Corosync{1,2,Crypto}/PacemakerRemoted/Pcsd
scripts/ha/ (migrated + updated from test-lab/ha/)
cluster-init.sh — generates corosync authkey, initialises DRBD/XFS/iSCSI,
creates NFS dataset dirs, configures Pacemaker with DRBD + XFS + iSCSI
+ nfs-server + VIP; STONITH disabled initially (enable separately)
cluster-enable-stonith.sh — enables fence_pve_ssh STONITH after key deploy
fence-pve-ssh.py — Proxmox SSH fence agent (node names updated to ha-server-1/2)
acceptance-tests.sh — T1–T7 production acceptance tests
test-lab/ha/ removed — all Nix config moved to modules/ha/ and
modules/build-types/; scripts moved to scripts/ha/
Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01HaH1cSGvhogRP5ExoF6nD8
168 lines
7.0 KiB
Bash
168 lines
7.0 KiB
Bash
#!/usr/bin/env bash
|
||
# acceptance-tests.sh — HA cluster acceptance tests (T1–T7)
|
||
#
|
||
# Run from a host with SSH access to both HA nodes (or from node1 itself).
|
||
# All 7 tests must pass before considering the cluster production-ready.
|
||
# Test values below must match variables.nix haServer* values.
|
||
set -euo pipefail
|
||
|
||
# ── Configuration ─────────────────────────────────────────────────────────
|
||
NODE1="ha-server-1"
|
||
NODE2="ha-server-2"
|
||
NODE1_IP="192.168.2.200" # vars.haServer1Ip
|
||
NODE2_IP="192.168.2.201" # vars.haServer2Ip
|
||
VIP="192.168.2.202" # vars.haServerVip
|
||
XFS_MOUNT="/srv/ha-data" # vars.haStorageRoot
|
||
ISCSI_IQN="iqn.2026-01.home.sweet:ha-storage" # vars.haIscsiIqn
|
||
# ──────────────────────────────────────────────────────────────────────────
|
||
|
||
PASS=0
|
||
FAIL=0
|
||
RESULTS=()
|
||
|
||
pass() { echo " PASS: $1"; ((PASS++)); RESULTS+=("PASS $1"); }
|
||
fail() { echo " FAIL: $1"; ((FAIL++)); RESULTS+=("FAIL $1"); }
|
||
|
||
n1() { ssh -o StrictHostKeyChecking=no -o ConnectTimeout=5 "root@${NODE1_IP}" "$@" 2>/dev/null; }
|
||
n2() { ssh -o StrictHostKeyChecking=no -o ConnectTimeout=5 "root@${NODE2_IP}" "$@" 2>/dev/null; }
|
||
|
||
echo "════════════════════════════════════════════════════"
|
||
echo " HA Cluster Acceptance Tests — $(date '+%Y-%m-%d %H:%M:%S')"
|
||
echo "════════════════════════════════════════════════════"
|
||
|
||
# ── T1: Corosync quorum established ──────────────────────────────────────
|
||
echo ""
|
||
echo "[T1] Corosync quorum"
|
||
if n1 "corosync-quorumtool -s" 2>/dev/null | grep -q "Quorate:.*Yes"; then
|
||
pass "cluster has quorum"
|
||
else
|
||
fail "cluster does not have quorum — check corosync on both nodes"
|
||
fi
|
||
|
||
# ── T2: DRBD Primary on node1, Secondary on node2 ────────────────────────
|
||
echo ""
|
||
echo "[T2] DRBD roles"
|
||
DRBD_ROLE=$(n1 "drbdadm role ha-data" 2>/dev/null || echo "unknown")
|
||
if [[ "$DRBD_ROLE" == "Primary/Secondary" || "$DRBD_ROLE" == "Primary" ]]; then
|
||
pass "DRBD Primary on $NODE1 ($DRBD_ROLE)"
|
||
else
|
||
fail "unexpected DRBD role on $NODE1: $DRBD_ROLE (expected Primary/Secondary)"
|
||
fi
|
||
|
||
DRBD_DSTATE=$(n1 "drbdadm dstate ha-data" 2>/dev/null || echo "unknown")
|
||
if echo "$DRBD_DSTATE" | grep -q "UpToDate"; then
|
||
pass "DRBD disk state UpToDate ($DRBD_DSTATE)"
|
||
else
|
||
fail "DRBD disk not UpToDate: $DRBD_DSTATE"
|
||
fi
|
||
|
||
# ── T3: XFS mounted at haStorageRoot on the Active node ──────────────────
|
||
echo ""
|
||
echo "[T3] XFS mount"
|
||
if n1 "mountpoint -q '${XFS_MOUNT}'" 2>/dev/null; then
|
||
pass "XFS mounted at ${XFS_MOUNT} on $NODE1"
|
||
else
|
||
fail "XFS not mounted at ${XFS_MOUNT} on $NODE1"
|
||
fi
|
||
|
||
if n2 "mountpoint -q '${XFS_MOUNT}'" 2>/dev/null; then
|
||
fail "XFS unexpectedly mounted on $NODE2 (should only be on Active node)"
|
||
else
|
||
pass "XFS not mounted on $NODE2 (correct — Secondary)"
|
||
fi
|
||
|
||
# ── T4: iSCSI target visible on both nodes ────────────────────────────────
|
||
echo ""
|
||
echo "[T4] iSCSI target"
|
||
IQN_COUNT=$(n1 "ls /sys/kernel/config/target/iscsi/ 2>/dev/null | grep -c iqn" || echo "0")
|
||
if [[ "$IQN_COUNT" -ge 1 ]]; then
|
||
pass "iSCSI IQN active on $NODE1 ($IQN_COUNT target(s))"
|
||
else
|
||
fail "no iSCSI IQN active on $NODE1"
|
||
fi
|
||
|
||
# iSCSI discovery from node2 via VIP
|
||
if n2 "iscsiadm -m discovery -t sendtargets -p '${VIP}' 2>/dev/null | grep -q '${ISCSI_IQN}'"; then
|
||
pass "iSCSI target discoverable from $NODE2 via VIP ${VIP}"
|
||
else
|
||
fail "iSCSI target not discoverable from $NODE2 via ${VIP}"
|
||
fi
|
||
|
||
# ── T5: Failover — standby node1, verify resources move to node2 ──────────
|
||
echo ""
|
||
echo "[T5] Failover (standby $NODE1)"
|
||
MYNODE=$(n1 "crm_node -n" 2>/dev/null || echo "")
|
||
n1 "crm_standby -N '${MYNODE}' -v on" 2>/dev/null || true
|
||
echo " Waiting up to 30 s for resources to move to $NODE2..."
|
||
MOVED=false
|
||
for i in $(seq 1 30); do
|
||
if n2 "mountpoint -q '${XFS_MOUNT}'" 2>/dev/null; then
|
||
MOVED=true
|
||
echo " Resources moved in ${i}s"
|
||
break
|
||
fi
|
||
sleep 1
|
||
done
|
||
|
||
if $MOVED; then
|
||
pass "XFS mounted on $NODE2 after failover"
|
||
IQN_ON_N2=$(n2 "ls /sys/kernel/config/target/iscsi/ 2>/dev/null | grep -c iqn" || echo "0")
|
||
[[ "$IQN_ON_N2" -ge 1 ]] \
|
||
&& pass "iSCSI target active on $NODE2 after failover" \
|
||
|| fail "iSCSI target NOT active on $NODE2 after failover"
|
||
else
|
||
fail "XFS did not mount on $NODE2 within 30 s — failover incomplete"
|
||
fi
|
||
|
||
# ── T6: Data integrity — file written pre-failover readable post-failover ─
|
||
echo ""
|
||
echo "[T6] Data integrity"
|
||
# Write a test file on node2 (now Active) and verify its content
|
||
TEST_FILE="${XFS_MOUNT}/.acceptance-test-$$"
|
||
TEST_CONTENT="ha-acceptance-test-$(date +%s)"
|
||
n2 "echo '${TEST_CONTENT}' > '${TEST_FILE}'" 2>/dev/null || true
|
||
READBACK=$(n2 "cat '${TEST_FILE}' 2>/dev/null" || echo "")
|
||
if [[ "$READBACK" == "$TEST_CONTENT" ]]; then
|
||
pass "test file written and read back correctly on $NODE2"
|
||
else
|
||
fail "data integrity check failed (wrote: '$TEST_CONTENT', read: '$READBACK')"
|
||
fi
|
||
n2 "rm -f '${TEST_FILE}'" 2>/dev/null || true
|
||
|
||
# ── T7: Node rejoin — un-standby node1, verify cluster is healthy ─────────
|
||
echo ""
|
||
echo "[T7] Node rejoin"
|
||
n1 "crm_standby -N '${MYNODE}' -v off" 2>/dev/null || true
|
||
n1 "crm_resource --cleanup" 2>/dev/null || true
|
||
sleep 5
|
||
|
||
ONLINE_NODES=$(n2 "crm_mon -1 2>/dev/null | grep -c 'Online:'" || echo "0")
|
||
if n1 "corosync-quorumtool -s 2>/dev/null | grep -q 'Quorate:.*Yes'"; then
|
||
pass "$NODE1 rejoined — cluster has quorum"
|
||
else
|
||
fail "$NODE1 did not rejoin with quorum"
|
||
fi
|
||
|
||
DRBD_ROLE_AFTER=$(n1 "drbdadm role ha-data" 2>/dev/null || echo "unknown")
|
||
if echo "$DRBD_ROLE_AFTER" | grep -q "Secondary"; then
|
||
pass "$NODE1 is DRBD Secondary after rejoin ($DRBD_ROLE_AFTER)"
|
||
else
|
||
fail "unexpected DRBD role on $NODE1 after rejoin: $DRBD_ROLE_AFTER"
|
||
fi
|
||
|
||
# ── Summary ───────────────────────────────────────────────────────────────
|
||
echo ""
|
||
echo "════════════════════════════════════════════════════"
|
||
echo " Results: ${PASS} PASS, ${FAIL} FAIL"
|
||
echo "════════════════════════════════════════════════════"
|
||
for r in "${RESULTS[@]}"; do echo " $r"; done
|
||
echo ""
|
||
|
||
if [[ "$FAIL" -eq 0 ]]; then
|
||
echo "ALL PASS — cluster is production-ready."
|
||
exit 0
|
||
else
|
||
echo "SOME TESTS FAILED — investigate before deploying."
|
||
exit 1
|
||
fi
|