fix(ha): use crm_mon for Active node detection and add cleanup before VIP wait
Check NixOS configurations / eval-hosts (push) Successful in 10m26s

acceptance-tests.sh: replace drbdadm-role-based Active node detection with
crm_mon-based detection. DRBD role briefly reads Secondary during Pacemaker
transitions, causing T2/T3/T4 to target the wrong node. crm_mon Promoted:/
Masters: lines are authoritative; wait up to 90 s for Pacemaker to settle
before falling back to NODE1.

cluster-init.sh: add crm_resource --cleanup before the VIP wait loop so
stale migration-threshold failure counters from previous cluster-init runs
are cleared before Pacemaker decides resource placement.

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
2026-07-29 00:24:47 +10:00
co-authored by Claude Sonnet 4.6
parent 49a10d7cc5
commit e7531b276e
2 changed files with 26 additions and 6 deletions
+23 -6
View File
@@ -35,17 +35,34 @@ echo " HA Cluster Acceptance Tests — $(date '+%Y-%m-%d %H:%M:%S')"
echo "════════════════════════════════════════════════════"
# ── Detect Active/Standby nodes ────────────────────────────────────────────
# Pacemaker can promote either node; determine which is currently Active
# (DRBD Primary / holds ha-group resources) before running tests.
# Use crm_mon to detect which node holds the Promoted (Primary) DRBD resource.
# Pacemaker is authoritative; DRBD role can briefly read as Secondary while
# Pacemaker is mid-transition, giving a false Active/Standby swap.
# Wait up to 90 s for Pacemaker to settle before giving up.
echo ""
echo "Detecting Active/Standby nodes..."
if n1 "drbdadm role ha-data" 2>/dev/null | grep -q "^Primary"; then
ACTIVE_NODE="$NODE1"; ACTIVE_IP="$NODE1_IP"
echo "Detecting Active/Standby nodes (waiting for Pacemaker to settle)..."
ACTIVE_NODE=""
for i in $(seq 1 30); do
# crm_mon -1 output contains "Promoted: [ <node> ]" for the DRBD master.
CRM_OUT=$(n1 "crm_mon -1 2>/dev/null" 2>/dev/null || n2 "crm_mon -1 2>/dev/null" 2>/dev/null || true)
# Match "Promoted:" (Pacemaker 2.x) or "Masters:" (Pacemaker 1.x), not "Unpromoted:"/"Slaves:"
ACTIVE_NODE=$(echo "$CRM_OUT" | grep -E '^\s*(Promoted|Masters):' | grep -oE '\b(ha-server-[0-9]+)\b' | head -1 || true)
[[ -n "$ACTIVE_NODE" ]] && break
sleep 3
done
if [[ -z "$ACTIVE_NODE" ]]; then
echo " WARNING: could not determine Active node from crm_mon after 90 s — defaulting to $NODE1"
ACTIVE_NODE="$NODE1"
fi
if [[ "$ACTIVE_NODE" == "$NODE1" ]]; then
ACTIVE_IP="$NODE1_IP"
STANDBY_NODE="$NODE2"; STANDBY_IP="$NODE2_IP"
na() { n1 "$@"; }
ns() { n2 "$@"; }
else
ACTIVE_NODE="$NODE2"; ACTIVE_IP="$NODE2_IP"
ACTIVE_IP="$NODE2_IP"
STANDBY_NODE="$NODE1"; STANDBY_IP="$NODE1_IP"
na() { n2 "$@"; }
ns() { n1 "$@"; }
+3
View File
@@ -399,6 +399,9 @@ cibadmin --replace --scope constraints --xml-text '<constraints>
<rsc_colocation id="coloc-group-with-drbd" score="INFINITY" rsc="ha-group" with-rsc="ms-drbd0" with-rsc-role="Promoted"/>
</constraints>'
log "Clearing stale Pacemaker failure history..."
crm_resource --cleanup 2>/dev/null || true
log "Waiting for resources to start..."
for i in $(seq 1 60); do
if crm_resource -r vip --locate 2>/dev/null | grep -q "running on"; then