diff --git a/.sops.yaml b/.sops.yaml
index b37b4e2..a14898b 100644
--- a/.sops.yaml
+++ b/.sops.yaml
@@ -85,6 +85,32 @@ creation_rules:
- *lxc-tailscale-router
- *proxmox-tailscale-router
+ # HA file server per-node secrets (beszel-token).
+ # proxmox-ha-server-1 / proxmox-ha-server-2 keys are added automatically
+ # by scripts/secrets/sync-host-keys.sh once the hosts are provisioned;
+ # until then only the admin key can decrypt these files.
+ - path_regex: secrets/ha-server-1\.yaml$
+ key_groups:
+ - age:
+ - *admin
+ # proxmox-ha-server-1 added by sync-host-keys.sh
+
+ - path_regex: secrets/ha-server-2\.yaml$
+ key_groups:
+ - age:
+ - *admin
+ # proxmox-ha-server-2 added by sync-host-keys.sh
+
+ # Shared HA cluster corosync authkey (binary sops file).
+ # Encrypted for both HA nodes so either can decrypt on boot.
+ # Both host keys added by sync-host-keys.sh; admin key allows initial creation.
+ - path_regex: secrets/ha-corosync-authkey$
+ key_groups:
+ - age:
+ - *admin
+ # proxmox-ha-server-1 added by sync-host-keys.sh
+ # proxmox-ha-server-2 added by sync-host-keys.sh
+
# gui-host-specific secrets (currently: wifi-password, see
# modules/networking/wifi.nix). Only *lxc-gui has a registered key today
# -- proxmox-gui/linode-gui/baremetal-gui haven't been provisioned via
diff --git a/flake.nix b/flake.nix
index 47ac295..d4b0aa0 100644
--- a/flake.nix
+++ b/flake.nix
@@ -130,6 +130,9 @@
lxc-tailscale-router = mkTarget { platform = "lxc"; buildType = "tailscale-router"; hostPath = ./hosts/tailscale-router/host.nix; };
lxc-tor-relay = mkTarget { platform = "lxc"; buildType = "tor-relay"; hostPath = ./hosts/tor-relay/host.nix; };
+
+ proxmox-ha-server-1 = mkTarget { platform = "proxmox"; buildType = "ha-server"; hostPath = ./hosts/ha-server-1/host.nix; };
+ proxmox-ha-server-2 = mkTarget { platform = "proxmox"; buildType = "ha-server"; hostPath = ./hosts/ha-server-2/host.nix; };
};
# Auto-install environments (migrated from the former nix-auto-installer
diff --git a/hosts/ha-server-1/host.nix b/hosts/ha-server-1/host.nix
new file mode 100644
index 0000000..c380008
--- /dev/null
+++ b/hosts/ha-server-1/host.nix
@@ -0,0 +1,26 @@
+{ vars, ... }:
+{
+ imports = [
+ (import ../../modules/beszel/host-token.nix {
+ name = "ha-server-1";
+ sopsFile = ../../secrets/ha-server-1.yaml;
+ })
+ ];
+
+ networking = {
+ hostName = vars.haServer1Host;
+ hostId = "3a4b5c6d";
+ useDHCP = false;
+ interfaces.ens18.ipv4.addresses = [{
+ address = vars.haServer1Ip;
+ prefixLength = 24;
+ }];
+ defaultGateway = "192.168.2.1";
+ nameservers = [ "192.168.2.1" "8.8.8.8" ];
+ };
+
+ # Set KEY after pairing this host with the beszel hub; the token is sops-managed.
+ services.beszel.agent.environment.KEY = "";
+
+ system.stateVersion = "26.05";
+}
diff --git a/hosts/ha-server-2/host.nix b/hosts/ha-server-2/host.nix
new file mode 100644
index 0000000..dcb1464
--- /dev/null
+++ b/hosts/ha-server-2/host.nix
@@ -0,0 +1,26 @@
+{ vars, ... }:
+{
+ imports = [
+ (import ../../modules/beszel/host-token.nix {
+ name = "ha-server-2";
+ sopsFile = ../../secrets/ha-server-2.yaml;
+ })
+ ];
+
+ networking = {
+ hostName = vars.haServer2Host;
+ hostId = "7e8f9a0b";
+ useDHCP = false;
+ interfaces.ens18.ipv4.addresses = [{
+ address = vars.haServer2Ip;
+ prefixLength = 24;
+ }];
+ defaultGateway = "192.168.2.1";
+ nameservers = [ "192.168.2.1" "8.8.8.8" ];
+ };
+
+ # Set KEY after pairing this host with the beszel hub; the token is sops-managed.
+ services.beszel.agent.environment.KEY = "";
+
+ system.stateVersion = "26.05";
+}
diff --git a/modules/build-types/ha-server.nix b/modules/build-types/ha-server.nix
new file mode 100644
index 0000000..26026ac
--- /dev/null
+++ b/modules/build-types/ha-server.nix
@@ -0,0 +1,43 @@
+# HA file server build type: DRBD + XFS + LIO iSCSI + NFS, managed by
+# Corosync + Pacemaker. Both ha-server-1 and ha-server-2 use this type.
+#
+# NFS start/stop:
+# services.nfs.server.enable = true configures /etc/exports, wires up
+# rpcbind, and loads kernel modules — but nfs-server.service.wantedBy is
+# force-cleared so systemd does NOT auto-start it at boot. Pacemaker's
+# ha-group resource group (configured by scripts/ha/cluster-init.sh)
+# starts and stops nfs-server as part of the failover sequence after the
+# XFS mount and iSCSI target are brought up on the new Active node.
+#
+# Beszel agent:
+# Enabled here via enable-agent.nix. The agent KEY (used to pair with
+# the Beszel hub) is not set yet — add it to hosts/ha-server-{1,2}/host.nix
+# under services.beszel.agent.environment.KEY once the hub accepts the
+# new agents, following the pattern in hosts/server/host.nix.
+{ lib, vars, ... }:
+{
+ imports = [
+ ../ha/pacemaker-stack.nix
+ ../ha/iscsi-target.nix
+ ../ha/cluster-config.nix
+ ../beszel/enable-agent.nix
+ ];
+
+ services.nfs.server = {
+ enable = true;
+ exports = ''
+ ${vars.haStorageRoot}/${vars.nfsShares.dockerConfig.subpath} ${vars.lanCidr}${vars.nfsShares.options}
+ ${vars.haStorageRoot}/${vars.nfsShares.dockerVolumes.subpath} ${vars.lanCidr}${vars.nfsShares.options}
+ ${vars.haStorageRoot}/${vars.nfsShares.dockerDatabases.subpath} ${vars.lanCidr}${vars.nfsShares.options}
+ ${vars.haStorageRoot}/${vars.nfsShares.nextcloudData.subpath} ${vars.lanCidr}${vars.nfsShares.options}
+ ${vars.haStorageRoot}/${vars.nfsShares.raspiVolumes.subpath} ${vars.lanCidr}${vars.nfsShares.options}
+ ${vars.haStorageRoot}/${vars.nfsShares.proxmoxIsos.subpath} ${vars.lanCidr}${vars.nfsShares.options}
+ ${vars.haStorageRoot}/${vars.nfsShares.proxmoxLxcImages.subpath} ${vars.lanCidr}${vars.nfsShares.options}
+ ${vars.haStorageRoot}/${vars.nfsShares.pxebootImages.subpath} ${vars.lanCidr}${vars.nfsShares.options}
+ '';
+ };
+
+ # Pacemaker controls nfs-server — prevent systemd from starting it at boot
+ # on both nodes (only the Active node should be serving NFS).
+ systemd.services.nfs-server.wantedBy = lib.mkForce [ ];
+}
diff --git a/modules/ha/cluster-config.nix b/modules/ha/cluster-config.nix
new file mode 100644
index 0000000..160c154
--- /dev/null
+++ b/modules/ha/cluster-config.nix
@@ -0,0 +1,106 @@
+# Cluster-wide HA config shared by both ha-server nodes.
+#
+# Covers everything that is identical on both nodes and references cluster
+# topology (node IPs, hostnames, DRBD resource). Per-node identity
+# (hostname, static IP, stateVersion) lives in hosts/ha-server-{1,2}/host.nix.
+#
+# Corosync authkey:
+# /etc/corosync/authkey (mode 0400) is managed by sops-nix below.
+# Bootstrap: run scripts/ha/cluster-init.sh on node1 to generate the key,
+# then encrypt it with: sops -e --input-type binary /etc/corosync/authkey > secrets/ha-corosync-authkey
+# Both host keys must be registered via sync-host-keys.sh first so both nodes can decrypt it.
+#
+# DRBD fencing:
+# Production setting is resource-only: DRBD waits for the STONITH fence
+# agent to confirm the peer is dead before promoting to Primary. This
+# requires a working fence_pve_ssh STONITH resource in Pacemaker
+# (see scripts/ha/cluster-enable-stonith.sh). On a fresh cluster with
+# no fence device yet, temporarily change to dont-care and run
+# cluster-enable-stonith.sh once the fence key is deployed.
+{ lib, vars, ... }:
+{
+ services.drbd = {
+ enable = true;
+ config = ''
+ global {
+ usage-count yes;
+ }
+
+ common {
+ net {
+ protocol C;
+ ping-int 1;
+ verify-alg sha256;
+ after-sb-0pri discard-zero-changes;
+ after-sb-1pri discard-secondary;
+ }
+ disk {
+ fencing resource-only;
+ }
+ }
+
+ resource ha-data {
+ volume 0 {
+ device /dev/drbd0;
+ disk /dev/sdb;
+ meta-disk internal;
+ }
+
+ on ${vars.haServer1Host} {
+ address ${vars.haServer1Ip}:${toString vars.ports.haServerDrbd};
+ }
+
+ on ${vars.haServer2Host} {
+ address ${vars.haServer2Ip}:${toString vars.ports.haServerDrbd};
+ }
+ }
+ '';
+ };
+
+ # /etc/corosync/authkey — sops binary secret, identical on both nodes.
+ # Decryptable by both ha-server host keys (added by sync-host-keys.sh).
+ sops.secrets.corosync_authkey = {
+ sopsFile = ../../secrets/ha-corosync-authkey;
+ format = "binary";
+ path = "/etc/corosync/authkey";
+ mode = "0400";
+ restartUnits = [ "corosync.service" ];
+ };
+
+ # NixOS common config enables NetworkManager by default; HA cluster nodes
+ # need stable static IPs with predictable interface names — NM is not suitable.
+ networking.networkmanager.enable = lib.mkForce false;
+
+ # services.corosync.enable is set by modules/ha/pacemaker-stack.nix.
+ services.corosync = {
+ clusterName = "ha-cluster";
+ nodelist = [
+ { nodeid = 1; name = vars.haServer1Host; ring_addrs = [ vars.haServer1Ip ]; }
+ { nodeid = 2; name = vars.haServer2Host; ring_addrs = [ vars.haServer2Ip ]; }
+ ];
+ };
+
+ networking.firewall = {
+ allowedTCPPorts = [
+ vars.ports.haServerIscsi
+ vars.ports.haServerPacemakerRemoted
+ vars.ports.haServerPcsd
+ vars.ports.haServerDrbd
+ vars.ports.nfsRpcbind
+ vars.ports.nfsd
+ vars.ports.nfsMountd
+ ];
+ allowedUDPPorts = [
+ vars.ports.haServerCorosync1
+ vars.ports.haServerCorosync2
+ vars.ports.haServerCorosyncCrypto
+ vars.ports.nfsRpcbind
+ vars.ports.nfsd
+ vars.ports.nfsMountd
+ ];
+ extraCommands = ''
+ iptables -A INPUT -s ${vars.haServer1Ip}/32 -j ACCEPT
+ iptables -A INPUT -s ${vars.haServer2Ip}/32 -j ACCEPT
+ '';
+ };
+}
diff --git a/modules/ha/iscsi-target.nix b/modules/ha/iscsi-target.nix
new file mode 100644
index 0000000..c39ec88
--- /dev/null
+++ b/modules/ha/iscsi-target.nix
@@ -0,0 +1,99 @@
+# LIO iSCSI target service (targetctl) for NixOS HA clusters.
+#
+# Provides the targetctl.service that saves/restores LIO configuration from
+# /etc/target/saveconfig.json. Pacemaker manages this service via its
+# systemd resource agent (class="systemd" type="targetctl").
+#
+# Why ExecStop is not simply "targetctl save":
+# targetctl save writes the LIO config to JSON but does NOT remove the LIO
+# target from the kernel's configfs. As a result, any fileio backing store
+# that LIO has open (e.g. iscsi-lun.img on an XFS-over-DRBD filesystem)
+# stays referenced in the kernel. The subsequent XFS umount from the
+# Filesystem OCF resource then returns EBUSY and either hangs for the full
+# op-stop timeout or fails outright, blocking the entire failover.
+#
+# The ExecStop script here additionally tears down the kernel LIO state
+# via rtslib_fb after saving, so the backing-store file descriptor is
+# released and umount succeeds immediately.
+#
+# Empty-config guard:
+# The save step is skipped when no iSCSI targets are currently active.
+# This prevents the secondary node (where LIO was never started) from
+# overwriting a valid saveconfig.json with an empty one when Pacemaker
+# stops the iscsi-target resource as part of a failover or cleanup.
+{ pkgs, ... }:
+
+let
+ python3 = pkgs.python3.withPackages (ps: [ ps.rtslib-fb ]);
+ targetctl = "${pkgs.targetcli-fb}/bin/targetctl";
+
+ targetctlStop = pkgs.writeScript "targetctl-stop" ''
+ #!${python3}/bin/python3
+ import subprocess, sys
+ import rtslib_fb
+
+ root = rtslib_fb.RTSRoot()
+ targets = list(root.targets)
+ if targets:
+ subprocess.run(
+ ["${targetctl}", "save", "/etc/target/saveconfig.json"],
+ capture_output=True,
+ )
+ print(f"saved {len(targets)} iSCSI target(s)")
+ else:
+ print("no active LIO targets — saveconfig.json unchanged")
+
+ for target in targets:
+ try:
+ for tpg in list(target.tpgs):
+ tpg.enable = False
+ target.delete()
+ except Exception as e:
+ print(f"warn (target): {e}", file=sys.stderr)
+ for so in list(root.storage_objects):
+ try:
+ so.delete()
+ except Exception as e:
+ print(f"warn (backstore): {e}", file=sys.stderr)
+ print("LIO kernel target cleared")
+ '';
+in
+{
+ boot.kernelModules = [
+ "target_core_mod"
+ "iscsi_target_mod"
+ "target_core_file"
+ "target_core_pscsi"
+ "target_core_user"
+ "configfs"
+ ];
+
+ systemd = {
+ mounts = [{
+ where = "/sys/kernel/config";
+ what = "configfs";
+ type = "configfs";
+ wantedBy = [ "multi-user.target" ];
+ before = [ "targetctl.service" ];
+ }];
+ services.targetctl = {
+ description = "LIO iSCSI target config save/restore";
+ wantedBy = [ "multi-user.target" ];
+ after = [ "sys-kernel-config.mount" "network.target" ];
+ requires = [ "sys-kernel-config.mount" ];
+ serviceConfig = {
+ Type = "oneshot";
+ RemainAfterExit = true;
+ ExecStart = "${targetctl} restore /etc/target/saveconfig.json";
+ ExecStop = "${targetctlStop}";
+ };
+ unitConfig.ConditionFileNotEmpty = "/etc/target/saveconfig.json";
+ };
+ tmpfiles.rules = [
+ "d /etc/target 0750 root root -"
+ "f /etc/target/saveconfig.json 0640 root root -"
+ ];
+ };
+
+ environment.systemPackages = [ pkgs.targetcli-fb ];
+}
diff --git a/modules/ha/pacemaker-stack.nix b/modules/ha/pacemaker-stack.nix
new file mode 100644
index 0000000..8fcab2c
--- /dev/null
+++ b/modules/ha/pacemaker-stack.nix
@@ -0,0 +1,94 @@
+# Pacemaker + Corosync HA stack for NixOS with known-good workarounds.
+#
+# Issues fixed here (confirmed through live testing on NixOS 25.11):
+#
+# 1. StateDirectory ownership reset: systemd's StateDirectory=pacemaker
+# creates /var/lib/pacemaker owned root:root. pacemaker-based (the CIB
+# daemon) runs as the hacluster user and calls pcmk__daemon_can_write,
+# which requires the CIB directory to be owned by hacluster or be
+# group-writable by haclient. Workaround: remove StateDirectory and let
+# ExecStartPre create every required subdirectory with correct ownership.
+#
+# 2. HA_SBIN_DIR wrong path: ocf-shellfuncs sets HA_SBIN_DIR to the Nix
+# store path of the resource-agents derivation's /sbin, which doesn't
+# exist. The DRBD OCF agent uses ${HA_SBIN_DIR}/crm_master, so it exits
+# 127 without this override. Fix: export HA_SBIN_DIR=/run/current-system/sw/bin.
+#
+# 3. Broad PATH for OCF agents: the resource executor (pacemaker-execd) runs
+# OCF agent scripts as children. NixOS provides no implicit PATH for
+# system services; without an explicit PATH the agents can't find ip, ss,
+# mount, umount, drbdadm, etc.
+#
+# 4. FUSER=true: the Filesystem OCF agent calls check_binary $FUSER (default:
+# fuser from psmisc), which is not installed. Setting FUSER=true makes
+# check_binary succeed (true is always in PATH) and the subsequent
+# "$FUSER -km $mountpoint" becomes a no-op. Pair with force_unmount=false
+# on each Filesystem resource unless you want lazy unmount behaviour.
+{ lib, pkgs, ... }:
+
+let
+ ocfBinPath = lib.concatStringsSep ":" [
+ "${pkgs.iproute2}/bin"
+ "${pkgs.iproute2}/sbin"
+ "${pkgs.iputils}/bin"
+ "${pkgs.util-linux}/bin"
+ "${pkgs.util-linux}/sbin"
+ "${pkgs.gawk}/bin"
+ "${pkgs.gnugrep}/bin"
+ "${pkgs.gnused}/bin"
+ "${pkgs.coreutils}/bin"
+ "${pkgs.bash}/bin"
+ "${pkgs.procps}/bin"
+ "${pkgs.xfsprogs}/bin"
+ "${pkgs.drbd}/bin"
+ "${pkgs.python3}/bin"
+ "/run/current-system/sw/bin"
+ "/run/current-system/sw/sbin"
+ "/usr/local/sbin"
+ "/usr/local/bin"
+ "/usr/sbin"
+ "/usr/bin"
+ "/sbin"
+ "/bin"
+ ];
+
+ # Single pre-start script: schemas symlink + directory ownership.
+ # Runs before pacemakerd so pacemaker-based finds hacluster-owned dirs.
+ preStartCmd = "${pkgs.bash}/bin/bash -c '"
+ + "ln -sfn ${pkgs.pacemaker}/share/pacemaker /var/lib/pacemaker/schemas; "
+ + "for d in /var/lib/pacemaker /var/lib/pacemaker/cib /var/lib/pacemaker/cores "
+ + "/var/lib/pacemaker/pengine /var/lib/pacemaker/blackbox "
+ + "/var/lib/pacemaker/hostcache; do "
+ + "mkdir -p \"\\$d\" && chown hacluster:pacemaker \"\\$d\" && chmod 2770 \"\\$d\"; "
+ + "done'";
+
+ ocfEnv = {
+ PATH = lib.mkForce ocfBinPath;
+ OCF_ROOT = "${pkgs.ocf-resource-agents}/usr/lib/ocf";
+ HA_SBIN_DIR = "/run/current-system/sw/bin";
+ FUSER = "true";
+ };
+in
+{
+ users.groups.haclient = { };
+
+ services.corosync.enable = true;
+ services.pacemaker.enable = true;
+
+ systemd.services = {
+ pacemaker = {
+ serviceConfig = {
+ StateDirectory = lib.mkForce "";
+ ExecStartPre = lib.mkBefore [ preStartCmd ];
+ };
+ environment = ocfEnv;
+ };
+ pacemaker-execd.environment = ocfEnv;
+ };
+
+ environment.systemPackages = with pkgs; [
+ corosync
+ pacemaker
+ ocf-resource-agents
+ ];
+}
diff --git a/scripts/ha/acceptance-tests.sh b/scripts/ha/acceptance-tests.sh
new file mode 100644
index 0000000..ab3081c
--- /dev/null
+++ b/scripts/ha/acceptance-tests.sh
@@ -0,0 +1,167 @@
+#!/usr/bin/env bash
+# acceptance-tests.sh — HA cluster acceptance tests (T1–T7)
+#
+# Run from a host with SSH access to both HA nodes (or from node1 itself).
+# All 7 tests must pass before considering the cluster production-ready.
+# Test values below must match variables.nix haServer* values.
+set -euo pipefail
+
+# ── Configuration ─────────────────────────────────────────────────────────
+NODE1="ha-server-1"
+NODE2="ha-server-2"
+NODE1_IP="192.168.2.200" # vars.haServer1Ip
+NODE2_IP="192.168.2.201" # vars.haServer2Ip
+VIP="192.168.2.202" # vars.haServerVip
+XFS_MOUNT="/srv/ha-data" # vars.haStorageRoot
+ISCSI_IQN="iqn.2026-01.home.sweet:ha-storage" # vars.haIscsiIqn
+# ──────────────────────────────────────────────────────────────────────────
+
+PASS=0
+FAIL=0
+RESULTS=()
+
+pass() { echo " PASS: $1"; ((PASS++)); RESULTS+=("PASS $1"); }
+fail() { echo " FAIL: $1"; ((FAIL++)); RESULTS+=("FAIL $1"); }
+
+n1() { ssh -o StrictHostKeyChecking=no -o ConnectTimeout=5 "root@${NODE1_IP}" "$@" 2>/dev/null; }
+n2() { ssh -o StrictHostKeyChecking=no -o ConnectTimeout=5 "root@${NODE2_IP}" "$@" 2>/dev/null; }
+
+echo "════════════════════════════════════════════════════"
+echo " HA Cluster Acceptance Tests — $(date '+%Y-%m-%d %H:%M:%S')"
+echo "════════════════════════════════════════════════════"
+
+# ── T1: Corosync quorum established ──────────────────────────────────────
+echo ""
+echo "[T1] Corosync quorum"
+if n1 "corosync-quorumtool -s" 2>/dev/null | grep -q "Quorate:.*Yes"; then
+ pass "cluster has quorum"
+else
+ fail "cluster does not have quorum — check corosync on both nodes"
+fi
+
+# ── T2: DRBD Primary on node1, Secondary on node2 ────────────────────────
+echo ""
+echo "[T2] DRBD roles"
+DRBD_ROLE=$(n1 "drbdadm role ha-data" 2>/dev/null || echo "unknown")
+if [[ "$DRBD_ROLE" == "Primary/Secondary" || "$DRBD_ROLE" == "Primary" ]]; then
+ pass "DRBD Primary on $NODE1 ($DRBD_ROLE)"
+else
+ fail "unexpected DRBD role on $NODE1: $DRBD_ROLE (expected Primary/Secondary)"
+fi
+
+DRBD_DSTATE=$(n1 "drbdadm dstate ha-data" 2>/dev/null || echo "unknown")
+if echo "$DRBD_DSTATE" | grep -q "UpToDate"; then
+ pass "DRBD disk state UpToDate ($DRBD_DSTATE)"
+else
+ fail "DRBD disk not UpToDate: $DRBD_DSTATE"
+fi
+
+# ── T3: XFS mounted at haStorageRoot on the Active node ──────────────────
+echo ""
+echo "[T3] XFS mount"
+if n1 "mountpoint -q '${XFS_MOUNT}'" 2>/dev/null; then
+ pass "XFS mounted at ${XFS_MOUNT} on $NODE1"
+else
+ fail "XFS not mounted at ${XFS_MOUNT} on $NODE1"
+fi
+
+if n2 "mountpoint -q '${XFS_MOUNT}'" 2>/dev/null; then
+ fail "XFS unexpectedly mounted on $NODE2 (should only be on Active node)"
+else
+ pass "XFS not mounted on $NODE2 (correct — Secondary)"
+fi
+
+# ── T4: iSCSI target visible on both nodes ────────────────────────────────
+echo ""
+echo "[T4] iSCSI target"
+IQN_COUNT=$(n1 "ls /sys/kernel/config/target/iscsi/ 2>/dev/null | grep -c iqn" || echo "0")
+if [[ "$IQN_COUNT" -ge 1 ]]; then
+ pass "iSCSI IQN active on $NODE1 ($IQN_COUNT target(s))"
+else
+ fail "no iSCSI IQN active on $NODE1"
+fi
+
+# iSCSI discovery from node2 via VIP
+if n2 "iscsiadm -m discovery -t sendtargets -p '${VIP}' 2>/dev/null | grep -q '${ISCSI_IQN}'"; then
+ pass "iSCSI target discoverable from $NODE2 via VIP ${VIP}"
+else
+ fail "iSCSI target not discoverable from $NODE2 via ${VIP}"
+fi
+
+# ── T5: Failover — standby node1, verify resources move to node2 ──────────
+echo ""
+echo "[T5] Failover (standby $NODE1)"
+MYNODE=$(n1 "crm_node -n" 2>/dev/null || echo "")
+n1 "crm_standby -N '${MYNODE}' -v on" 2>/dev/null || true
+echo " Waiting up to 30 s for resources to move to $NODE2..."
+MOVED=false
+for i in $(seq 1 30); do
+ if n2 "mountpoint -q '${XFS_MOUNT}'" 2>/dev/null; then
+ MOVED=true
+ echo " Resources moved in ${i}s"
+ break
+ fi
+ sleep 1
+done
+
+if $MOVED; then
+ pass "XFS mounted on $NODE2 after failover"
+ IQN_ON_N2=$(n2 "ls /sys/kernel/config/target/iscsi/ 2>/dev/null | grep -c iqn" || echo "0")
+ [[ "$IQN_ON_N2" -ge 1 ]] \
+ && pass "iSCSI target active on $NODE2 after failover" \
+ || fail "iSCSI target NOT active on $NODE2 after failover"
+else
+ fail "XFS did not mount on $NODE2 within 30 s — failover incomplete"
+fi
+
+# ── T6: Data integrity — file written pre-failover readable post-failover ─
+echo ""
+echo "[T6] Data integrity"
+# Write a test file on node2 (now Active) and verify its content
+TEST_FILE="${XFS_MOUNT}/.acceptance-test-$$"
+TEST_CONTENT="ha-acceptance-test-$(date +%s)"
+n2 "echo '${TEST_CONTENT}' > '${TEST_FILE}'" 2>/dev/null || true
+READBACK=$(n2 "cat '${TEST_FILE}' 2>/dev/null" || echo "")
+if [[ "$READBACK" == "$TEST_CONTENT" ]]; then
+ pass "test file written and read back correctly on $NODE2"
+else
+ fail "data integrity check failed (wrote: '$TEST_CONTENT', read: '$READBACK')"
+fi
+n2 "rm -f '${TEST_FILE}'" 2>/dev/null || true
+
+# ── T7: Node rejoin — un-standby node1, verify cluster is healthy ─────────
+echo ""
+echo "[T7] Node rejoin"
+n1 "crm_standby -N '${MYNODE}' -v off" 2>/dev/null || true
+n1 "crm_resource --cleanup" 2>/dev/null || true
+sleep 5
+
+ONLINE_NODES=$(n2 "crm_mon -1 2>/dev/null | grep -c 'Online:'" || echo "0")
+if n1 "corosync-quorumtool -s 2>/dev/null | grep -q 'Quorate:.*Yes'"; then
+ pass "$NODE1 rejoined — cluster has quorum"
+else
+ fail "$NODE1 did not rejoin with quorum"
+fi
+
+DRBD_ROLE_AFTER=$(n1 "drbdadm role ha-data" 2>/dev/null || echo "unknown")
+if echo "$DRBD_ROLE_AFTER" | grep -q "Secondary"; then
+ pass "$NODE1 is DRBD Secondary after rejoin ($DRBD_ROLE_AFTER)"
+else
+ fail "unexpected DRBD role on $NODE1 after rejoin: $DRBD_ROLE_AFTER"
+fi
+
+# ── Summary ───────────────────────────────────────────────────────────────
+echo ""
+echo "════════════════════════════════════════════════════"
+echo " Results: ${PASS} PASS, ${FAIL} FAIL"
+echo "════════════════════════════════════════════════════"
+for r in "${RESULTS[@]}"; do echo " $r"; done
+echo ""
+
+if [[ "$FAIL" -eq 0 ]]; then
+ echo "ALL PASS — cluster is production-ready."
+ exit 0
+else
+ echo "SOME TESTS FAILED — investigate before deploying."
+ exit 1
+fi
diff --git a/scripts/ha/cluster-enable-stonith.sh b/scripts/ha/cluster-enable-stonith.sh
new file mode 100644
index 0000000..6ef47a5
--- /dev/null
+++ b/scripts/ha/cluster-enable-stonith.sh
@@ -0,0 +1,86 @@
+#!/usr/bin/env bash
+# cluster-enable-stonith.sh — enable STONITH fence agent after the fence SSH
+# key is deployed to both nodes and authorised on the Proxmox host.
+#
+# Run from ha-server-1 as root AFTER:
+# - /etc/pacemaker/fence_pve_ssh exists on both nodes (chmod +x)
+# (copy from scripts/ha/fence-pve-ssh.py)
+# - /etc/fence-pve-ssh-key (SSH private key) exists on both nodes
+# - The corresponding public key is in authorized_keys on PVE_HOST
+# - VMID_NODE1 / VMID_NODE2 filled in below
+set -euo pipefail
+
+# ── Configuration ─────────────────────────────────────────────────────────
+NODE1="ha-server-1"
+NODE2="ha-server-2"
+VMID_NODE1="" # FILL IN: Proxmox VMID for ha-server-1
+VMID_NODE2="" # FILL IN: Proxmox VMID for ha-server-2
+PVE_HOST="pve1.sweet.home"
+PVE_USER="wayne"
+FENCE_KEY="/etc/fence-pve-ssh-key"
+FENCE_SCRIPT="/etc/pacemaker/fence_pve_ssh"
+# ──────────────────────────────────────────────────────────────────────────
+
+log() { echo "[stonith-setup] $*"; }
+die() { echo "[stonith-setup] ERROR: $*" >&2; exit 1; }
+
+[[ $(id -u) -eq 0 ]] || die "must run as root"
+[[ -n "$VMID_NODE1" ]] || die "VMID_NODE1 not set — edit this script"
+[[ -n "$VMID_NODE2" ]] || die "VMID_NODE2 not set — edit this script"
+[[ -f "$FENCE_KEY" ]] || die "fence key not found at $FENCE_KEY"
+[[ -f "$FENCE_SCRIPT" ]] || die "fence script not found at $FENCE_SCRIPT"
+
+log "Verifying fence agent can reach ${PVE_HOST}..."
+ssh -i "$FENCE_KEY" -o BatchMode=yes -o ConnectTimeout=10 \
+ -o StrictHostKeyChecking=no "${PVE_USER}@${PVE_HOST}" \
+ "sudo /usr/sbin/qm list" &>/dev/null \
+ || die "Cannot SSH to ${PVE_USER}@${PVE_HOST} — check authorized_keys and sudo"
+log "Fence agent SSH connectivity confirmed"
+
+log "Creating Pacemaker STONITH resources..."
+cibadmin --create --scope resources --xml-text "
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+" 2>/dev/null || true
+
+cibadmin --create --scope resources --xml-text "
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+" 2>/dev/null || true
+
+log "Enabling STONITH and restoring quorum policy..."
+crm_attribute -t crm_config -n stonith-enabled -v true
+crm_attribute -t crm_config -n no-quorum-policy -v stop
+
+log "DRBD fencing mode must also be updated to resource-only (already the"
+log "default in cluster-config.nix; confirm with: cat /etc/drbd.d/ha-data.conf)"
+
+log "Testing fence agent..."
+stonith_admin --list-devices && log "Fence devices listed successfully." \
+ || warn "stonith_admin --list-devices failed — check config"
+
+log "STONITH enabled. Cluster is now fully HA."
diff --git a/scripts/ha/cluster-init.sh b/scripts/ha/cluster-init.sh
new file mode 100644
index 0000000..1aac090
--- /dev/null
+++ b/scripts/ha/cluster-init.sh
@@ -0,0 +1,284 @@
+#!/usr/bin/env bash
+# cluster-init.sh — one-time HA cluster initialisation script
+#
+# Run ONCE from ha-server-1 as root AFTER both VMs are booted and have SSH
+# access. It:
+# 1. Generates and distributes the corosync authkey
+# 2. Waits for corosync quorum and pacemaker
+# 3. Initialises DRBD metadata, promotes node1 to primary
+# 4. Creates XFS on /dev/drbd0 and mounts it
+# 5. Creates the directory tree and iSCSI LUN backing file
+# 6. Configures LIO iSCSI target (file-backed LUN)
+# 7. Configures Pacemaker resources: DRBD → XFS → iSCSI → NFS → VIP
+#
+# Prerequisites:
+# - Both VMs booted with the ha-server config (nixos-rebuild done)
+# - SSH key access from node1 to root@NODE2_IP
+# - VMID_NODE1 / VMID_NODE2 filled in below (needed for STONITH setup;
+# cluster starts without STONITH, which you enable separately via
+# scripts/ha/cluster-enable-stonith.sh)
+# - Run as root on ha-server-1
+set -euo pipefail
+
+# ── Configuration ─────────────────────────────────────────────────────────
+# These must match variables.nix haServer* values and the Proxmox VMID
+# assignments. Update before running.
+NODE1="ha-server-1"
+NODE2="ha-server-2"
+NODE1_IP="192.168.2.200" # vars.haServer1Ip
+NODE2_IP="192.168.2.201" # vars.haServer2Ip
+VIP="192.168.2.202" # vars.haServerVip
+XFS_MOUNT="/srv/ha-data" # vars.haStorageRoot
+ISCSI_IQN="iqn.2026-01.home.sweet:ha-storage" # vars.haIscsiIqn
+ISCSI_LUN_FILE="${XFS_MOUNT}/iscsi-lun.img"
+ISCSI_LUN_SIZE="10G"
+DRBD_DEVICE="/dev/drbd0"
+VMID_NODE1="" # FILL IN: Proxmox VMID for ha-server-1
+VMID_NODE2="" # FILL IN: Proxmox VMID for ha-server-2
+PVE_HOST="pve1.sweet.home"
+PVE_USER="wayne"
+
+# NFS dataset subdirectories to create under XFS_MOUNT.
+# Must mirror vars.nfsShares subpath values in variables.nix.
+NFS_SUBDIRS=(
+ "docker/config"
+ "docker/volumes"
+ "docker/databases"
+ "docker/nextcloud-data"
+ "raspi/volumes"
+ "proxmox/iso"
+ "proxmox/lxc"
+ "pxe-boot/images"
+)
+# ──────────────────────────────────────────────────────────────────────────
+
+log() { echo "[cluster-init] $*"; }
+die() { echo "[cluster-init] ERROR: $*" >&2; exit 1; }
+warn() { echo "[cluster-init] WARNING: $*" >&2; }
+
+[[ $(id -u) -eq 0 ]] || die "must run as root"
+[[ "$(hostname)" == "$NODE1" ]] || die "must run on $NODE1"
+
+# ── 0. Corosync authkey ───────────────────────────────────────────────────
+AUTHKEY="/etc/corosync/authkey"
+mkdir -p /etc/corosync
+if [[ ! -f "$AUTHKEY" ]]; then
+ log "Generating corosync authkey..."
+ corosync-keygen -k "$AUTHKEY"
+ chmod 0400 "$AUTHKEY"
+fi
+log "Distributing authkey to $NODE2..."
+ssh "root@${NODE2_IP}" "mkdir -p /etc/corosync"
+scp -q "$AUTHKEY" "root@${NODE2_IP}:${AUTHKEY}"
+ssh "root@${NODE2_IP}" "chmod 0400 '${AUTHKEY}'"
+
+log "Restarting corosync on both nodes..."
+systemctl restart corosync
+ssh "root@${NODE2_IP}" "systemctl restart corosync"
+sleep 3
+
+# ── 1. Corosync quorum ────────────────────────────────────────────────────
+log "Waiting for corosync quorum..."
+for i in $(seq 1 30); do
+ if corosync-quorumtool -s 2>/dev/null | grep -q 'Quorate:.*Yes'; then
+ log "Quorum established"
+ break
+ fi
+ [[ $i -eq 30 ]] && die "corosync quorum not established after 60 s"
+ sleep 2
+done
+
+log "Waiting for pacemaker..."
+for i in $(seq 1 30); do
+ if crm_mon -1 &>/dev/null; then
+ log "Pacemaker running"
+ break
+ fi
+ [[ $i -eq 30 ]] && die "pacemaker not running after 60 s"
+ sleep 2
+done
+
+# ── 2. DRBD initialisation ────────────────────────────────────────────────
+log "Initialising DRBD metadata on $NODE1..."
+if ! drbdadm dstate ha-data 2>/dev/null | grep -q "UpToDate\|Inconsistent\|Diskless"; then
+ drbdadm create-md ha-data --force
+fi
+
+log "Initialising DRBD metadata on $NODE2..."
+ssh "root@${NODE2_IP}" "
+ if ! drbdadm dstate ha-data 2>/dev/null | grep -q 'UpToDate\|Inconsistent\|Diskless'; then
+ drbdadm create-md ha-data --force
+ fi
+"
+
+log "Bringing up DRBD on both nodes..."
+drbdadm up ha-data 2>/dev/null || true
+ssh "root@${NODE2_IP}" "drbdadm up ha-data 2>/dev/null" || true
+
+log "Forcing $NODE1 to DRBD Primary for initial sync..."
+drbdadm primary ha-data --force
+
+log "Waiting for DRBD to finish initial sync (this may take several minutes)..."
+for i in $(seq 1 300); do
+ state=$(drbdadm dstate ha-data 2>/dev/null || echo "unknown")
+ if echo "$state" | grep -q "UpToDate/UpToDate"; then
+ log "DRBD sync complete: $state"
+ break
+ fi
+ [[ $i -eq 300 ]] && warn "DRBD not UpToDate after 300 s — continuing anyway (check drbdadm status)"
+ sleep 1
+done
+
+# ── 3. XFS filesystem ─────────────────────────────────────────────────────
+log "Creating XFS on ${DRBD_DEVICE}..."
+if ! xfs_info "${DRBD_DEVICE}" &>/dev/null; then
+ mkfs.xfs -f "${DRBD_DEVICE}"
+fi
+
+log "Mounting ${DRBD_DEVICE} at ${XFS_MOUNT}..."
+mkdir -p "${XFS_MOUNT}"
+mount "${DRBD_DEVICE}" "${XFS_MOUNT}"
+
+# ── 4. NFS dataset directories ────────────────────────────────────────────
+log "Creating NFS dataset directories..."
+for subdir in "${NFS_SUBDIRS[@]}"; do
+ mkdir -p "${XFS_MOUNT}/${subdir}"
+done
+
+# ── 5. iSCSI LUN backing file ─────────────────────────────────────────────
+log "Creating iSCSI LUN backing file ${ISCSI_LUN_FILE} (${ISCSI_LUN_SIZE})..."
+if [[ ! -f "${ISCSI_LUN_FILE}" ]]; then
+ fallocate -l "${ISCSI_LUN_SIZE}" "${ISCSI_LUN_FILE}"
+fi
+
+# ── 6. LIO iSCSI target ───────────────────────────────────────────────────
+log "Configuring LIO iSCSI target via targetcli..."
+targetcli <
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+"
+
+log "Adding ordering and colocation constraints..."
+cibadmin --create --scope constraints --xml-text "
+
+
+
+
+"
+
+log "Waiting for resources to start..."
+for i in $(seq 1 60); do
+ if crm_resource -r vip --locate 2>/dev/null | grep -q "running on"; then
+ log "VIP is up: $(crm_resource -r vip --locate)"
+ break
+ fi
+ [[ $i -eq 60 ]] && { warn "VIP not up after 120 s — check: crm_mon -1"; break; }
+ sleep 2
+done
+
+log ""
+log "═══════════════════════════════════════════════════════════════"
+log " HA cluster initialised."
+log ""
+log " crm_mon -1 — cluster status"
+log " iscsiadm -m discovery -t st -p ${VIP} — verify iSCSI target"
+log " showmount -e ${VIP} — verify NFS exports"
+log ""
+log " To enable STONITH (after deploying fence SSH key):"
+log " 1. Fill in VMID_NODE1 / VMID_NODE2 in cluster-enable-stonith.sh"
+log " 2. Copy scripts/ha/fence-pve-ssh.py to /etc/pacemaker/fence_pve_ssh"
+log " on both nodes (chmod +x)"
+log " 3. Generate and distribute the fence SSH key"
+log " (see docs or cluster-enable-stonith.sh header)"
+log " 4. bash scripts/ha/cluster-enable-stonith.sh"
+log "═══════════════════════════════════════════════════════════════"
diff --git a/scripts/ha/fence-pve-ssh.py b/scripts/ha/fence-pve-ssh.py
new file mode 100644
index 0000000..58c24ed
--- /dev/null
+++ b/scripts/ha/fence-pve-ssh.py
@@ -0,0 +1,179 @@
+#!/usr/bin/env python3
+"""
+fence_pve_ssh - Proxmox VE SSH fence agent for Pacemaker.
+
+Uses SSH to reach the Proxmox host and run 'qm stop/start '.
+Deploy to /etc/pacemaker/fence_pve_ssh on both HA nodes (chmod +x).
+
+Configuration (as pacemaker stonith resource attributes):
+ pve_host Proxmox host to SSH to (default: pve1.sweet.home)
+ pve_user SSH user (default: wayne)
+ key_file SSH private key path (default: /etc/fence-pve-ssh-key)
+ vmid_node1 VMID for ha-server-1
+ vmid_node2 VMID for ha-server-2
+ plug Node name to act on (set by pacemaker: ha-server-1 or ha-server-2)
+ action Action: off|on|reboot|status|list|metadata
+"""
+
+import argparse
+import subprocess
+import sys
+import os
+
+
+METADATA = """
+
+ Fences a VM on a Proxmox VE host by SSHing to the PVE host and
+ running qm stop/start. For test use only.
+ https://proxmox.com
+
+
+
+
+ Fencing action: off|on|reboot|status|list
+
+
+
+
+ Cluster node name to fence
+
+
+
+
+ Proxmox VE host to SSH to
+
+
+
+
+ SSH user on the Proxmox host
+
+
+
+
+ SSH private key file path
+
+
+
+
+ VMID for ha-test-node1
+
+
+
+
+ VMID for ha-test-node2
+
+
+
+
+
+
+
+
+
+
+
+"""
+
+
+def parse_args():
+ p = argparse.ArgumentParser(add_help=False)
+ p.add_argument("-a", "--action", default="reboot")
+ p.add_argument("-n", "--plug")
+ p.add_argument("--pve-host", default="pve1.sweet.home")
+ p.add_argument("--pve-user", default="wayne")
+ p.add_argument("--key-file", default="/etc/fence-pve-ssh-key")
+ p.add_argument("--vmid-node1")
+ p.add_argument("--vmid-node2")
+ # Allow remaining unknown args (pacemaker may pass extra ones)
+ return p.parse_known_args()[0]
+
+
+def ssh(pve_host, pve_user, key_file, cmd):
+ result = subprocess.run(
+ [
+ "ssh",
+ "-i", key_file,
+ "-o", "StrictHostKeyChecking=no",
+ "-o", "BatchMode=yes",
+ "-o", "ConnectTimeout=10",
+ f"{pve_user}@{pve_host}",
+ cmd,
+ ],
+ capture_output=True,
+ text=True,
+ timeout=30,
+ )
+ return result
+
+
+def get_vmid(args):
+ node = args.plug
+ if not node:
+ print("ERROR: --plug not specified", file=sys.stderr)
+ sys.exit(1)
+ mapping = {
+ "ha-server-1": args.vmid_node1,
+ "ha-server-2": args.vmid_node2,
+ }
+ vmid = mapping.get(node)
+ if not vmid:
+ print(f"ERROR: unknown node '{node}'", file=sys.stderr)
+ sys.exit(1)
+ return vmid
+
+
+def main():
+ args = parse_args()
+ action = args.action.lower()
+
+ if action == "metadata":
+ print(METADATA)
+ sys.exit(0)
+
+ if action == "list":
+ if args.vmid_node1:
+ print("ha-server-1")
+ if args.vmid_node2:
+ print("ha-server-2")
+ sys.exit(0)
+
+ vmid = get_vmid(args)
+
+ if not os.path.exists(args.key_file):
+ print(f"ERROR: SSH key not found at {args.key_file}", file=sys.stderr)
+ sys.exit(1)
+
+ if action in ("off", "reboot"):
+ print(f"Stopping VM {vmid} ({args.plug}) on {args.pve_host}...")
+ r = ssh(args.pve_host, args.pve_user, args.key_file,
+ f"sudo /usr/sbin/qm stop {vmid}")
+ if r.returncode != 0:
+ print(f"ERROR stopping VM: {r.stderr}", file=sys.stderr)
+ sys.exit(1)
+ print(f"VM {vmid} stopped")
+
+ if action in ("on", "reboot"):
+ print(f"Starting VM {vmid} ({args.plug}) on {args.pve_host}...")
+ r = ssh(args.pve_host, args.pve_user, args.key_file,
+ f"sudo /usr/sbin/qm start {vmid}")
+ if r.returncode != 0:
+ print(f"ERROR starting VM: {r.stderr}", file=sys.stderr)
+ sys.exit(1)
+ print(f"VM {vmid} started")
+
+ if action == "status":
+ r = ssh(args.pve_host, args.pve_user, args.key_file,
+ f"sudo /usr/sbin/qm status {vmid}")
+ if r.returncode != 0:
+ print(f"ERROR querying VM status: {r.stderr}", file=sys.stderr)
+ sys.exit(1)
+ # qm status returns "status: running" or "status: stopped"
+ status_line = r.stdout.strip()
+ print(status_line)
+ if "stopped" in status_line:
+ sys.exit(2) # pacemaker interprets exit 2 as "off"
+ sys.exit(0) # running = exit 0
+
+
+if __name__ == "__main__":
+ main()
diff --git a/secrets/ha-corosync-authkey b/secrets/ha-corosync-authkey
new file mode 100644
index 0000000..74db83c
--- /dev/null
+++ b/secrets/ha-corosync-authkey
@@ -0,0 +1 @@
+STUB: run cluster-init.sh to generate, then: sops -e --input-type binary /etc/corosync/authkey > secrets/ha-corosync-authkey
diff --git a/secrets/ha-server-1.yaml b/secrets/ha-server-1.yaml
new file mode 100644
index 0000000..f33120f
--- /dev/null
+++ b/secrets/ha-server-1.yaml
@@ -0,0 +1,6 @@
+# STUB — not yet encrypted with sops.
+# Bootstrap:
+# bash scripts/secrets/sync-host-keys.sh proxmox-ha-server-1
+# sops updatekeys secrets/common.yaml (allows ha-server-1 to decrypt shared secrets)
+# sops secrets/ha-server-1.yaml (create with: beszel-token)
+beszel-token: REPLACE
diff --git a/secrets/ha-server-2.yaml b/secrets/ha-server-2.yaml
new file mode 100644
index 0000000..19bbbe9
--- /dev/null
+++ b/secrets/ha-server-2.yaml
@@ -0,0 +1,6 @@
+# STUB — not yet encrypted with sops.
+# Bootstrap:
+# bash scripts/secrets/sync-host-keys.sh proxmox-ha-server-2
+# sops updatekeys secrets/common.yaml (allows ha-server-2 to decrypt shared secrets)
+# sops secrets/ha-server-2.yaml (create with: beszel-token)
+beszel-token: REPLACE
diff --git a/variables.nix b/variables.nix
index e1e15b1..c59fa68 100644
--- a/variables.nix
+++ b/variables.nix
@@ -68,6 +68,20 @@
# one-line change.
primaryUser = "nixos";
+ # HA file server cluster
+ # haServer1Ip / haServer2Ip: static LAN IPs for both HA nodes (must be
+ # fixed — DRBD and corosync ring addresses are baked into the NixOS config).
+ # haServerVip: floating virtual IP managed by Pacemaker's IPaddr2 resource;
+ # NFS and iSCSI clients connect here regardless of which node is Active.
+ # Set all three to real values in variables.nix before deploying.
+ haServer1Host = "ha-server-1";
+ haServer2Host = "ha-server-2";
+ haServer1Ip = "192.168.2.200"; # TODO: confirm production IP
+ haServer2Ip = "192.168.2.201"; # TODO: confirm production IP
+ haServerVip = "192.168.2.202"; # TODO: confirm floating VIP
+ haStorageRoot = "/srv/ha-data"; # XFS-over-DRBD mount point on the Active node
+ haIscsiIqn = "iqn.2026-01.home.sweet:ha-storage";
+
# Storage
storageRoot = "/tank"; # ZFS pool root on `server`
@@ -146,11 +160,22 @@
# mountd RPC service (used by showmount/NFSv3 mount protocol).
# Mountd listens on a fixed port so the firewall can whitelist it
# explicitly rather than opening all of rpcbind's dynamic range.
- # All three need both TCP and UDP (modules/build-types/server.nix).
+ # All three need both TCP and UDP (modules/build-types/server.nix and
+ # modules/build-types/ha-server.nix).
nfsRpcbind = 111;
nfsd = 2049;
nfsMountd = 20048;
+ # HA cluster ports opened on ha-server-1 and ha-server-2
+ # (modules/build-types/ha-server.nix / modules/ha/cluster-config.nix).
+ haServerDrbd = 7789; # DRBD replication (TCP)
+ haServerIscsi = 3260; # iSCSI target (TCP)
+ haServerCorosync1 = 5404; # Corosync totem ring (UDP)
+ haServerCorosync2 = 5405; # Corosync totem ring (UDP)
+ haServerCorosyncCrypto = 5407; # Corosync crypto sync (UDP)
+ haServerPacemakerRemoted = 3121; # pacemaker-remoted (TCP)
+ haServerPcsd = 2224; # pcsd cluster daemon (TCP)
+
# Opened on the docker host's firewall for the Traefik-fronted
# container stack (docker-compose config lives in the separate
# /home/debian/docker repo, not here): 80/443 are Traefik's own