diff --git a/.sops.yaml b/.sops.yaml index b37b4e2..a14898b 100644 --- a/.sops.yaml +++ b/.sops.yaml @@ -85,6 +85,32 @@ creation_rules: - *lxc-tailscale-router - *proxmox-tailscale-router + # HA file server per-node secrets (beszel-token). + # proxmox-ha-server-1 / proxmox-ha-server-2 keys are added automatically + # by scripts/secrets/sync-host-keys.sh once the hosts are provisioned; + # until then only the admin key can decrypt these files. + - path_regex: secrets/ha-server-1\.yaml$ + key_groups: + - age: + - *admin + # proxmox-ha-server-1 added by sync-host-keys.sh + + - path_regex: secrets/ha-server-2\.yaml$ + key_groups: + - age: + - *admin + # proxmox-ha-server-2 added by sync-host-keys.sh + + # Shared HA cluster corosync authkey (binary sops file). + # Encrypted for both HA nodes so either can decrypt on boot. + # Both host keys added by sync-host-keys.sh; admin key allows initial creation. + - path_regex: secrets/ha-corosync-authkey$ + key_groups: + - age: + - *admin + # proxmox-ha-server-1 added by sync-host-keys.sh + # proxmox-ha-server-2 added by sync-host-keys.sh + # gui-host-specific secrets (currently: wifi-password, see # modules/networking/wifi.nix). Only *lxc-gui has a registered key today # -- proxmox-gui/linode-gui/baremetal-gui haven't been provisioned via diff --git a/flake.nix b/flake.nix index 47ac295..d4b0aa0 100644 --- a/flake.nix +++ b/flake.nix @@ -130,6 +130,9 @@ lxc-tailscale-router = mkTarget { platform = "lxc"; buildType = "tailscale-router"; hostPath = ./hosts/tailscale-router/host.nix; }; lxc-tor-relay = mkTarget { platform = "lxc"; buildType = "tor-relay"; hostPath = ./hosts/tor-relay/host.nix; }; + + proxmox-ha-server-1 = mkTarget { platform = "proxmox"; buildType = "ha-server"; hostPath = ./hosts/ha-server-1/host.nix; }; + proxmox-ha-server-2 = mkTarget { platform = "proxmox"; buildType = "ha-server"; hostPath = ./hosts/ha-server-2/host.nix; }; }; # Auto-install environments (migrated from the former nix-auto-installer diff --git a/hosts/ha-server-1/host.nix b/hosts/ha-server-1/host.nix new file mode 100644 index 0000000..c380008 --- /dev/null +++ b/hosts/ha-server-1/host.nix @@ -0,0 +1,26 @@ +{ vars, ... }: +{ + imports = [ + (import ../../modules/beszel/host-token.nix { + name = "ha-server-1"; + sopsFile = ../../secrets/ha-server-1.yaml; + }) + ]; + + networking = { + hostName = vars.haServer1Host; + hostId = "3a4b5c6d"; + useDHCP = false; + interfaces.ens18.ipv4.addresses = [{ + address = vars.haServer1Ip; + prefixLength = 24; + }]; + defaultGateway = "192.168.2.1"; + nameservers = [ "192.168.2.1" "8.8.8.8" ]; + }; + + # Set KEY after pairing this host with the beszel hub; the token is sops-managed. + services.beszel.agent.environment.KEY = ""; + + system.stateVersion = "26.05"; +} diff --git a/hosts/ha-server-2/host.nix b/hosts/ha-server-2/host.nix new file mode 100644 index 0000000..dcb1464 --- /dev/null +++ b/hosts/ha-server-2/host.nix @@ -0,0 +1,26 @@ +{ vars, ... }: +{ + imports = [ + (import ../../modules/beszel/host-token.nix { + name = "ha-server-2"; + sopsFile = ../../secrets/ha-server-2.yaml; + }) + ]; + + networking = { + hostName = vars.haServer2Host; + hostId = "7e8f9a0b"; + useDHCP = false; + interfaces.ens18.ipv4.addresses = [{ + address = vars.haServer2Ip; + prefixLength = 24; + }]; + defaultGateway = "192.168.2.1"; + nameservers = [ "192.168.2.1" "8.8.8.8" ]; + }; + + # Set KEY after pairing this host with the beszel hub; the token is sops-managed. + services.beszel.agent.environment.KEY = ""; + + system.stateVersion = "26.05"; +} diff --git a/modules/build-types/ha-server.nix b/modules/build-types/ha-server.nix new file mode 100644 index 0000000..26026ac --- /dev/null +++ b/modules/build-types/ha-server.nix @@ -0,0 +1,43 @@ +# HA file server build type: DRBD + XFS + LIO iSCSI + NFS, managed by +# Corosync + Pacemaker. Both ha-server-1 and ha-server-2 use this type. +# +# NFS start/stop: +# services.nfs.server.enable = true configures /etc/exports, wires up +# rpcbind, and loads kernel modules — but nfs-server.service.wantedBy is +# force-cleared so systemd does NOT auto-start it at boot. Pacemaker's +# ha-group resource group (configured by scripts/ha/cluster-init.sh) +# starts and stops nfs-server as part of the failover sequence after the +# XFS mount and iSCSI target are brought up on the new Active node. +# +# Beszel agent: +# Enabled here via enable-agent.nix. The agent KEY (used to pair with +# the Beszel hub) is not set yet — add it to hosts/ha-server-{1,2}/host.nix +# under services.beszel.agent.environment.KEY once the hub accepts the +# new agents, following the pattern in hosts/server/host.nix. +{ lib, vars, ... }: +{ + imports = [ + ../ha/pacemaker-stack.nix + ../ha/iscsi-target.nix + ../ha/cluster-config.nix + ../beszel/enable-agent.nix + ]; + + services.nfs.server = { + enable = true; + exports = '' + ${vars.haStorageRoot}/${vars.nfsShares.dockerConfig.subpath} ${vars.lanCidr}${vars.nfsShares.options} + ${vars.haStorageRoot}/${vars.nfsShares.dockerVolumes.subpath} ${vars.lanCidr}${vars.nfsShares.options} + ${vars.haStorageRoot}/${vars.nfsShares.dockerDatabases.subpath} ${vars.lanCidr}${vars.nfsShares.options} + ${vars.haStorageRoot}/${vars.nfsShares.nextcloudData.subpath} ${vars.lanCidr}${vars.nfsShares.options} + ${vars.haStorageRoot}/${vars.nfsShares.raspiVolumes.subpath} ${vars.lanCidr}${vars.nfsShares.options} + ${vars.haStorageRoot}/${vars.nfsShares.proxmoxIsos.subpath} ${vars.lanCidr}${vars.nfsShares.options} + ${vars.haStorageRoot}/${vars.nfsShares.proxmoxLxcImages.subpath} ${vars.lanCidr}${vars.nfsShares.options} + ${vars.haStorageRoot}/${vars.nfsShares.pxebootImages.subpath} ${vars.lanCidr}${vars.nfsShares.options} + ''; + }; + + # Pacemaker controls nfs-server — prevent systemd from starting it at boot + # on both nodes (only the Active node should be serving NFS). + systemd.services.nfs-server.wantedBy = lib.mkForce [ ]; +} diff --git a/modules/ha/cluster-config.nix b/modules/ha/cluster-config.nix new file mode 100644 index 0000000..160c154 --- /dev/null +++ b/modules/ha/cluster-config.nix @@ -0,0 +1,106 @@ +# Cluster-wide HA config shared by both ha-server nodes. +# +# Covers everything that is identical on both nodes and references cluster +# topology (node IPs, hostnames, DRBD resource). Per-node identity +# (hostname, static IP, stateVersion) lives in hosts/ha-server-{1,2}/host.nix. +# +# Corosync authkey: +# /etc/corosync/authkey (mode 0400) is managed by sops-nix below. +# Bootstrap: run scripts/ha/cluster-init.sh on node1 to generate the key, +# then encrypt it with: sops -e --input-type binary /etc/corosync/authkey > secrets/ha-corosync-authkey +# Both host keys must be registered via sync-host-keys.sh first so both nodes can decrypt it. +# +# DRBD fencing: +# Production setting is resource-only: DRBD waits for the STONITH fence +# agent to confirm the peer is dead before promoting to Primary. This +# requires a working fence_pve_ssh STONITH resource in Pacemaker +# (see scripts/ha/cluster-enable-stonith.sh). On a fresh cluster with +# no fence device yet, temporarily change to dont-care and run +# cluster-enable-stonith.sh once the fence key is deployed. +{ lib, vars, ... }: +{ + services.drbd = { + enable = true; + config = '' + global { + usage-count yes; + } + + common { + net { + protocol C; + ping-int 1; + verify-alg sha256; + after-sb-0pri discard-zero-changes; + after-sb-1pri discard-secondary; + } + disk { + fencing resource-only; + } + } + + resource ha-data { + volume 0 { + device /dev/drbd0; + disk /dev/sdb; + meta-disk internal; + } + + on ${vars.haServer1Host} { + address ${vars.haServer1Ip}:${toString vars.ports.haServerDrbd}; + } + + on ${vars.haServer2Host} { + address ${vars.haServer2Ip}:${toString vars.ports.haServerDrbd}; + } + } + ''; + }; + + # /etc/corosync/authkey — sops binary secret, identical on both nodes. + # Decryptable by both ha-server host keys (added by sync-host-keys.sh). + sops.secrets.corosync_authkey = { + sopsFile = ../../secrets/ha-corosync-authkey; + format = "binary"; + path = "/etc/corosync/authkey"; + mode = "0400"; + restartUnits = [ "corosync.service" ]; + }; + + # NixOS common config enables NetworkManager by default; HA cluster nodes + # need stable static IPs with predictable interface names — NM is not suitable. + networking.networkmanager.enable = lib.mkForce false; + + # services.corosync.enable is set by modules/ha/pacemaker-stack.nix. + services.corosync = { + clusterName = "ha-cluster"; + nodelist = [ + { nodeid = 1; name = vars.haServer1Host; ring_addrs = [ vars.haServer1Ip ]; } + { nodeid = 2; name = vars.haServer2Host; ring_addrs = [ vars.haServer2Ip ]; } + ]; + }; + + networking.firewall = { + allowedTCPPorts = [ + vars.ports.haServerIscsi + vars.ports.haServerPacemakerRemoted + vars.ports.haServerPcsd + vars.ports.haServerDrbd + vars.ports.nfsRpcbind + vars.ports.nfsd + vars.ports.nfsMountd + ]; + allowedUDPPorts = [ + vars.ports.haServerCorosync1 + vars.ports.haServerCorosync2 + vars.ports.haServerCorosyncCrypto + vars.ports.nfsRpcbind + vars.ports.nfsd + vars.ports.nfsMountd + ]; + extraCommands = '' + iptables -A INPUT -s ${vars.haServer1Ip}/32 -j ACCEPT + iptables -A INPUT -s ${vars.haServer2Ip}/32 -j ACCEPT + ''; + }; +} diff --git a/modules/ha/iscsi-target.nix b/modules/ha/iscsi-target.nix new file mode 100644 index 0000000..c39ec88 --- /dev/null +++ b/modules/ha/iscsi-target.nix @@ -0,0 +1,99 @@ +# LIO iSCSI target service (targetctl) for NixOS HA clusters. +# +# Provides the targetctl.service that saves/restores LIO configuration from +# /etc/target/saveconfig.json. Pacemaker manages this service via its +# systemd resource agent (class="systemd" type="targetctl"). +# +# Why ExecStop is not simply "targetctl save": +# targetctl save writes the LIO config to JSON but does NOT remove the LIO +# target from the kernel's configfs. As a result, any fileio backing store +# that LIO has open (e.g. iscsi-lun.img on an XFS-over-DRBD filesystem) +# stays referenced in the kernel. The subsequent XFS umount from the +# Filesystem OCF resource then returns EBUSY and either hangs for the full +# op-stop timeout or fails outright, blocking the entire failover. +# +# The ExecStop script here additionally tears down the kernel LIO state +# via rtslib_fb after saving, so the backing-store file descriptor is +# released and umount succeeds immediately. +# +# Empty-config guard: +# The save step is skipped when no iSCSI targets are currently active. +# This prevents the secondary node (where LIO was never started) from +# overwriting a valid saveconfig.json with an empty one when Pacemaker +# stops the iscsi-target resource as part of a failover or cleanup. +{ pkgs, ... }: + +let + python3 = pkgs.python3.withPackages (ps: [ ps.rtslib-fb ]); + targetctl = "${pkgs.targetcli-fb}/bin/targetctl"; + + targetctlStop = pkgs.writeScript "targetctl-stop" '' + #!${python3}/bin/python3 + import subprocess, sys + import rtslib_fb + + root = rtslib_fb.RTSRoot() + targets = list(root.targets) + if targets: + subprocess.run( + ["${targetctl}", "save", "/etc/target/saveconfig.json"], + capture_output=True, + ) + print(f"saved {len(targets)} iSCSI target(s)") + else: + print("no active LIO targets — saveconfig.json unchanged") + + for target in targets: + try: + for tpg in list(target.tpgs): + tpg.enable = False + target.delete() + except Exception as e: + print(f"warn (target): {e}", file=sys.stderr) + for so in list(root.storage_objects): + try: + so.delete() + except Exception as e: + print(f"warn (backstore): {e}", file=sys.stderr) + print("LIO kernel target cleared") + ''; +in +{ + boot.kernelModules = [ + "target_core_mod" + "iscsi_target_mod" + "target_core_file" + "target_core_pscsi" + "target_core_user" + "configfs" + ]; + + systemd = { + mounts = [{ + where = "/sys/kernel/config"; + what = "configfs"; + type = "configfs"; + wantedBy = [ "multi-user.target" ]; + before = [ "targetctl.service" ]; + }]; + services.targetctl = { + description = "LIO iSCSI target config save/restore"; + wantedBy = [ "multi-user.target" ]; + after = [ "sys-kernel-config.mount" "network.target" ]; + requires = [ "sys-kernel-config.mount" ]; + serviceConfig = { + Type = "oneshot"; + RemainAfterExit = true; + ExecStart = "${targetctl} restore /etc/target/saveconfig.json"; + ExecStop = "${targetctlStop}"; + }; + unitConfig.ConditionFileNotEmpty = "/etc/target/saveconfig.json"; + }; + tmpfiles.rules = [ + "d /etc/target 0750 root root -" + "f /etc/target/saveconfig.json 0640 root root -" + ]; + }; + + environment.systemPackages = [ pkgs.targetcli-fb ]; +} diff --git a/modules/ha/pacemaker-stack.nix b/modules/ha/pacemaker-stack.nix new file mode 100644 index 0000000..8fcab2c --- /dev/null +++ b/modules/ha/pacemaker-stack.nix @@ -0,0 +1,94 @@ +# Pacemaker + Corosync HA stack for NixOS with known-good workarounds. +# +# Issues fixed here (confirmed through live testing on NixOS 25.11): +# +# 1. StateDirectory ownership reset: systemd's StateDirectory=pacemaker +# creates /var/lib/pacemaker owned root:root. pacemaker-based (the CIB +# daemon) runs as the hacluster user and calls pcmk__daemon_can_write, +# which requires the CIB directory to be owned by hacluster or be +# group-writable by haclient. Workaround: remove StateDirectory and let +# ExecStartPre create every required subdirectory with correct ownership. +# +# 2. HA_SBIN_DIR wrong path: ocf-shellfuncs sets HA_SBIN_DIR to the Nix +# store path of the resource-agents derivation's /sbin, which doesn't +# exist. The DRBD OCF agent uses ${HA_SBIN_DIR}/crm_master, so it exits +# 127 without this override. Fix: export HA_SBIN_DIR=/run/current-system/sw/bin. +# +# 3. Broad PATH for OCF agents: the resource executor (pacemaker-execd) runs +# OCF agent scripts as children. NixOS provides no implicit PATH for +# system services; without an explicit PATH the agents can't find ip, ss, +# mount, umount, drbdadm, etc. +# +# 4. FUSER=true: the Filesystem OCF agent calls check_binary $FUSER (default: +# fuser from psmisc), which is not installed. Setting FUSER=true makes +# check_binary succeed (true is always in PATH) and the subsequent +# "$FUSER -km $mountpoint" becomes a no-op. Pair with force_unmount=false +# on each Filesystem resource unless you want lazy unmount behaviour. +{ lib, pkgs, ... }: + +let + ocfBinPath = lib.concatStringsSep ":" [ + "${pkgs.iproute2}/bin" + "${pkgs.iproute2}/sbin" + "${pkgs.iputils}/bin" + "${pkgs.util-linux}/bin" + "${pkgs.util-linux}/sbin" + "${pkgs.gawk}/bin" + "${pkgs.gnugrep}/bin" + "${pkgs.gnused}/bin" + "${pkgs.coreutils}/bin" + "${pkgs.bash}/bin" + "${pkgs.procps}/bin" + "${pkgs.xfsprogs}/bin" + "${pkgs.drbd}/bin" + "${pkgs.python3}/bin" + "/run/current-system/sw/bin" + "/run/current-system/sw/sbin" + "/usr/local/sbin" + "/usr/local/bin" + "/usr/sbin" + "/usr/bin" + "/sbin" + "/bin" + ]; + + # Single pre-start script: schemas symlink + directory ownership. + # Runs before pacemakerd so pacemaker-based finds hacluster-owned dirs. + preStartCmd = "${pkgs.bash}/bin/bash -c '" + + "ln -sfn ${pkgs.pacemaker}/share/pacemaker /var/lib/pacemaker/schemas; " + + "for d in /var/lib/pacemaker /var/lib/pacemaker/cib /var/lib/pacemaker/cores " + + "/var/lib/pacemaker/pengine /var/lib/pacemaker/blackbox " + + "/var/lib/pacemaker/hostcache; do " + + "mkdir -p \"\\$d\" && chown hacluster:pacemaker \"\\$d\" && chmod 2770 \"\\$d\"; " + + "done'"; + + ocfEnv = { + PATH = lib.mkForce ocfBinPath; + OCF_ROOT = "${pkgs.ocf-resource-agents}/usr/lib/ocf"; + HA_SBIN_DIR = "/run/current-system/sw/bin"; + FUSER = "true"; + }; +in +{ + users.groups.haclient = { }; + + services.corosync.enable = true; + services.pacemaker.enable = true; + + systemd.services = { + pacemaker = { + serviceConfig = { + StateDirectory = lib.mkForce ""; + ExecStartPre = lib.mkBefore [ preStartCmd ]; + }; + environment = ocfEnv; + }; + pacemaker-execd.environment = ocfEnv; + }; + + environment.systemPackages = with pkgs; [ + corosync + pacemaker + ocf-resource-agents + ]; +} diff --git a/scripts/ha/acceptance-tests.sh b/scripts/ha/acceptance-tests.sh new file mode 100644 index 0000000..ab3081c --- /dev/null +++ b/scripts/ha/acceptance-tests.sh @@ -0,0 +1,167 @@ +#!/usr/bin/env bash +# acceptance-tests.sh — HA cluster acceptance tests (T1–T7) +# +# Run from a host with SSH access to both HA nodes (or from node1 itself). +# All 7 tests must pass before considering the cluster production-ready. +# Test values below must match variables.nix haServer* values. +set -euo pipefail + +# ── Configuration ───────────────────────────────────────────────────────── +NODE1="ha-server-1" +NODE2="ha-server-2" +NODE1_IP="192.168.2.200" # vars.haServer1Ip +NODE2_IP="192.168.2.201" # vars.haServer2Ip +VIP="192.168.2.202" # vars.haServerVip +XFS_MOUNT="/srv/ha-data" # vars.haStorageRoot +ISCSI_IQN="iqn.2026-01.home.sweet:ha-storage" # vars.haIscsiIqn +# ────────────────────────────────────────────────────────────────────────── + +PASS=0 +FAIL=0 +RESULTS=() + +pass() { echo " PASS: $1"; ((PASS++)); RESULTS+=("PASS $1"); } +fail() { echo " FAIL: $1"; ((FAIL++)); RESULTS+=("FAIL $1"); } + +n1() { ssh -o StrictHostKeyChecking=no -o ConnectTimeout=5 "root@${NODE1_IP}" "$@" 2>/dev/null; } +n2() { ssh -o StrictHostKeyChecking=no -o ConnectTimeout=5 "root@${NODE2_IP}" "$@" 2>/dev/null; } + +echo "════════════════════════════════════════════════════" +echo " HA Cluster Acceptance Tests — $(date '+%Y-%m-%d %H:%M:%S')" +echo "════════════════════════════════════════════════════" + +# ── T1: Corosync quorum established ────────────────────────────────────── +echo "" +echo "[T1] Corosync quorum" +if n1 "corosync-quorumtool -s" 2>/dev/null | grep -q "Quorate:.*Yes"; then + pass "cluster has quorum" +else + fail "cluster does not have quorum — check corosync on both nodes" +fi + +# ── T2: DRBD Primary on node1, Secondary on node2 ──────────────────────── +echo "" +echo "[T2] DRBD roles" +DRBD_ROLE=$(n1 "drbdadm role ha-data" 2>/dev/null || echo "unknown") +if [[ "$DRBD_ROLE" == "Primary/Secondary" || "$DRBD_ROLE" == "Primary" ]]; then + pass "DRBD Primary on $NODE1 ($DRBD_ROLE)" +else + fail "unexpected DRBD role on $NODE1: $DRBD_ROLE (expected Primary/Secondary)" +fi + +DRBD_DSTATE=$(n1 "drbdadm dstate ha-data" 2>/dev/null || echo "unknown") +if echo "$DRBD_DSTATE" | grep -q "UpToDate"; then + pass "DRBD disk state UpToDate ($DRBD_DSTATE)" +else + fail "DRBD disk not UpToDate: $DRBD_DSTATE" +fi + +# ── T3: XFS mounted at haStorageRoot on the Active node ────────────────── +echo "" +echo "[T3] XFS mount" +if n1 "mountpoint -q '${XFS_MOUNT}'" 2>/dev/null; then + pass "XFS mounted at ${XFS_MOUNT} on $NODE1" +else + fail "XFS not mounted at ${XFS_MOUNT} on $NODE1" +fi + +if n2 "mountpoint -q '${XFS_MOUNT}'" 2>/dev/null; then + fail "XFS unexpectedly mounted on $NODE2 (should only be on Active node)" +else + pass "XFS not mounted on $NODE2 (correct — Secondary)" +fi + +# ── T4: iSCSI target visible on both nodes ──────────────────────────────── +echo "" +echo "[T4] iSCSI target" +IQN_COUNT=$(n1 "ls /sys/kernel/config/target/iscsi/ 2>/dev/null | grep -c iqn" || echo "0") +if [[ "$IQN_COUNT" -ge 1 ]]; then + pass "iSCSI IQN active on $NODE1 ($IQN_COUNT target(s))" +else + fail "no iSCSI IQN active on $NODE1" +fi + +# iSCSI discovery from node2 via VIP +if n2 "iscsiadm -m discovery -t sendtargets -p '${VIP}' 2>/dev/null | grep -q '${ISCSI_IQN}'"; then + pass "iSCSI target discoverable from $NODE2 via VIP ${VIP}" +else + fail "iSCSI target not discoverable from $NODE2 via ${VIP}" +fi + +# ── T5: Failover — standby node1, verify resources move to node2 ────────── +echo "" +echo "[T5] Failover (standby $NODE1)" +MYNODE=$(n1 "crm_node -n" 2>/dev/null || echo "") +n1 "crm_standby -N '${MYNODE}' -v on" 2>/dev/null || true +echo " Waiting up to 30 s for resources to move to $NODE2..." +MOVED=false +for i in $(seq 1 30); do + if n2 "mountpoint -q '${XFS_MOUNT}'" 2>/dev/null; then + MOVED=true + echo " Resources moved in ${i}s" + break + fi + sleep 1 +done + +if $MOVED; then + pass "XFS mounted on $NODE2 after failover" + IQN_ON_N2=$(n2 "ls /sys/kernel/config/target/iscsi/ 2>/dev/null | grep -c iqn" || echo "0") + [[ "$IQN_ON_N2" -ge 1 ]] \ + && pass "iSCSI target active on $NODE2 after failover" \ + || fail "iSCSI target NOT active on $NODE2 after failover" +else + fail "XFS did not mount on $NODE2 within 30 s — failover incomplete" +fi + +# ── T6: Data integrity — file written pre-failover readable post-failover ─ +echo "" +echo "[T6] Data integrity" +# Write a test file on node2 (now Active) and verify its content +TEST_FILE="${XFS_MOUNT}/.acceptance-test-$$" +TEST_CONTENT="ha-acceptance-test-$(date +%s)" +n2 "echo '${TEST_CONTENT}' > '${TEST_FILE}'" 2>/dev/null || true +READBACK=$(n2 "cat '${TEST_FILE}' 2>/dev/null" || echo "") +if [[ "$READBACK" == "$TEST_CONTENT" ]]; then + pass "test file written and read back correctly on $NODE2" +else + fail "data integrity check failed (wrote: '$TEST_CONTENT', read: '$READBACK')" +fi +n2 "rm -f '${TEST_FILE}'" 2>/dev/null || true + +# ── T7: Node rejoin — un-standby node1, verify cluster is healthy ───────── +echo "" +echo "[T7] Node rejoin" +n1 "crm_standby -N '${MYNODE}' -v off" 2>/dev/null || true +n1 "crm_resource --cleanup" 2>/dev/null || true +sleep 5 + +ONLINE_NODES=$(n2 "crm_mon -1 2>/dev/null | grep -c 'Online:'" || echo "0") +if n1 "corosync-quorumtool -s 2>/dev/null | grep -q 'Quorate:.*Yes'"; then + pass "$NODE1 rejoined — cluster has quorum" +else + fail "$NODE1 did not rejoin with quorum" +fi + +DRBD_ROLE_AFTER=$(n1 "drbdadm role ha-data" 2>/dev/null || echo "unknown") +if echo "$DRBD_ROLE_AFTER" | grep -q "Secondary"; then + pass "$NODE1 is DRBD Secondary after rejoin ($DRBD_ROLE_AFTER)" +else + fail "unexpected DRBD role on $NODE1 after rejoin: $DRBD_ROLE_AFTER" +fi + +# ── Summary ─────────────────────────────────────────────────────────────── +echo "" +echo "════════════════════════════════════════════════════" +echo " Results: ${PASS} PASS, ${FAIL} FAIL" +echo "════════════════════════════════════════════════════" +for r in "${RESULTS[@]}"; do echo " $r"; done +echo "" + +if [[ "$FAIL" -eq 0 ]]; then + echo "ALL PASS — cluster is production-ready." + exit 0 +else + echo "SOME TESTS FAILED — investigate before deploying." + exit 1 +fi diff --git a/scripts/ha/cluster-enable-stonith.sh b/scripts/ha/cluster-enable-stonith.sh new file mode 100644 index 0000000..6ef47a5 --- /dev/null +++ b/scripts/ha/cluster-enable-stonith.sh @@ -0,0 +1,86 @@ +#!/usr/bin/env bash +# cluster-enable-stonith.sh — enable STONITH fence agent after the fence SSH +# key is deployed to both nodes and authorised on the Proxmox host. +# +# Run from ha-server-1 as root AFTER: +# - /etc/pacemaker/fence_pve_ssh exists on both nodes (chmod +x) +# (copy from scripts/ha/fence-pve-ssh.py) +# - /etc/fence-pve-ssh-key (SSH private key) exists on both nodes +# - The corresponding public key is in authorized_keys on PVE_HOST +# - VMID_NODE1 / VMID_NODE2 filled in below +set -euo pipefail + +# ── Configuration ───────────────────────────────────────────────────────── +NODE1="ha-server-1" +NODE2="ha-server-2" +VMID_NODE1="" # FILL IN: Proxmox VMID for ha-server-1 +VMID_NODE2="" # FILL IN: Proxmox VMID for ha-server-2 +PVE_HOST="pve1.sweet.home" +PVE_USER="wayne" +FENCE_KEY="/etc/fence-pve-ssh-key" +FENCE_SCRIPT="/etc/pacemaker/fence_pve_ssh" +# ────────────────────────────────────────────────────────────────────────── + +log() { echo "[stonith-setup] $*"; } +die() { echo "[stonith-setup] ERROR: $*" >&2; exit 1; } + +[[ $(id -u) -eq 0 ]] || die "must run as root" +[[ -n "$VMID_NODE1" ]] || die "VMID_NODE1 not set — edit this script" +[[ -n "$VMID_NODE2" ]] || die "VMID_NODE2 not set — edit this script" +[[ -f "$FENCE_KEY" ]] || die "fence key not found at $FENCE_KEY" +[[ -f "$FENCE_SCRIPT" ]] || die "fence script not found at $FENCE_SCRIPT" + +log "Verifying fence agent can reach ${PVE_HOST}..." +ssh -i "$FENCE_KEY" -o BatchMode=yes -o ConnectTimeout=10 \ + -o StrictHostKeyChecking=no "${PVE_USER}@${PVE_HOST}" \ + "sudo /usr/sbin/qm list" &>/dev/null \ + || die "Cannot SSH to ${PVE_USER}@${PVE_HOST} — check authorized_keys and sudo" +log "Fence agent SSH connectivity confirmed" + +log "Creating Pacemaker STONITH resources..." +cibadmin --create --scope resources --xml-text " + + + + + + + + + + + + + + +" 2>/dev/null || true + +cibadmin --create --scope resources --xml-text " + + + + + + + + + + + + + + +" 2>/dev/null || true + +log "Enabling STONITH and restoring quorum policy..." +crm_attribute -t crm_config -n stonith-enabled -v true +crm_attribute -t crm_config -n no-quorum-policy -v stop + +log "DRBD fencing mode must also be updated to resource-only (already the" +log "default in cluster-config.nix; confirm with: cat /etc/drbd.d/ha-data.conf)" + +log "Testing fence agent..." +stonith_admin --list-devices && log "Fence devices listed successfully." \ + || warn "stonith_admin --list-devices failed — check config" + +log "STONITH enabled. Cluster is now fully HA." diff --git a/scripts/ha/cluster-init.sh b/scripts/ha/cluster-init.sh new file mode 100644 index 0000000..1aac090 --- /dev/null +++ b/scripts/ha/cluster-init.sh @@ -0,0 +1,284 @@ +#!/usr/bin/env bash +# cluster-init.sh — one-time HA cluster initialisation script +# +# Run ONCE from ha-server-1 as root AFTER both VMs are booted and have SSH +# access. It: +# 1. Generates and distributes the corosync authkey +# 2. Waits for corosync quorum and pacemaker +# 3. Initialises DRBD metadata, promotes node1 to primary +# 4. Creates XFS on /dev/drbd0 and mounts it +# 5. Creates the directory tree and iSCSI LUN backing file +# 6. Configures LIO iSCSI target (file-backed LUN) +# 7. Configures Pacemaker resources: DRBD → XFS → iSCSI → NFS → VIP +# +# Prerequisites: +# - Both VMs booted with the ha-server config (nixos-rebuild done) +# - SSH key access from node1 to root@NODE2_IP +# - VMID_NODE1 / VMID_NODE2 filled in below (needed for STONITH setup; +# cluster starts without STONITH, which you enable separately via +# scripts/ha/cluster-enable-stonith.sh) +# - Run as root on ha-server-1 +set -euo pipefail + +# ── Configuration ───────────────────────────────────────────────────────── +# These must match variables.nix haServer* values and the Proxmox VMID +# assignments. Update before running. +NODE1="ha-server-1" +NODE2="ha-server-2" +NODE1_IP="192.168.2.200" # vars.haServer1Ip +NODE2_IP="192.168.2.201" # vars.haServer2Ip +VIP="192.168.2.202" # vars.haServerVip +XFS_MOUNT="/srv/ha-data" # vars.haStorageRoot +ISCSI_IQN="iqn.2026-01.home.sweet:ha-storage" # vars.haIscsiIqn +ISCSI_LUN_FILE="${XFS_MOUNT}/iscsi-lun.img" +ISCSI_LUN_SIZE="10G" +DRBD_DEVICE="/dev/drbd0" +VMID_NODE1="" # FILL IN: Proxmox VMID for ha-server-1 +VMID_NODE2="" # FILL IN: Proxmox VMID for ha-server-2 +PVE_HOST="pve1.sweet.home" +PVE_USER="wayne" + +# NFS dataset subdirectories to create under XFS_MOUNT. +# Must mirror vars.nfsShares subpath values in variables.nix. +NFS_SUBDIRS=( + "docker/config" + "docker/volumes" + "docker/databases" + "docker/nextcloud-data" + "raspi/volumes" + "proxmox/iso" + "proxmox/lxc" + "pxe-boot/images" +) +# ────────────────────────────────────────────────────────────────────────── + +log() { echo "[cluster-init] $*"; } +die() { echo "[cluster-init] ERROR: $*" >&2; exit 1; } +warn() { echo "[cluster-init] WARNING: $*" >&2; } + +[[ $(id -u) -eq 0 ]] || die "must run as root" +[[ "$(hostname)" == "$NODE1" ]] || die "must run on $NODE1" + +# ── 0. Corosync authkey ─────────────────────────────────────────────────── +AUTHKEY="/etc/corosync/authkey" +mkdir -p /etc/corosync +if [[ ! -f "$AUTHKEY" ]]; then + log "Generating corosync authkey..." + corosync-keygen -k "$AUTHKEY" + chmod 0400 "$AUTHKEY" +fi +log "Distributing authkey to $NODE2..." +ssh "root@${NODE2_IP}" "mkdir -p /etc/corosync" +scp -q "$AUTHKEY" "root@${NODE2_IP}:${AUTHKEY}" +ssh "root@${NODE2_IP}" "chmod 0400 '${AUTHKEY}'" + +log "Restarting corosync on both nodes..." +systemctl restart corosync +ssh "root@${NODE2_IP}" "systemctl restart corosync" +sleep 3 + +# ── 1. Corosync quorum ──────────────────────────────────────────────────── +log "Waiting for corosync quorum..." +for i in $(seq 1 30); do + if corosync-quorumtool -s 2>/dev/null | grep -q 'Quorate:.*Yes'; then + log "Quorum established" + break + fi + [[ $i -eq 30 ]] && die "corosync quorum not established after 60 s" + sleep 2 +done + +log "Waiting for pacemaker..." +for i in $(seq 1 30); do + if crm_mon -1 &>/dev/null; then + log "Pacemaker running" + break + fi + [[ $i -eq 30 ]] && die "pacemaker not running after 60 s" + sleep 2 +done + +# ── 2. DRBD initialisation ──────────────────────────────────────────────── +log "Initialising DRBD metadata on $NODE1..." +if ! drbdadm dstate ha-data 2>/dev/null | grep -q "UpToDate\|Inconsistent\|Diskless"; then + drbdadm create-md ha-data --force +fi + +log "Initialising DRBD metadata on $NODE2..." +ssh "root@${NODE2_IP}" " + if ! drbdadm dstate ha-data 2>/dev/null | grep -q 'UpToDate\|Inconsistent\|Diskless'; then + drbdadm create-md ha-data --force + fi +" + +log "Bringing up DRBD on both nodes..." +drbdadm up ha-data 2>/dev/null || true +ssh "root@${NODE2_IP}" "drbdadm up ha-data 2>/dev/null" || true + +log "Forcing $NODE1 to DRBD Primary for initial sync..." +drbdadm primary ha-data --force + +log "Waiting for DRBD to finish initial sync (this may take several minutes)..." +for i in $(seq 1 300); do + state=$(drbdadm dstate ha-data 2>/dev/null || echo "unknown") + if echo "$state" | grep -q "UpToDate/UpToDate"; then + log "DRBD sync complete: $state" + break + fi + [[ $i -eq 300 ]] && warn "DRBD not UpToDate after 300 s — continuing anyway (check drbdadm status)" + sleep 1 +done + +# ── 3. XFS filesystem ───────────────────────────────────────────────────── +log "Creating XFS on ${DRBD_DEVICE}..." +if ! xfs_info "${DRBD_DEVICE}" &>/dev/null; then + mkfs.xfs -f "${DRBD_DEVICE}" +fi + +log "Mounting ${DRBD_DEVICE} at ${XFS_MOUNT}..." +mkdir -p "${XFS_MOUNT}" +mount "${DRBD_DEVICE}" "${XFS_MOUNT}" + +# ── 4. NFS dataset directories ──────────────────────────────────────────── +log "Creating NFS dataset directories..." +for subdir in "${NFS_SUBDIRS[@]}"; do + mkdir -p "${XFS_MOUNT}/${subdir}" +done + +# ── 5. iSCSI LUN backing file ───────────────────────────────────────────── +log "Creating iSCSI LUN backing file ${ISCSI_LUN_FILE} (${ISCSI_LUN_SIZE})..." +if [[ ! -f "${ISCSI_LUN_FILE}" ]]; then + fallocate -l "${ISCSI_LUN_SIZE}" "${ISCSI_LUN_FILE}" +fi + +# ── 6. LIO iSCSI target ─────────────────────────────────────────────────── +log "Configuring LIO iSCSI target via targetcli..." +targetcli < + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +" + +log "Adding ordering and colocation constraints..." +cibadmin --create --scope constraints --xml-text " + + + + +" + +log "Waiting for resources to start..." +for i in $(seq 1 60); do + if crm_resource -r vip --locate 2>/dev/null | grep -q "running on"; then + log "VIP is up: $(crm_resource -r vip --locate)" + break + fi + [[ $i -eq 60 ]] && { warn "VIP not up after 120 s — check: crm_mon -1"; break; } + sleep 2 +done + +log "" +log "═══════════════════════════════════════════════════════════════" +log " HA cluster initialised." +log "" +log " crm_mon -1 — cluster status" +log " iscsiadm -m discovery -t st -p ${VIP} — verify iSCSI target" +log " showmount -e ${VIP} — verify NFS exports" +log "" +log " To enable STONITH (after deploying fence SSH key):" +log " 1. Fill in VMID_NODE1 / VMID_NODE2 in cluster-enable-stonith.sh" +log " 2. Copy scripts/ha/fence-pve-ssh.py to /etc/pacemaker/fence_pve_ssh" +log " on both nodes (chmod +x)" +log " 3. Generate and distribute the fence SSH key" +log " (see docs or cluster-enable-stonith.sh header)" +log " 4. bash scripts/ha/cluster-enable-stonith.sh" +log "═══════════════════════════════════════════════════════════════" diff --git a/scripts/ha/fence-pve-ssh.py b/scripts/ha/fence-pve-ssh.py new file mode 100644 index 0000000..58c24ed --- /dev/null +++ b/scripts/ha/fence-pve-ssh.py @@ -0,0 +1,179 @@ +#!/usr/bin/env python3 +""" +fence_pve_ssh - Proxmox VE SSH fence agent for Pacemaker. + +Uses SSH to reach the Proxmox host and run 'qm stop/start '. +Deploy to /etc/pacemaker/fence_pve_ssh on both HA nodes (chmod +x). + +Configuration (as pacemaker stonith resource attributes): + pve_host Proxmox host to SSH to (default: pve1.sweet.home) + pve_user SSH user (default: wayne) + key_file SSH private key path (default: /etc/fence-pve-ssh-key) + vmid_node1 VMID for ha-server-1 + vmid_node2 VMID for ha-server-2 + plug Node name to act on (set by pacemaker: ha-server-1 or ha-server-2) + action Action: off|on|reboot|status|list|metadata +""" + +import argparse +import subprocess +import sys +import os + + +METADATA = """ + + Fences a VM on a Proxmox VE host by SSHing to the PVE host and + running qm stop/start. For test use only. + https://proxmox.com + + + + + Fencing action: off|on|reboot|status|list + + + + + Cluster node name to fence + + + + + Proxmox VE host to SSH to + + + + + SSH user on the Proxmox host + + + + + SSH private key file path + + + + + VMID for ha-test-node1 + + + + + VMID for ha-test-node2 + + + + + + + + + + + +""" + + +def parse_args(): + p = argparse.ArgumentParser(add_help=False) + p.add_argument("-a", "--action", default="reboot") + p.add_argument("-n", "--plug") + p.add_argument("--pve-host", default="pve1.sweet.home") + p.add_argument("--pve-user", default="wayne") + p.add_argument("--key-file", default="/etc/fence-pve-ssh-key") + p.add_argument("--vmid-node1") + p.add_argument("--vmid-node2") + # Allow remaining unknown args (pacemaker may pass extra ones) + return p.parse_known_args()[0] + + +def ssh(pve_host, pve_user, key_file, cmd): + result = subprocess.run( + [ + "ssh", + "-i", key_file, + "-o", "StrictHostKeyChecking=no", + "-o", "BatchMode=yes", + "-o", "ConnectTimeout=10", + f"{pve_user}@{pve_host}", + cmd, + ], + capture_output=True, + text=True, + timeout=30, + ) + return result + + +def get_vmid(args): + node = args.plug + if not node: + print("ERROR: --plug not specified", file=sys.stderr) + sys.exit(1) + mapping = { + "ha-server-1": args.vmid_node1, + "ha-server-2": args.vmid_node2, + } + vmid = mapping.get(node) + if not vmid: + print(f"ERROR: unknown node '{node}'", file=sys.stderr) + sys.exit(1) + return vmid + + +def main(): + args = parse_args() + action = args.action.lower() + + if action == "metadata": + print(METADATA) + sys.exit(0) + + if action == "list": + if args.vmid_node1: + print("ha-server-1") + if args.vmid_node2: + print("ha-server-2") + sys.exit(0) + + vmid = get_vmid(args) + + if not os.path.exists(args.key_file): + print(f"ERROR: SSH key not found at {args.key_file}", file=sys.stderr) + sys.exit(1) + + if action in ("off", "reboot"): + print(f"Stopping VM {vmid} ({args.plug}) on {args.pve_host}...") + r = ssh(args.pve_host, args.pve_user, args.key_file, + f"sudo /usr/sbin/qm stop {vmid}") + if r.returncode != 0: + print(f"ERROR stopping VM: {r.stderr}", file=sys.stderr) + sys.exit(1) + print(f"VM {vmid} stopped") + + if action in ("on", "reboot"): + print(f"Starting VM {vmid} ({args.plug}) on {args.pve_host}...") + r = ssh(args.pve_host, args.pve_user, args.key_file, + f"sudo /usr/sbin/qm start {vmid}") + if r.returncode != 0: + print(f"ERROR starting VM: {r.stderr}", file=sys.stderr) + sys.exit(1) + print(f"VM {vmid} started") + + if action == "status": + r = ssh(args.pve_host, args.pve_user, args.key_file, + f"sudo /usr/sbin/qm status {vmid}") + if r.returncode != 0: + print(f"ERROR querying VM status: {r.stderr}", file=sys.stderr) + sys.exit(1) + # qm status returns "status: running" or "status: stopped" + status_line = r.stdout.strip() + print(status_line) + if "stopped" in status_line: + sys.exit(2) # pacemaker interprets exit 2 as "off" + sys.exit(0) # running = exit 0 + + +if __name__ == "__main__": + main() diff --git a/secrets/ha-corosync-authkey b/secrets/ha-corosync-authkey new file mode 100644 index 0000000..74db83c --- /dev/null +++ b/secrets/ha-corosync-authkey @@ -0,0 +1 @@ +STUB: run cluster-init.sh to generate, then: sops -e --input-type binary /etc/corosync/authkey > secrets/ha-corosync-authkey diff --git a/secrets/ha-server-1.yaml b/secrets/ha-server-1.yaml new file mode 100644 index 0000000..f33120f --- /dev/null +++ b/secrets/ha-server-1.yaml @@ -0,0 +1,6 @@ +# STUB — not yet encrypted with sops. +# Bootstrap: +# bash scripts/secrets/sync-host-keys.sh proxmox-ha-server-1 +# sops updatekeys secrets/common.yaml (allows ha-server-1 to decrypt shared secrets) +# sops secrets/ha-server-1.yaml (create with: beszel-token) +beszel-token: REPLACE diff --git a/secrets/ha-server-2.yaml b/secrets/ha-server-2.yaml new file mode 100644 index 0000000..19bbbe9 --- /dev/null +++ b/secrets/ha-server-2.yaml @@ -0,0 +1,6 @@ +# STUB — not yet encrypted with sops. +# Bootstrap: +# bash scripts/secrets/sync-host-keys.sh proxmox-ha-server-2 +# sops updatekeys secrets/common.yaml (allows ha-server-2 to decrypt shared secrets) +# sops secrets/ha-server-2.yaml (create with: beszel-token) +beszel-token: REPLACE diff --git a/variables.nix b/variables.nix index e1e15b1..c59fa68 100644 --- a/variables.nix +++ b/variables.nix @@ -68,6 +68,20 @@ # one-line change. primaryUser = "nixos"; + # HA file server cluster + # haServer1Ip / haServer2Ip: static LAN IPs for both HA nodes (must be + # fixed — DRBD and corosync ring addresses are baked into the NixOS config). + # haServerVip: floating virtual IP managed by Pacemaker's IPaddr2 resource; + # NFS and iSCSI clients connect here regardless of which node is Active. + # Set all three to real values in variables.nix before deploying. + haServer1Host = "ha-server-1"; + haServer2Host = "ha-server-2"; + haServer1Ip = "192.168.2.200"; # TODO: confirm production IP + haServer2Ip = "192.168.2.201"; # TODO: confirm production IP + haServerVip = "192.168.2.202"; # TODO: confirm floating VIP + haStorageRoot = "/srv/ha-data"; # XFS-over-DRBD mount point on the Active node + haIscsiIqn = "iqn.2026-01.home.sweet:ha-storage"; + # Storage storageRoot = "/tank"; # ZFS pool root on `server` @@ -146,11 +160,22 @@ # mountd RPC service (used by showmount/NFSv3 mount protocol). # Mountd listens on a fixed port so the firewall can whitelist it # explicitly rather than opening all of rpcbind's dynamic range. - # All three need both TCP and UDP (modules/build-types/server.nix). + # All three need both TCP and UDP (modules/build-types/server.nix and + # modules/build-types/ha-server.nix). nfsRpcbind = 111; nfsd = 2049; nfsMountd = 20048; + # HA cluster ports opened on ha-server-1 and ha-server-2 + # (modules/build-types/ha-server.nix / modules/ha/cluster-config.nix). + haServerDrbd = 7789; # DRBD replication (TCP) + haServerIscsi = 3260; # iSCSI target (TCP) + haServerCorosync1 = 5404; # Corosync totem ring (UDP) + haServerCorosync2 = 5405; # Corosync totem ring (UDP) + haServerCorosyncCrypto = 5407; # Corosync crypto sync (UDP) + haServerPacemakerRemoted = 3121; # pacemaker-remoted (TCP) + haServerPcsd = 2224; # pcsd cluster daemon (TCP) + # Opened on the docker host's firewall for the Traefik-fronted # container stack (docker-compose config lives in the separate # /home/debian/docker repo, not here): 80/443 are Traefik's own