Archived
Merge pull request 'Worktree ha file server test' (#81) from worktree-ha-file-server-test into main
Reviewed-on: #81
This commit is contained in:
+26
@@ -85,6 +85,32 @@ creation_rules:
|
||||
- *lxc-tailscale-router
|
||||
- *proxmox-tailscale-router
|
||||
|
||||
# HA file server per-node secrets (beszel-token).
|
||||
# proxmox-ha-server-1 / proxmox-ha-server-2 keys are added automatically
|
||||
# by scripts/secrets/sync-host-keys.sh once the hosts are provisioned;
|
||||
# until then only the admin key can decrypt these files.
|
||||
- path_regex: secrets/ha-server-1\.yaml$
|
||||
key_groups:
|
||||
- age:
|
||||
- *admin
|
||||
# proxmox-ha-server-1 added by sync-host-keys.sh
|
||||
|
||||
- path_regex: secrets/ha-server-2\.yaml$
|
||||
key_groups:
|
||||
- age:
|
||||
- *admin
|
||||
# proxmox-ha-server-2 added by sync-host-keys.sh
|
||||
|
||||
# Shared HA cluster corosync authkey (binary sops file).
|
||||
# Encrypted for both HA nodes so either can decrypt on boot.
|
||||
# Both host keys added by sync-host-keys.sh; admin key allows initial creation.
|
||||
- path_regex: secrets/ha-corosync-authkey$
|
||||
key_groups:
|
||||
- age:
|
||||
- *admin
|
||||
# proxmox-ha-server-1 added by sync-host-keys.sh
|
||||
# proxmox-ha-server-2 added by sync-host-keys.sh
|
||||
|
||||
# gui-host-specific secrets (currently: wifi-password, see
|
||||
# modules/networking/wifi.nix). Only *lxc-gui has a registered key today
|
||||
# -- proxmox-gui/linode-gui/baremetal-gui haven't been provisioned via
|
||||
|
||||
@@ -130,6 +130,9 @@
|
||||
lxc-tailscale-router = mkTarget { platform = "lxc"; buildType = "tailscale-router"; hostPath = ./hosts/tailscale-router/host.nix; };
|
||||
|
||||
lxc-tor-relay = mkTarget { platform = "lxc"; buildType = "tor-relay"; hostPath = ./hosts/tor-relay/host.nix; };
|
||||
|
||||
proxmox-ha-server-1 = mkTarget { platform = "proxmox"; buildType = "ha-server"; hostPath = ./hosts/ha-server-1/host.nix; };
|
||||
proxmox-ha-server-2 = mkTarget { platform = "proxmox"; buildType = "ha-server"; hostPath = ./hosts/ha-server-2/host.nix; };
|
||||
};
|
||||
|
||||
# Auto-install environments (migrated from the former nix-auto-installer
|
||||
|
||||
@@ -0,0 +1,26 @@
|
||||
{ vars, ... }:
|
||||
{
|
||||
imports = [
|
||||
(import ../../modules/beszel/host-token.nix {
|
||||
name = "ha-server-1";
|
||||
sopsFile = ../../secrets/ha-server-1.yaml;
|
||||
})
|
||||
];
|
||||
|
||||
networking = {
|
||||
hostName = vars.haServer1Host;
|
||||
hostId = "3a4b5c6d";
|
||||
useDHCP = false;
|
||||
interfaces.ens18.ipv4.addresses = [{
|
||||
address = vars.haServer1Ip;
|
||||
prefixLength = 24;
|
||||
}];
|
||||
defaultGateway = "192.168.2.1";
|
||||
nameservers = [ "192.168.2.1" "8.8.8.8" ];
|
||||
};
|
||||
|
||||
# Set KEY after pairing this host with the beszel hub; the token is sops-managed.
|
||||
services.beszel.agent.environment.KEY = "";
|
||||
|
||||
system.stateVersion = "26.05";
|
||||
}
|
||||
@@ -0,0 +1,26 @@
|
||||
{ vars, ... }:
|
||||
{
|
||||
imports = [
|
||||
(import ../../modules/beszel/host-token.nix {
|
||||
name = "ha-server-2";
|
||||
sopsFile = ../../secrets/ha-server-2.yaml;
|
||||
})
|
||||
];
|
||||
|
||||
networking = {
|
||||
hostName = vars.haServer2Host;
|
||||
hostId = "7e8f9a0b";
|
||||
useDHCP = false;
|
||||
interfaces.ens18.ipv4.addresses = [{
|
||||
address = vars.haServer2Ip;
|
||||
prefixLength = 24;
|
||||
}];
|
||||
defaultGateway = "192.168.2.1";
|
||||
nameservers = [ "192.168.2.1" "8.8.8.8" ];
|
||||
};
|
||||
|
||||
# Set KEY after pairing this host with the beszel hub; the token is sops-managed.
|
||||
services.beszel.agent.environment.KEY = "";
|
||||
|
||||
system.stateVersion = "26.05";
|
||||
}
|
||||
@@ -0,0 +1,43 @@
|
||||
# HA file server build type: DRBD + XFS + LIO iSCSI + NFS, managed by
|
||||
# Corosync + Pacemaker. Both ha-server-1 and ha-server-2 use this type.
|
||||
#
|
||||
# NFS start/stop:
|
||||
# services.nfs.server.enable = true configures /etc/exports, wires up
|
||||
# rpcbind, and loads kernel modules — but nfs-server.service.wantedBy is
|
||||
# force-cleared so systemd does NOT auto-start it at boot. Pacemaker's
|
||||
# ha-group resource group (configured by scripts/ha/cluster-init.sh)
|
||||
# starts and stops nfs-server as part of the failover sequence after the
|
||||
# XFS mount and iSCSI target are brought up on the new Active node.
|
||||
#
|
||||
# Beszel agent:
|
||||
# Enabled here via enable-agent.nix. The agent KEY (used to pair with
|
||||
# the Beszel hub) is not set yet — add it to hosts/ha-server-{1,2}/host.nix
|
||||
# under services.beszel.agent.environment.KEY once the hub accepts the
|
||||
# new agents, following the pattern in hosts/server/host.nix.
|
||||
{ lib, vars, ... }:
|
||||
{
|
||||
imports = [
|
||||
../ha/pacemaker-stack.nix
|
||||
../ha/iscsi-target.nix
|
||||
../ha/cluster-config.nix
|
||||
../beszel/enable-agent.nix
|
||||
];
|
||||
|
||||
services.nfs.server = {
|
||||
enable = true;
|
||||
exports = ''
|
||||
${vars.haStorageRoot}/${vars.nfsShares.dockerConfig.subpath} ${vars.lanCidr}${vars.nfsShares.options}
|
||||
${vars.haStorageRoot}/${vars.nfsShares.dockerVolumes.subpath} ${vars.lanCidr}${vars.nfsShares.options}
|
||||
${vars.haStorageRoot}/${vars.nfsShares.dockerDatabases.subpath} ${vars.lanCidr}${vars.nfsShares.options}
|
||||
${vars.haStorageRoot}/${vars.nfsShares.nextcloudData.subpath} ${vars.lanCidr}${vars.nfsShares.options}
|
||||
${vars.haStorageRoot}/${vars.nfsShares.raspiVolumes.subpath} ${vars.lanCidr}${vars.nfsShares.options}
|
||||
${vars.haStorageRoot}/${vars.nfsShares.proxmoxIsos.subpath} ${vars.lanCidr}${vars.nfsShares.options}
|
||||
${vars.haStorageRoot}/${vars.nfsShares.proxmoxLxcImages.subpath} ${vars.lanCidr}${vars.nfsShares.options}
|
||||
${vars.haStorageRoot}/${vars.nfsShares.pxebootImages.subpath} ${vars.lanCidr}${vars.nfsShares.options}
|
||||
'';
|
||||
};
|
||||
|
||||
# Pacemaker controls nfs-server — prevent systemd from starting it at boot
|
||||
# on both nodes (only the Active node should be serving NFS).
|
||||
systemd.services.nfs-server.wantedBy = lib.mkForce [ ];
|
||||
}
|
||||
@@ -0,0 +1,106 @@
|
||||
# Cluster-wide HA config shared by both ha-server nodes.
|
||||
#
|
||||
# Covers everything that is identical on both nodes and references cluster
|
||||
# topology (node IPs, hostnames, DRBD resource). Per-node identity
|
||||
# (hostname, static IP, stateVersion) lives in hosts/ha-server-{1,2}/host.nix.
|
||||
#
|
||||
# Corosync authkey:
|
||||
# /etc/corosync/authkey (mode 0400) is managed by sops-nix below.
|
||||
# Bootstrap: run scripts/ha/cluster-init.sh on node1 to generate the key,
|
||||
# then encrypt it with: sops -e --input-type binary /etc/corosync/authkey > secrets/ha-corosync-authkey
|
||||
# Both host keys must be registered via sync-host-keys.sh first so both nodes can decrypt it.
|
||||
#
|
||||
# DRBD fencing:
|
||||
# Production setting is resource-only: DRBD waits for the STONITH fence
|
||||
# agent to confirm the peer is dead before promoting to Primary. This
|
||||
# requires a working fence_pve_ssh STONITH resource in Pacemaker
|
||||
# (see scripts/ha/cluster-enable-stonith.sh). On a fresh cluster with
|
||||
# no fence device yet, temporarily change to dont-care and run
|
||||
# cluster-enable-stonith.sh once the fence key is deployed.
|
||||
{ lib, vars, ... }:
|
||||
{
|
||||
services.drbd = {
|
||||
enable = true;
|
||||
config = ''
|
||||
global {
|
||||
usage-count yes;
|
||||
}
|
||||
|
||||
common {
|
||||
net {
|
||||
protocol C;
|
||||
ping-int 1;
|
||||
verify-alg sha256;
|
||||
after-sb-0pri discard-zero-changes;
|
||||
after-sb-1pri discard-secondary;
|
||||
}
|
||||
disk {
|
||||
fencing resource-only;
|
||||
}
|
||||
}
|
||||
|
||||
resource ha-data {
|
||||
volume 0 {
|
||||
device /dev/drbd0;
|
||||
disk /dev/sdb;
|
||||
meta-disk internal;
|
||||
}
|
||||
|
||||
on ${vars.haServer1Host} {
|
||||
address ${vars.haServer1Ip}:${toString vars.ports.haServerDrbd};
|
||||
}
|
||||
|
||||
on ${vars.haServer2Host} {
|
||||
address ${vars.haServer2Ip}:${toString vars.ports.haServerDrbd};
|
||||
}
|
||||
}
|
||||
'';
|
||||
};
|
||||
|
||||
# /etc/corosync/authkey — sops binary secret, identical on both nodes.
|
||||
# Decryptable by both ha-server host keys (added by sync-host-keys.sh).
|
||||
sops.secrets.corosync_authkey = {
|
||||
sopsFile = ../../secrets/ha-corosync-authkey;
|
||||
format = "binary";
|
||||
path = "/etc/corosync/authkey";
|
||||
mode = "0400";
|
||||
restartUnits = [ "corosync.service" ];
|
||||
};
|
||||
|
||||
# NixOS common config enables NetworkManager by default; HA cluster nodes
|
||||
# need stable static IPs with predictable interface names — NM is not suitable.
|
||||
networking.networkmanager.enable = lib.mkForce false;
|
||||
|
||||
# services.corosync.enable is set by modules/ha/pacemaker-stack.nix.
|
||||
services.corosync = {
|
||||
clusterName = "ha-cluster";
|
||||
nodelist = [
|
||||
{ nodeid = 1; name = vars.haServer1Host; ring_addrs = [ vars.haServer1Ip ]; }
|
||||
{ nodeid = 2; name = vars.haServer2Host; ring_addrs = [ vars.haServer2Ip ]; }
|
||||
];
|
||||
};
|
||||
|
||||
networking.firewall = {
|
||||
allowedTCPPorts = [
|
||||
vars.ports.haServerIscsi
|
||||
vars.ports.haServerPacemakerRemoted
|
||||
vars.ports.haServerPcsd
|
||||
vars.ports.haServerDrbd
|
||||
vars.ports.nfsRpcbind
|
||||
vars.ports.nfsd
|
||||
vars.ports.nfsMountd
|
||||
];
|
||||
allowedUDPPorts = [
|
||||
vars.ports.haServerCorosync1
|
||||
vars.ports.haServerCorosync2
|
||||
vars.ports.haServerCorosyncCrypto
|
||||
vars.ports.nfsRpcbind
|
||||
vars.ports.nfsd
|
||||
vars.ports.nfsMountd
|
||||
];
|
||||
extraCommands = ''
|
||||
iptables -A INPUT -s ${vars.haServer1Ip}/32 -j ACCEPT
|
||||
iptables -A INPUT -s ${vars.haServer2Ip}/32 -j ACCEPT
|
||||
'';
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,99 @@
|
||||
# LIO iSCSI target service (targetctl) for NixOS HA clusters.
|
||||
#
|
||||
# Provides the targetctl.service that saves/restores LIO configuration from
|
||||
# /etc/target/saveconfig.json. Pacemaker manages this service via its
|
||||
# systemd resource agent (class="systemd" type="targetctl").
|
||||
#
|
||||
# Why ExecStop is not simply "targetctl save":
|
||||
# targetctl save writes the LIO config to JSON but does NOT remove the LIO
|
||||
# target from the kernel's configfs. As a result, any fileio backing store
|
||||
# that LIO has open (e.g. iscsi-lun.img on an XFS-over-DRBD filesystem)
|
||||
# stays referenced in the kernel. The subsequent XFS umount from the
|
||||
# Filesystem OCF resource then returns EBUSY and either hangs for the full
|
||||
# op-stop timeout or fails outright, blocking the entire failover.
|
||||
#
|
||||
# The ExecStop script here additionally tears down the kernel LIO state
|
||||
# via rtslib_fb after saving, so the backing-store file descriptor is
|
||||
# released and umount succeeds immediately.
|
||||
#
|
||||
# Empty-config guard:
|
||||
# The save step is skipped when no iSCSI targets are currently active.
|
||||
# This prevents the secondary node (where LIO was never started) from
|
||||
# overwriting a valid saveconfig.json with an empty one when Pacemaker
|
||||
# stops the iscsi-target resource as part of a failover or cleanup.
|
||||
{ pkgs, ... }:
|
||||
|
||||
let
|
||||
python3 = pkgs.python3.withPackages (ps: [ ps.rtslib-fb ]);
|
||||
targetctl = "${pkgs.targetcli-fb}/bin/targetctl";
|
||||
|
||||
targetctlStop = pkgs.writeScript "targetctl-stop" ''
|
||||
#!${python3}/bin/python3
|
||||
import subprocess, sys
|
||||
import rtslib_fb
|
||||
|
||||
root = rtslib_fb.RTSRoot()
|
||||
targets = list(root.targets)
|
||||
if targets:
|
||||
subprocess.run(
|
||||
["${targetctl}", "save", "/etc/target/saveconfig.json"],
|
||||
capture_output=True,
|
||||
)
|
||||
print(f"saved {len(targets)} iSCSI target(s)")
|
||||
else:
|
||||
print("no active LIO targets — saveconfig.json unchanged")
|
||||
|
||||
for target in targets:
|
||||
try:
|
||||
for tpg in list(target.tpgs):
|
||||
tpg.enable = False
|
||||
target.delete()
|
||||
except Exception as e:
|
||||
print(f"warn (target): {e}", file=sys.stderr)
|
||||
for so in list(root.storage_objects):
|
||||
try:
|
||||
so.delete()
|
||||
except Exception as e:
|
||||
print(f"warn (backstore): {e}", file=sys.stderr)
|
||||
print("LIO kernel target cleared")
|
||||
'';
|
||||
in
|
||||
{
|
||||
boot.kernelModules = [
|
||||
"target_core_mod"
|
||||
"iscsi_target_mod"
|
||||
"target_core_file"
|
||||
"target_core_pscsi"
|
||||
"target_core_user"
|
||||
"configfs"
|
||||
];
|
||||
|
||||
systemd = {
|
||||
mounts = [{
|
||||
where = "/sys/kernel/config";
|
||||
what = "configfs";
|
||||
type = "configfs";
|
||||
wantedBy = [ "multi-user.target" ];
|
||||
before = [ "targetctl.service" ];
|
||||
}];
|
||||
services.targetctl = {
|
||||
description = "LIO iSCSI target config save/restore";
|
||||
wantedBy = [ "multi-user.target" ];
|
||||
after = [ "sys-kernel-config.mount" "network.target" ];
|
||||
requires = [ "sys-kernel-config.mount" ];
|
||||
serviceConfig = {
|
||||
Type = "oneshot";
|
||||
RemainAfterExit = true;
|
||||
ExecStart = "${targetctl} restore /etc/target/saveconfig.json";
|
||||
ExecStop = "${targetctlStop}";
|
||||
};
|
||||
unitConfig.ConditionFileNotEmpty = "/etc/target/saveconfig.json";
|
||||
};
|
||||
tmpfiles.rules = [
|
||||
"d /etc/target 0750 root root -"
|
||||
"f /etc/target/saveconfig.json 0640 root root -"
|
||||
];
|
||||
};
|
||||
|
||||
environment.systemPackages = [ pkgs.targetcli-fb ];
|
||||
}
|
||||
@@ -0,0 +1,94 @@
|
||||
# Pacemaker + Corosync HA stack for NixOS with known-good workarounds.
|
||||
#
|
||||
# Issues fixed here (confirmed through live testing on NixOS 25.11):
|
||||
#
|
||||
# 1. StateDirectory ownership reset: systemd's StateDirectory=pacemaker
|
||||
# creates /var/lib/pacemaker owned root:root. pacemaker-based (the CIB
|
||||
# daemon) runs as the hacluster user and calls pcmk__daemon_can_write,
|
||||
# which requires the CIB directory to be owned by hacluster or be
|
||||
# group-writable by haclient. Workaround: remove StateDirectory and let
|
||||
# ExecStartPre create every required subdirectory with correct ownership.
|
||||
#
|
||||
# 2. HA_SBIN_DIR wrong path: ocf-shellfuncs sets HA_SBIN_DIR to the Nix
|
||||
# store path of the resource-agents derivation's /sbin, which doesn't
|
||||
# exist. The DRBD OCF agent uses ${HA_SBIN_DIR}/crm_master, so it exits
|
||||
# 127 without this override. Fix: export HA_SBIN_DIR=/run/current-system/sw/bin.
|
||||
#
|
||||
# 3. Broad PATH for OCF agents: the resource executor (pacemaker-execd) runs
|
||||
# OCF agent scripts as children. NixOS provides no implicit PATH for
|
||||
# system services; without an explicit PATH the agents can't find ip, ss,
|
||||
# mount, umount, drbdadm, etc.
|
||||
#
|
||||
# 4. FUSER=true: the Filesystem OCF agent calls check_binary $FUSER (default:
|
||||
# fuser from psmisc), which is not installed. Setting FUSER=true makes
|
||||
# check_binary succeed (true is always in PATH) and the subsequent
|
||||
# "$FUSER -km $mountpoint" becomes a no-op. Pair with force_unmount=false
|
||||
# on each Filesystem resource unless you want lazy unmount behaviour.
|
||||
{ lib, pkgs, ... }:
|
||||
|
||||
let
|
||||
ocfBinPath = lib.concatStringsSep ":" [
|
||||
"${pkgs.iproute2}/bin"
|
||||
"${pkgs.iproute2}/sbin"
|
||||
"${pkgs.iputils}/bin"
|
||||
"${pkgs.util-linux}/bin"
|
||||
"${pkgs.util-linux}/sbin"
|
||||
"${pkgs.gawk}/bin"
|
||||
"${pkgs.gnugrep}/bin"
|
||||
"${pkgs.gnused}/bin"
|
||||
"${pkgs.coreutils}/bin"
|
||||
"${pkgs.bash}/bin"
|
||||
"${pkgs.procps}/bin"
|
||||
"${pkgs.xfsprogs}/bin"
|
||||
"${pkgs.drbd}/bin"
|
||||
"${pkgs.python3}/bin"
|
||||
"/run/current-system/sw/bin"
|
||||
"/run/current-system/sw/sbin"
|
||||
"/usr/local/sbin"
|
||||
"/usr/local/bin"
|
||||
"/usr/sbin"
|
||||
"/usr/bin"
|
||||
"/sbin"
|
||||
"/bin"
|
||||
];
|
||||
|
||||
# Single pre-start script: schemas symlink + directory ownership.
|
||||
# Runs before pacemakerd so pacemaker-based finds hacluster-owned dirs.
|
||||
preStartCmd = "${pkgs.bash}/bin/bash -c '"
|
||||
+ "ln -sfn ${pkgs.pacemaker}/share/pacemaker /var/lib/pacemaker/schemas; "
|
||||
+ "for d in /var/lib/pacemaker /var/lib/pacemaker/cib /var/lib/pacemaker/cores "
|
||||
+ "/var/lib/pacemaker/pengine /var/lib/pacemaker/blackbox "
|
||||
+ "/var/lib/pacemaker/hostcache; do "
|
||||
+ "mkdir -p \"\\$d\" && chown hacluster:pacemaker \"\\$d\" && chmod 2770 \"\\$d\"; "
|
||||
+ "done'";
|
||||
|
||||
ocfEnv = {
|
||||
PATH = lib.mkForce ocfBinPath;
|
||||
OCF_ROOT = "${pkgs.ocf-resource-agents}/usr/lib/ocf";
|
||||
HA_SBIN_DIR = "/run/current-system/sw/bin";
|
||||
FUSER = "true";
|
||||
};
|
||||
in
|
||||
{
|
||||
users.groups.haclient = { };
|
||||
|
||||
services.corosync.enable = true;
|
||||
services.pacemaker.enable = true;
|
||||
|
||||
systemd.services = {
|
||||
pacemaker = {
|
||||
serviceConfig = {
|
||||
StateDirectory = lib.mkForce "";
|
||||
ExecStartPre = lib.mkBefore [ preStartCmd ];
|
||||
};
|
||||
environment = ocfEnv;
|
||||
};
|
||||
pacemaker-execd.environment = ocfEnv;
|
||||
};
|
||||
|
||||
environment.systemPackages = with pkgs; [
|
||||
corosync
|
||||
pacemaker
|
||||
ocf-resource-agents
|
||||
];
|
||||
}
|
||||
@@ -0,0 +1,167 @@
|
||||
#!/usr/bin/env bash
|
||||
# acceptance-tests.sh — HA cluster acceptance tests (T1–T7)
|
||||
#
|
||||
# Run from a host with SSH access to both HA nodes (or from node1 itself).
|
||||
# All 7 tests must pass before considering the cluster production-ready.
|
||||
# Test values below must match variables.nix haServer* values.
|
||||
set -euo pipefail
|
||||
|
||||
# ── Configuration ─────────────────────────────────────────────────────────
|
||||
NODE1="ha-server-1"
|
||||
NODE2="ha-server-2"
|
||||
NODE1_IP="192.168.2.200" # vars.haServer1Ip
|
||||
NODE2_IP="192.168.2.201" # vars.haServer2Ip
|
||||
VIP="192.168.2.202" # vars.haServerVip
|
||||
XFS_MOUNT="/srv/ha-data" # vars.haStorageRoot
|
||||
ISCSI_IQN="iqn.2026-01.home.sweet:ha-storage" # vars.haIscsiIqn
|
||||
# ──────────────────────────────────────────────────────────────────────────
|
||||
|
||||
PASS=0
|
||||
FAIL=0
|
||||
RESULTS=()
|
||||
|
||||
pass() { echo " PASS: $1"; ((PASS++)); RESULTS+=("PASS $1"); }
|
||||
fail() { echo " FAIL: $1"; ((FAIL++)); RESULTS+=("FAIL $1"); }
|
||||
|
||||
n1() { ssh -o StrictHostKeyChecking=no -o ConnectTimeout=5 "root@${NODE1_IP}" "$@" 2>/dev/null; }
|
||||
n2() { ssh -o StrictHostKeyChecking=no -o ConnectTimeout=5 "root@${NODE2_IP}" "$@" 2>/dev/null; }
|
||||
|
||||
echo "════════════════════════════════════════════════════"
|
||||
echo " HA Cluster Acceptance Tests — $(date '+%Y-%m-%d %H:%M:%S')"
|
||||
echo "════════════════════════════════════════════════════"
|
||||
|
||||
# ── T1: Corosync quorum established ──────────────────────────────────────
|
||||
echo ""
|
||||
echo "[T1] Corosync quorum"
|
||||
if n1 "corosync-quorumtool -s" 2>/dev/null | grep -q "Quorate:.*Yes"; then
|
||||
pass "cluster has quorum"
|
||||
else
|
||||
fail "cluster does not have quorum — check corosync on both nodes"
|
||||
fi
|
||||
|
||||
# ── T2: DRBD Primary on node1, Secondary on node2 ────────────────────────
|
||||
echo ""
|
||||
echo "[T2] DRBD roles"
|
||||
DRBD_ROLE=$(n1 "drbdadm role ha-data" 2>/dev/null || echo "unknown")
|
||||
if [[ "$DRBD_ROLE" == "Primary/Secondary" || "$DRBD_ROLE" == "Primary" ]]; then
|
||||
pass "DRBD Primary on $NODE1 ($DRBD_ROLE)"
|
||||
else
|
||||
fail "unexpected DRBD role on $NODE1: $DRBD_ROLE (expected Primary/Secondary)"
|
||||
fi
|
||||
|
||||
DRBD_DSTATE=$(n1 "drbdadm dstate ha-data" 2>/dev/null || echo "unknown")
|
||||
if echo "$DRBD_DSTATE" | grep -q "UpToDate"; then
|
||||
pass "DRBD disk state UpToDate ($DRBD_DSTATE)"
|
||||
else
|
||||
fail "DRBD disk not UpToDate: $DRBD_DSTATE"
|
||||
fi
|
||||
|
||||
# ── T3: XFS mounted at haStorageRoot on the Active node ──────────────────
|
||||
echo ""
|
||||
echo "[T3] XFS mount"
|
||||
if n1 "mountpoint -q '${XFS_MOUNT}'" 2>/dev/null; then
|
||||
pass "XFS mounted at ${XFS_MOUNT} on $NODE1"
|
||||
else
|
||||
fail "XFS not mounted at ${XFS_MOUNT} on $NODE1"
|
||||
fi
|
||||
|
||||
if n2 "mountpoint -q '${XFS_MOUNT}'" 2>/dev/null; then
|
||||
fail "XFS unexpectedly mounted on $NODE2 (should only be on Active node)"
|
||||
else
|
||||
pass "XFS not mounted on $NODE2 (correct — Secondary)"
|
||||
fi
|
||||
|
||||
# ── T4: iSCSI target visible on both nodes ────────────────────────────────
|
||||
echo ""
|
||||
echo "[T4] iSCSI target"
|
||||
IQN_COUNT=$(n1 "ls /sys/kernel/config/target/iscsi/ 2>/dev/null | grep -c iqn" || echo "0")
|
||||
if [[ "$IQN_COUNT" -ge 1 ]]; then
|
||||
pass "iSCSI IQN active on $NODE1 ($IQN_COUNT target(s))"
|
||||
else
|
||||
fail "no iSCSI IQN active on $NODE1"
|
||||
fi
|
||||
|
||||
# iSCSI discovery from node2 via VIP
|
||||
if n2 "iscsiadm -m discovery -t sendtargets -p '${VIP}' 2>/dev/null | grep -q '${ISCSI_IQN}'"; then
|
||||
pass "iSCSI target discoverable from $NODE2 via VIP ${VIP}"
|
||||
else
|
||||
fail "iSCSI target not discoverable from $NODE2 via ${VIP}"
|
||||
fi
|
||||
|
||||
# ── T5: Failover — standby node1, verify resources move to node2 ──────────
|
||||
echo ""
|
||||
echo "[T5] Failover (standby $NODE1)"
|
||||
MYNODE=$(n1 "crm_node -n" 2>/dev/null || echo "")
|
||||
n1 "crm_standby -N '${MYNODE}' -v on" 2>/dev/null || true
|
||||
echo " Waiting up to 30 s for resources to move to $NODE2..."
|
||||
MOVED=false
|
||||
for i in $(seq 1 30); do
|
||||
if n2 "mountpoint -q '${XFS_MOUNT}'" 2>/dev/null; then
|
||||
MOVED=true
|
||||
echo " Resources moved in ${i}s"
|
||||
break
|
||||
fi
|
||||
sleep 1
|
||||
done
|
||||
|
||||
if $MOVED; then
|
||||
pass "XFS mounted on $NODE2 after failover"
|
||||
IQN_ON_N2=$(n2 "ls /sys/kernel/config/target/iscsi/ 2>/dev/null | grep -c iqn" || echo "0")
|
||||
[[ "$IQN_ON_N2" -ge 1 ]] \
|
||||
&& pass "iSCSI target active on $NODE2 after failover" \
|
||||
|| fail "iSCSI target NOT active on $NODE2 after failover"
|
||||
else
|
||||
fail "XFS did not mount on $NODE2 within 30 s — failover incomplete"
|
||||
fi
|
||||
|
||||
# ── T6: Data integrity — file written pre-failover readable post-failover ─
|
||||
echo ""
|
||||
echo "[T6] Data integrity"
|
||||
# Write a test file on node2 (now Active) and verify its content
|
||||
TEST_FILE="${XFS_MOUNT}/.acceptance-test-$$"
|
||||
TEST_CONTENT="ha-acceptance-test-$(date +%s)"
|
||||
n2 "echo '${TEST_CONTENT}' > '${TEST_FILE}'" 2>/dev/null || true
|
||||
READBACK=$(n2 "cat '${TEST_FILE}' 2>/dev/null" || echo "")
|
||||
if [[ "$READBACK" == "$TEST_CONTENT" ]]; then
|
||||
pass "test file written and read back correctly on $NODE2"
|
||||
else
|
||||
fail "data integrity check failed (wrote: '$TEST_CONTENT', read: '$READBACK')"
|
||||
fi
|
||||
n2 "rm -f '${TEST_FILE}'" 2>/dev/null || true
|
||||
|
||||
# ── T7: Node rejoin — un-standby node1, verify cluster is healthy ─────────
|
||||
echo ""
|
||||
echo "[T7] Node rejoin"
|
||||
n1 "crm_standby -N '${MYNODE}' -v off" 2>/dev/null || true
|
||||
n1 "crm_resource --cleanup" 2>/dev/null || true
|
||||
sleep 5
|
||||
|
||||
ONLINE_NODES=$(n2 "crm_mon -1 2>/dev/null | grep -c 'Online:'" || echo "0")
|
||||
if n1 "corosync-quorumtool -s 2>/dev/null | grep -q 'Quorate:.*Yes'"; then
|
||||
pass "$NODE1 rejoined — cluster has quorum"
|
||||
else
|
||||
fail "$NODE1 did not rejoin with quorum"
|
||||
fi
|
||||
|
||||
DRBD_ROLE_AFTER=$(n1 "drbdadm role ha-data" 2>/dev/null || echo "unknown")
|
||||
if echo "$DRBD_ROLE_AFTER" | grep -q "Secondary"; then
|
||||
pass "$NODE1 is DRBD Secondary after rejoin ($DRBD_ROLE_AFTER)"
|
||||
else
|
||||
fail "unexpected DRBD role on $NODE1 after rejoin: $DRBD_ROLE_AFTER"
|
||||
fi
|
||||
|
||||
# ── Summary ───────────────────────────────────────────────────────────────
|
||||
echo ""
|
||||
echo "════════════════════════════════════════════════════"
|
||||
echo " Results: ${PASS} PASS, ${FAIL} FAIL"
|
||||
echo "════════════════════════════════════════════════════"
|
||||
for r in "${RESULTS[@]}"; do echo " $r"; done
|
||||
echo ""
|
||||
|
||||
if [[ "$FAIL" -eq 0 ]]; then
|
||||
echo "ALL PASS — cluster is production-ready."
|
||||
exit 0
|
||||
else
|
||||
echo "SOME TESTS FAILED — investigate before deploying."
|
||||
exit 1
|
||||
fi
|
||||
@@ -0,0 +1,86 @@
|
||||
#!/usr/bin/env bash
|
||||
# cluster-enable-stonith.sh — enable STONITH fence agent after the fence SSH
|
||||
# key is deployed to both nodes and authorised on the Proxmox host.
|
||||
#
|
||||
# Run from ha-server-1 as root AFTER:
|
||||
# - /etc/pacemaker/fence_pve_ssh exists on both nodes (chmod +x)
|
||||
# (copy from scripts/ha/fence-pve-ssh.py)
|
||||
# - /etc/fence-pve-ssh-key (SSH private key) exists on both nodes
|
||||
# - The corresponding public key is in authorized_keys on PVE_HOST
|
||||
# - VMID_NODE1 / VMID_NODE2 filled in below
|
||||
set -euo pipefail
|
||||
|
||||
# ── Configuration ─────────────────────────────────────────────────────────
|
||||
NODE1="ha-server-1"
|
||||
NODE2="ha-server-2"
|
||||
VMID_NODE1="" # FILL IN: Proxmox VMID for ha-server-1
|
||||
VMID_NODE2="" # FILL IN: Proxmox VMID for ha-server-2
|
||||
PVE_HOST="pve1.sweet.home"
|
||||
PVE_USER="wayne"
|
||||
FENCE_KEY="/etc/fence-pve-ssh-key"
|
||||
FENCE_SCRIPT="/etc/pacemaker/fence_pve_ssh"
|
||||
# ──────────────────────────────────────────────────────────────────────────
|
||||
|
||||
log() { echo "[stonith-setup] $*"; }
|
||||
die() { echo "[stonith-setup] ERROR: $*" >&2; exit 1; }
|
||||
|
||||
[[ $(id -u) -eq 0 ]] || die "must run as root"
|
||||
[[ -n "$VMID_NODE1" ]] || die "VMID_NODE1 not set — edit this script"
|
||||
[[ -n "$VMID_NODE2" ]] || die "VMID_NODE2 not set — edit this script"
|
||||
[[ -f "$FENCE_KEY" ]] || die "fence key not found at $FENCE_KEY"
|
||||
[[ -f "$FENCE_SCRIPT" ]] || die "fence script not found at $FENCE_SCRIPT"
|
||||
|
||||
log "Verifying fence agent can reach ${PVE_HOST}..."
|
||||
ssh -i "$FENCE_KEY" -o BatchMode=yes -o ConnectTimeout=10 \
|
||||
-o StrictHostKeyChecking=no "${PVE_USER}@${PVE_HOST}" \
|
||||
"sudo /usr/sbin/qm list" &>/dev/null \
|
||||
|| die "Cannot SSH to ${PVE_USER}@${PVE_HOST} — check authorized_keys and sudo"
|
||||
log "Fence agent SSH connectivity confirmed"
|
||||
|
||||
log "Creating Pacemaker STONITH resources..."
|
||||
cibadmin --create --scope resources --xml-text "
|
||||
<primitive id=\"stonith-${NODE1}\" class=\"stonith\" type=\"external/fence_pve_ssh\">
|
||||
<instance_attributes id=\"stonith-${NODE1}-attrs\">
|
||||
<nvpair id=\"stonith-${NODE1}-plug\" name=\"plug\" value=\"${NODE1}\"/>
|
||||
<nvpair id=\"stonith-${NODE1}-pve-host\" name=\"pve_host\" value=\"${PVE_HOST}\"/>
|
||||
<nvpair id=\"stonith-${NODE1}-pve-user\" name=\"pve_user\" value=\"${PVE_USER}\"/>
|
||||
<nvpair id=\"stonith-${NODE1}-key-file\" name=\"key_file\" value=\"${FENCE_KEY}\"/>
|
||||
<nvpair id=\"stonith-${NODE1}-vmid1\" name=\"vmid_node1\" value=\"${VMID_NODE1}\"/>
|
||||
<nvpair id=\"stonith-${NODE1}-vmid2\" name=\"vmid_node2\" value=\"${VMID_NODE2}\"/>
|
||||
<nvpair id=\"stonith-${NODE1}-host-list\" name=\"pcmk_host_list\" value=\"${NODE1}\"/>
|
||||
</instance_attributes>
|
||||
<operations>
|
||||
<op id=\"stonith-${NODE1}-monitor\" name=\"monitor\" interval=\"30s\" timeout=\"30s\"/>
|
||||
</operations>
|
||||
</primitive>
|
||||
" 2>/dev/null || true
|
||||
|
||||
cibadmin --create --scope resources --xml-text "
|
||||
<primitive id=\"stonith-${NODE2}\" class=\"stonith\" type=\"external/fence_pve_ssh\">
|
||||
<instance_attributes id=\"stonith-${NODE2}-attrs\">
|
||||
<nvpair id=\"stonith-${NODE2}-plug\" name=\"plug\" value=\"${NODE2}\"/>
|
||||
<nvpair id=\"stonith-${NODE2}-pve-host\" name=\"pve_host\" value=\"${PVE_HOST}\"/>
|
||||
<nvpair id=\"stonith-${NODE2}-pve-user\" name=\"pve_user\" value=\"${PVE_USER}\"/>
|
||||
<nvpair id=\"stonith-${NODE2}-key-file\" name=\"key_file\" value=\"${FENCE_KEY}\"/>
|
||||
<nvpair id=\"stonith-${NODE2}-vmid1\" name=\"vmid_node1\" value=\"${VMID_NODE1}\"/>
|
||||
<nvpair id=\"stonith-${NODE2}-vmid2\" name=\"vmid_node2\" value=\"${VMID_NODE2}\"/>
|
||||
<nvpair id=\"stonith-${NODE2}-host-list\" name=\"pcmk_host_list\" value=\"${NODE2}\"/>
|
||||
</instance_attributes>
|
||||
<operations>
|
||||
<op id=\"stonith-${NODE2}-monitor\" name=\"monitor\" interval=\"30s\" timeout=\"30s\"/>
|
||||
</operations>
|
||||
</primitive>
|
||||
" 2>/dev/null || true
|
||||
|
||||
log "Enabling STONITH and restoring quorum policy..."
|
||||
crm_attribute -t crm_config -n stonith-enabled -v true
|
||||
crm_attribute -t crm_config -n no-quorum-policy -v stop
|
||||
|
||||
log "DRBD fencing mode must also be updated to resource-only (already the"
|
||||
log "default in cluster-config.nix; confirm with: cat /etc/drbd.d/ha-data.conf)"
|
||||
|
||||
log "Testing fence agent..."
|
||||
stonith_admin --list-devices && log "Fence devices listed successfully." \
|
||||
|| warn "stonith_admin --list-devices failed — check config"
|
||||
|
||||
log "STONITH enabled. Cluster is now fully HA."
|
||||
@@ -0,0 +1,284 @@
|
||||
#!/usr/bin/env bash
|
||||
# cluster-init.sh — one-time HA cluster initialisation script
|
||||
#
|
||||
# Run ONCE from ha-server-1 as root AFTER both VMs are booted and have SSH
|
||||
# access. It:
|
||||
# 1. Generates and distributes the corosync authkey
|
||||
# 2. Waits for corosync quorum and pacemaker
|
||||
# 3. Initialises DRBD metadata, promotes node1 to primary
|
||||
# 4. Creates XFS on /dev/drbd0 and mounts it
|
||||
# 5. Creates the directory tree and iSCSI LUN backing file
|
||||
# 6. Configures LIO iSCSI target (file-backed LUN)
|
||||
# 7. Configures Pacemaker resources: DRBD → XFS → iSCSI → NFS → VIP
|
||||
#
|
||||
# Prerequisites:
|
||||
# - Both VMs booted with the ha-server config (nixos-rebuild done)
|
||||
# - SSH key access from node1 to root@NODE2_IP
|
||||
# - VMID_NODE1 / VMID_NODE2 filled in below (needed for STONITH setup;
|
||||
# cluster starts without STONITH, which you enable separately via
|
||||
# scripts/ha/cluster-enable-stonith.sh)
|
||||
# - Run as root on ha-server-1
|
||||
set -euo pipefail
|
||||
|
||||
# ── Configuration ─────────────────────────────────────────────────────────
|
||||
# These must match variables.nix haServer* values and the Proxmox VMID
|
||||
# assignments. Update before running.
|
||||
NODE1="ha-server-1"
|
||||
NODE2="ha-server-2"
|
||||
NODE1_IP="192.168.2.200" # vars.haServer1Ip
|
||||
NODE2_IP="192.168.2.201" # vars.haServer2Ip
|
||||
VIP="192.168.2.202" # vars.haServerVip
|
||||
XFS_MOUNT="/srv/ha-data" # vars.haStorageRoot
|
||||
ISCSI_IQN="iqn.2026-01.home.sweet:ha-storage" # vars.haIscsiIqn
|
||||
ISCSI_LUN_FILE="${XFS_MOUNT}/iscsi-lun.img"
|
||||
ISCSI_LUN_SIZE="10G"
|
||||
DRBD_DEVICE="/dev/drbd0"
|
||||
VMID_NODE1="" # FILL IN: Proxmox VMID for ha-server-1
|
||||
VMID_NODE2="" # FILL IN: Proxmox VMID for ha-server-2
|
||||
PVE_HOST="pve1.sweet.home"
|
||||
PVE_USER="wayne"
|
||||
|
||||
# NFS dataset subdirectories to create under XFS_MOUNT.
|
||||
# Must mirror vars.nfsShares subpath values in variables.nix.
|
||||
NFS_SUBDIRS=(
|
||||
"docker/config"
|
||||
"docker/volumes"
|
||||
"docker/databases"
|
||||
"docker/nextcloud-data"
|
||||
"raspi/volumes"
|
||||
"proxmox/iso"
|
||||
"proxmox/lxc"
|
||||
"pxe-boot/images"
|
||||
)
|
||||
# ──────────────────────────────────────────────────────────────────────────
|
||||
|
||||
log() { echo "[cluster-init] $*"; }
|
||||
die() { echo "[cluster-init] ERROR: $*" >&2; exit 1; }
|
||||
warn() { echo "[cluster-init] WARNING: $*" >&2; }
|
||||
|
||||
[[ $(id -u) -eq 0 ]] || die "must run as root"
|
||||
[[ "$(hostname)" == "$NODE1" ]] || die "must run on $NODE1"
|
||||
|
||||
# ── 0. Corosync authkey ───────────────────────────────────────────────────
|
||||
AUTHKEY="/etc/corosync/authkey"
|
||||
mkdir -p /etc/corosync
|
||||
if [[ ! -f "$AUTHKEY" ]]; then
|
||||
log "Generating corosync authkey..."
|
||||
corosync-keygen -k "$AUTHKEY"
|
||||
chmod 0400 "$AUTHKEY"
|
||||
fi
|
||||
log "Distributing authkey to $NODE2..."
|
||||
ssh "root@${NODE2_IP}" "mkdir -p /etc/corosync"
|
||||
scp -q "$AUTHKEY" "root@${NODE2_IP}:${AUTHKEY}"
|
||||
ssh "root@${NODE2_IP}" "chmod 0400 '${AUTHKEY}'"
|
||||
|
||||
log "Restarting corosync on both nodes..."
|
||||
systemctl restart corosync
|
||||
ssh "root@${NODE2_IP}" "systemctl restart corosync"
|
||||
sleep 3
|
||||
|
||||
# ── 1. Corosync quorum ────────────────────────────────────────────────────
|
||||
log "Waiting for corosync quorum..."
|
||||
for i in $(seq 1 30); do
|
||||
if corosync-quorumtool -s 2>/dev/null | grep -q 'Quorate:.*Yes'; then
|
||||
log "Quorum established"
|
||||
break
|
||||
fi
|
||||
[[ $i -eq 30 ]] && die "corosync quorum not established after 60 s"
|
||||
sleep 2
|
||||
done
|
||||
|
||||
log "Waiting for pacemaker..."
|
||||
for i in $(seq 1 30); do
|
||||
if crm_mon -1 &>/dev/null; then
|
||||
log "Pacemaker running"
|
||||
break
|
||||
fi
|
||||
[[ $i -eq 30 ]] && die "pacemaker not running after 60 s"
|
||||
sleep 2
|
||||
done
|
||||
|
||||
# ── 2. DRBD initialisation ────────────────────────────────────────────────
|
||||
log "Initialising DRBD metadata on $NODE1..."
|
||||
if ! drbdadm dstate ha-data 2>/dev/null | grep -q "UpToDate\|Inconsistent\|Diskless"; then
|
||||
drbdadm create-md ha-data --force
|
||||
fi
|
||||
|
||||
log "Initialising DRBD metadata on $NODE2..."
|
||||
ssh "root@${NODE2_IP}" "
|
||||
if ! drbdadm dstate ha-data 2>/dev/null | grep -q 'UpToDate\|Inconsistent\|Diskless'; then
|
||||
drbdadm create-md ha-data --force
|
||||
fi
|
||||
"
|
||||
|
||||
log "Bringing up DRBD on both nodes..."
|
||||
drbdadm up ha-data 2>/dev/null || true
|
||||
ssh "root@${NODE2_IP}" "drbdadm up ha-data 2>/dev/null" || true
|
||||
|
||||
log "Forcing $NODE1 to DRBD Primary for initial sync..."
|
||||
drbdadm primary ha-data --force
|
||||
|
||||
log "Waiting for DRBD to finish initial sync (this may take several minutes)..."
|
||||
for i in $(seq 1 300); do
|
||||
state=$(drbdadm dstate ha-data 2>/dev/null || echo "unknown")
|
||||
if echo "$state" | grep -q "UpToDate/UpToDate"; then
|
||||
log "DRBD sync complete: $state"
|
||||
break
|
||||
fi
|
||||
[[ $i -eq 300 ]] && warn "DRBD not UpToDate after 300 s — continuing anyway (check drbdadm status)"
|
||||
sleep 1
|
||||
done
|
||||
|
||||
# ── 3. XFS filesystem ─────────────────────────────────────────────────────
|
||||
log "Creating XFS on ${DRBD_DEVICE}..."
|
||||
if ! xfs_info "${DRBD_DEVICE}" &>/dev/null; then
|
||||
mkfs.xfs -f "${DRBD_DEVICE}"
|
||||
fi
|
||||
|
||||
log "Mounting ${DRBD_DEVICE} at ${XFS_MOUNT}..."
|
||||
mkdir -p "${XFS_MOUNT}"
|
||||
mount "${DRBD_DEVICE}" "${XFS_MOUNT}"
|
||||
|
||||
# ── 4. NFS dataset directories ────────────────────────────────────────────
|
||||
log "Creating NFS dataset directories..."
|
||||
for subdir in "${NFS_SUBDIRS[@]}"; do
|
||||
mkdir -p "${XFS_MOUNT}/${subdir}"
|
||||
done
|
||||
|
||||
# ── 5. iSCSI LUN backing file ─────────────────────────────────────────────
|
||||
log "Creating iSCSI LUN backing file ${ISCSI_LUN_FILE} (${ISCSI_LUN_SIZE})..."
|
||||
if [[ ! -f "${ISCSI_LUN_FILE}" ]]; then
|
||||
fallocate -l "${ISCSI_LUN_SIZE}" "${ISCSI_LUN_FILE}"
|
||||
fi
|
||||
|
||||
# ── 6. LIO iSCSI target ───────────────────────────────────────────────────
|
||||
log "Configuring LIO iSCSI target via targetcli..."
|
||||
targetcli <<EOF
|
||||
/backstores/fileio create name=ha-lun0 file_or_dev=${ISCSI_LUN_FILE} size=0 write_back=false
|
||||
/iscsi create ${ISCSI_IQN}
|
||||
/iscsi/${ISCSI_IQN}/tpg1/luns create /backstores/fileio/ha-lun0
|
||||
/iscsi/${ISCSI_IQN}/tpg1/portals create ${VIP}
|
||||
/iscsi/${ISCSI_IQN}/tpg1 set attribute authentication=0
|
||||
/iscsi/${ISCSI_IQN}/tpg1 set attribute demo_mode_write_protect=0
|
||||
saveconfig /etc/target/saveconfig.json
|
||||
EOF
|
||||
|
||||
log "Distributing iSCSI saveconfig to $NODE2..."
|
||||
scp -q /etc/target/saveconfig.json "root@${NODE2_IP}:/etc/target/saveconfig.json"
|
||||
|
||||
log "Unmounting ${XFS_MOUNT} — Pacemaker manages it..."
|
||||
umount "${XFS_MOUNT}"
|
||||
|
||||
log "Demoting DRBD to Secondary — Pacemaker manages primary role..."
|
||||
drbdadm secondary ha-data
|
||||
|
||||
# ── 7. Pacemaker resources ────────────────────────────────────────────────
|
||||
log "Configuring Pacemaker cluster properties..."
|
||||
crm_attribute -t crm_config -n stonith-enabled -v false
|
||||
crm_attribute -t crm_config -n no-quorum-policy -v ignore
|
||||
|
||||
log "Creating DRBD promotable clone resource..."
|
||||
cibadmin --replace --scope resources --xml-text "
|
||||
<resources>
|
||||
<clone id=\"ms-drbd0\" globally-unique=\"false\">
|
||||
<meta_attributes id=\"ms-drbd0-meta\">
|
||||
<nvpair id=\"ms-drbd0-promotable\" name=\"promotable\" value=\"true\"/>
|
||||
<nvpair id=\"ms-drbd0-master-max\" name=\"master-max\" value=\"1\"/>
|
||||
<nvpair id=\"ms-drbd0-master-node-max\" name=\"master-node-max\" value=\"1\"/>
|
||||
<nvpair id=\"ms-drbd0-clone-max\" name=\"clone-max\" value=\"2\"/>
|
||||
<nvpair id=\"ms-drbd0-clone-node-max\" name=\"clone-node-max\" value=\"1\"/>
|
||||
<nvpair id=\"ms-drbd0-notify\" name=\"notify\" value=\"true\"/>
|
||||
<nvpair id=\"ms-drbd0-interleave\" name=\"interleave\" value=\"true\"/>
|
||||
</meta_attributes>
|
||||
<primitive id=\"drbd0\" class=\"ocf\" type=\"drbd\" provider=\"linbit\">
|
||||
<instance_attributes id=\"drbd0-attrs\">
|
||||
<nvpair id=\"drbd0-resource\" name=\"drbd_resource\" value=\"ha-data\"/>
|
||||
</instance_attributes>
|
||||
<operations>
|
||||
<op id=\"drbd0-start\" name=\"start\" interval=\"0\" timeout=\"240s\"/>
|
||||
<op id=\"drbd0-stop\" name=\"stop\" interval=\"0\" timeout=\"120s\"/>
|
||||
<op id=\"drbd0-promote\" name=\"promote\" interval=\"0\" timeout=\"90s\"/>
|
||||
<op id=\"drbd0-demote\" name=\"demote\" interval=\"0\" timeout=\"90s\"/>
|
||||
<op id=\"drbd0-monitor-master\" name=\"monitor\" interval=\"20s\" timeout=\"20s\" role=\"Promoted\"/>
|
||||
<op id=\"drbd0-monitor-slave\" name=\"monitor\" interval=\"30s\" timeout=\"20s\" role=\"Unpromoted\"/>
|
||||
</operations>
|
||||
</primitive>
|
||||
</clone>
|
||||
<group id=\"ha-group\">
|
||||
<primitive id=\"xfs-data\" class=\"ocf\" type=\"Filesystem\" provider=\"heartbeat\">
|
||||
<instance_attributes id=\"xfs-data-attrs\">
|
||||
<nvpair id=\"xfs-data-device\" name=\"device\" value=\"${DRBD_DEVICE}\"/>
|
||||
<nvpair id=\"xfs-data-directory\" name=\"directory\" value=\"${XFS_MOUNT}\"/>
|
||||
<nvpair id=\"xfs-data-fstype\" name=\"fstype\" value=\"xfs\"/>
|
||||
<nvpair id=\"xfs-data-options\" name=\"options\" value=\"defaults\"/>
|
||||
<nvpair id=\"xfs-data-force_unmount\" name=\"force_unmount\" value=\"false\"/>
|
||||
</instance_attributes>
|
||||
<operations>
|
||||
<op id=\"xfs-data-start\" name=\"start\" interval=\"0\" timeout=\"60s\"/>
|
||||
<op id=\"xfs-data-stop\" name=\"stop\" interval=\"0\" timeout=\"60s\"/>
|
||||
<op id=\"xfs-data-monitor\" name=\"monitor\" interval=\"20s\" timeout=\"40s\"/>
|
||||
</operations>
|
||||
</primitive>
|
||||
<primitive id=\"iscsi-target\" class=\"systemd\" type=\"targetctl\">
|
||||
<operations>
|
||||
<op id=\"iscsi-start\" name=\"start\" interval=\"0\" timeout=\"60s\"/>
|
||||
<op id=\"iscsi-stop\" name=\"stop\" interval=\"0\" timeout=\"60s\"/>
|
||||
<op id=\"iscsi-monitor\" name=\"monitor\" interval=\"20s\" timeout=\"40s\"/>
|
||||
</operations>
|
||||
</primitive>
|
||||
<primitive id=\"nfs-server\" class=\"systemd\" type=\"nfs-server\">
|
||||
<operations>
|
||||
<op id=\"nfs-start\" name=\"start\" interval=\"0\" timeout=\"60s\"/>
|
||||
<op id=\"nfs-stop\" name=\"stop\" interval=\"0\" timeout=\"60s\"/>
|
||||
<op id=\"nfs-monitor\" name=\"monitor\" interval=\"30s\" timeout=\"40s\"/>
|
||||
</operations>
|
||||
</primitive>
|
||||
<primitive id=\"vip\" class=\"ocf\" type=\"IPaddr2\" provider=\"heartbeat\">
|
||||
<instance_attributes id=\"vip-attrs\">
|
||||
<nvpair id=\"vip-ip\" name=\"ip\" value=\"${VIP}\"/>
|
||||
<nvpair id=\"vip-cidr\" name=\"cidr_netmask\" value=\"24\"/>
|
||||
</instance_attributes>
|
||||
<operations>
|
||||
<op id=\"vip-start\" name=\"start\" interval=\"0\" timeout=\"20s\"/>
|
||||
<op id=\"vip-stop\" name=\"stop\" interval=\"0\" timeout=\"20s\"/>
|
||||
<op id=\"vip-monitor\" name=\"monitor\" interval=\"10s\" timeout=\"20s\"/>
|
||||
</operations>
|
||||
</primitive>
|
||||
</group>
|
||||
</resources>
|
||||
"
|
||||
|
||||
log "Adding ordering and colocation constraints..."
|
||||
cibadmin --create --scope constraints --xml-text "
|
||||
<constraints>
|
||||
<rsc_order id=\"order-drbd-group\" first=\"ms-drbd0\" first-action=\"promote\" then=\"ha-group\" then-action=\"start\"/>
|
||||
<rsc_colocation id=\"coloc-group-with-drbd\" rsc=\"ha-group\" with-rsc=\"ms-drbd0\" with-rsc-role=\"Master\" score=\"INFINITY\"/>
|
||||
</constraints>
|
||||
"
|
||||
|
||||
log "Waiting for resources to start..."
|
||||
for i in $(seq 1 60); do
|
||||
if crm_resource -r vip --locate 2>/dev/null | grep -q "running on"; then
|
||||
log "VIP is up: $(crm_resource -r vip --locate)"
|
||||
break
|
||||
fi
|
||||
[[ $i -eq 60 ]] && { warn "VIP not up after 120 s — check: crm_mon -1"; break; }
|
||||
sleep 2
|
||||
done
|
||||
|
||||
log ""
|
||||
log "═══════════════════════════════════════════════════════════════"
|
||||
log " HA cluster initialised."
|
||||
log ""
|
||||
log " crm_mon -1 — cluster status"
|
||||
log " iscsiadm -m discovery -t st -p ${VIP} — verify iSCSI target"
|
||||
log " showmount -e ${VIP} — verify NFS exports"
|
||||
log ""
|
||||
log " To enable STONITH (after deploying fence SSH key):"
|
||||
log " 1. Fill in VMID_NODE1 / VMID_NODE2 in cluster-enable-stonith.sh"
|
||||
log " 2. Copy scripts/ha/fence-pve-ssh.py to /etc/pacemaker/fence_pve_ssh"
|
||||
log " on both nodes (chmod +x)"
|
||||
log " 3. Generate and distribute the fence SSH key"
|
||||
log " (see docs or cluster-enable-stonith.sh header)"
|
||||
log " 4. bash scripts/ha/cluster-enable-stonith.sh"
|
||||
log "═══════════════════════════════════════════════════════════════"
|
||||
@@ -0,0 +1,179 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
fence_pve_ssh - Proxmox VE SSH fence agent for Pacemaker.
|
||||
|
||||
Uses SSH to reach the Proxmox host and run 'qm stop/start <vmid>'.
|
||||
Deploy to /etc/pacemaker/fence_pve_ssh on both HA nodes (chmod +x).
|
||||
|
||||
Configuration (as pacemaker stonith resource attributes):
|
||||
pve_host Proxmox host to SSH to (default: pve1.sweet.home)
|
||||
pve_user SSH user (default: wayne)
|
||||
key_file SSH private key path (default: /etc/fence-pve-ssh-key)
|
||||
vmid_node1 VMID for ha-server-1
|
||||
vmid_node2 VMID for ha-server-2
|
||||
plug Node name to act on (set by pacemaker: ha-server-1 or ha-server-2)
|
||||
action Action: off|on|reboot|status|list|metadata
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import subprocess
|
||||
import sys
|
||||
import os
|
||||
|
||||
|
||||
METADATA = """<?xml version="1.0" ?>
|
||||
<resource-agent name="fence_pve_ssh" shortdesc="Proxmox VE SSH fence agent (test lab)">
|
||||
<longdesc>Fences a VM on a Proxmox VE host by SSHing to the PVE host and
|
||||
running qm stop/start. For test use only.</longdesc>
|
||||
<vendor-url>https://proxmox.com</vendor-url>
|
||||
<parameters>
|
||||
<parameter name="action" required="1" unique="0">
|
||||
<getopt mixed="-a, --action=[action]"/>
|
||||
<content type="string" default="reboot"/>
|
||||
<shortdesc lang="en">Fencing action: off|on|reboot|status|list</shortdesc>
|
||||
</parameter>
|
||||
<parameter name="plug" required="0" unique="0">
|
||||
<getopt mixed="-n, --plug=[nodename]"/>
|
||||
<content type="string"/>
|
||||
<shortdesc lang="en">Cluster node name to fence</shortdesc>
|
||||
</parameter>
|
||||
<parameter name="pve_host" required="0" unique="0">
|
||||
<getopt mixed="--pve-host=[host]"/>
|
||||
<content type="string" default="pve1.sweet.home"/>
|
||||
<shortdesc lang="en">Proxmox VE host to SSH to</shortdesc>
|
||||
</parameter>
|
||||
<parameter name="pve_user" required="0" unique="0">
|
||||
<getopt mixed="--pve-user=[user]"/>
|
||||
<content type="string" default="wayne"/>
|
||||
<shortdesc lang="en">SSH user on the Proxmox host</shortdesc>
|
||||
</parameter>
|
||||
<parameter name="key_file" required="0" unique="0">
|
||||
<getopt mixed="--key-file=[path]"/>
|
||||
<content type="string" default="/etc/fence-pve-ssh-key"/>
|
||||
<shortdesc lang="en">SSH private key file path</shortdesc>
|
||||
</parameter>
|
||||
<parameter name="vmid_node1" required="1" unique="0">
|
||||
<getopt mixed="--vmid-node1=[vmid]"/>
|
||||
<content type="string"/>
|
||||
<shortdesc lang="en">VMID for ha-test-node1</shortdesc>
|
||||
</parameter>
|
||||
<parameter name="vmid_node2" required="1" unique="0">
|
||||
<getopt mixed="--vmid-node2=[vmid]"/>
|
||||
<content type="string"/>
|
||||
<shortdesc lang="en">VMID for ha-test-node2</shortdesc>
|
||||
</parameter>
|
||||
</parameters>
|
||||
<actions>
|
||||
<action name="off" timeout="60s"/>
|
||||
<action name="on" timeout="60s"/>
|
||||
<action name="reboot" timeout="60s"/>
|
||||
<action name="status" timeout="30s"/>
|
||||
<action name="list" timeout="10s"/>
|
||||
<action name="metadata" timeout="5s"/>
|
||||
</actions>
|
||||
</resource-agent>
|
||||
"""
|
||||
|
||||
|
||||
def parse_args():
|
||||
p = argparse.ArgumentParser(add_help=False)
|
||||
p.add_argument("-a", "--action", default="reboot")
|
||||
p.add_argument("-n", "--plug")
|
||||
p.add_argument("--pve-host", default="pve1.sweet.home")
|
||||
p.add_argument("--pve-user", default="wayne")
|
||||
p.add_argument("--key-file", default="/etc/fence-pve-ssh-key")
|
||||
p.add_argument("--vmid-node1")
|
||||
p.add_argument("--vmid-node2")
|
||||
# Allow remaining unknown args (pacemaker may pass extra ones)
|
||||
return p.parse_known_args()[0]
|
||||
|
||||
|
||||
def ssh(pve_host, pve_user, key_file, cmd):
|
||||
result = subprocess.run(
|
||||
[
|
||||
"ssh",
|
||||
"-i", key_file,
|
||||
"-o", "StrictHostKeyChecking=no",
|
||||
"-o", "BatchMode=yes",
|
||||
"-o", "ConnectTimeout=10",
|
||||
f"{pve_user}@{pve_host}",
|
||||
cmd,
|
||||
],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=30,
|
||||
)
|
||||
return result
|
||||
|
||||
|
||||
def get_vmid(args):
|
||||
node = args.plug
|
||||
if not node:
|
||||
print("ERROR: --plug not specified", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
mapping = {
|
||||
"ha-server-1": args.vmid_node1,
|
||||
"ha-server-2": args.vmid_node2,
|
||||
}
|
||||
vmid = mapping.get(node)
|
||||
if not vmid:
|
||||
print(f"ERROR: unknown node '{node}'", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
return vmid
|
||||
|
||||
|
||||
def main():
|
||||
args = parse_args()
|
||||
action = args.action.lower()
|
||||
|
||||
if action == "metadata":
|
||||
print(METADATA)
|
||||
sys.exit(0)
|
||||
|
||||
if action == "list":
|
||||
if args.vmid_node1:
|
||||
print("ha-server-1")
|
||||
if args.vmid_node2:
|
||||
print("ha-server-2")
|
||||
sys.exit(0)
|
||||
|
||||
vmid = get_vmid(args)
|
||||
|
||||
if not os.path.exists(args.key_file):
|
||||
print(f"ERROR: SSH key not found at {args.key_file}", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
if action in ("off", "reboot"):
|
||||
print(f"Stopping VM {vmid} ({args.plug}) on {args.pve_host}...")
|
||||
r = ssh(args.pve_host, args.pve_user, args.key_file,
|
||||
f"sudo /usr/sbin/qm stop {vmid}")
|
||||
if r.returncode != 0:
|
||||
print(f"ERROR stopping VM: {r.stderr}", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
print(f"VM {vmid} stopped")
|
||||
|
||||
if action in ("on", "reboot"):
|
||||
print(f"Starting VM {vmid} ({args.plug}) on {args.pve_host}...")
|
||||
r = ssh(args.pve_host, args.pve_user, args.key_file,
|
||||
f"sudo /usr/sbin/qm start {vmid}")
|
||||
if r.returncode != 0:
|
||||
print(f"ERROR starting VM: {r.stderr}", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
print(f"VM {vmid} started")
|
||||
|
||||
if action == "status":
|
||||
r = ssh(args.pve_host, args.pve_user, args.key_file,
|
||||
f"sudo /usr/sbin/qm status {vmid}")
|
||||
if r.returncode != 0:
|
||||
print(f"ERROR querying VM status: {r.stderr}", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
# qm status returns "status: running" or "status: stopped"
|
||||
status_line = r.stdout.strip()
|
||||
print(status_line)
|
||||
if "stopped" in status_line:
|
||||
sys.exit(2) # pacemaker interprets exit 2 as "off"
|
||||
sys.exit(0) # running = exit 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1 @@
|
||||
STUB: run cluster-init.sh to generate, then: sops -e --input-type binary /etc/corosync/authkey > secrets/ha-corosync-authkey
|
||||
@@ -0,0 +1,6 @@
|
||||
# STUB — not yet encrypted with sops.
|
||||
# Bootstrap:
|
||||
# bash scripts/secrets/sync-host-keys.sh proxmox-ha-server-1
|
||||
# sops updatekeys secrets/common.yaml (allows ha-server-1 to decrypt shared secrets)
|
||||
# sops secrets/ha-server-1.yaml (create with: beszel-token)
|
||||
beszel-token: REPLACE
|
||||
@@ -0,0 +1,6 @@
|
||||
# STUB — not yet encrypted with sops.
|
||||
# Bootstrap:
|
||||
# bash scripts/secrets/sync-host-keys.sh proxmox-ha-server-2
|
||||
# sops updatekeys secrets/common.yaml (allows ha-server-2 to decrypt shared secrets)
|
||||
# sops secrets/ha-server-2.yaml (create with: beszel-token)
|
||||
beszel-token: REPLACE
|
||||
+26
-1
@@ -68,6 +68,20 @@
|
||||
# one-line change.
|
||||
primaryUser = "nixos";
|
||||
|
||||
# HA file server cluster
|
||||
# haServer1Ip / haServer2Ip: static LAN IPs for both HA nodes (must be
|
||||
# fixed — DRBD and corosync ring addresses are baked into the NixOS config).
|
||||
# haServerVip: floating virtual IP managed by Pacemaker's IPaddr2 resource;
|
||||
# NFS and iSCSI clients connect here regardless of which node is Active.
|
||||
# Set all three to real values in variables.nix before deploying.
|
||||
haServer1Host = "ha-server-1";
|
||||
haServer2Host = "ha-server-2";
|
||||
haServer1Ip = "192.168.2.200"; # TODO: confirm production IP
|
||||
haServer2Ip = "192.168.2.201"; # TODO: confirm production IP
|
||||
haServerVip = "192.168.2.202"; # TODO: confirm floating VIP
|
||||
haStorageRoot = "/srv/ha-data"; # XFS-over-DRBD mount point on the Active node
|
||||
haIscsiIqn = "iqn.2026-01.home.sweet:ha-storage";
|
||||
|
||||
# Storage
|
||||
storageRoot = "/tank"; # ZFS pool root on `server`
|
||||
|
||||
@@ -146,11 +160,22 @@
|
||||
# mountd RPC service (used by showmount/NFSv3 mount protocol).
|
||||
# Mountd listens on a fixed port so the firewall can whitelist it
|
||||
# explicitly rather than opening all of rpcbind's dynamic range.
|
||||
# All three need both TCP and UDP (modules/build-types/server.nix).
|
||||
# All three need both TCP and UDP (modules/build-types/server.nix and
|
||||
# modules/build-types/ha-server.nix).
|
||||
nfsRpcbind = 111;
|
||||
nfsd = 2049;
|
||||
nfsMountd = 20048;
|
||||
|
||||
# HA cluster ports opened on ha-server-1 and ha-server-2
|
||||
# (modules/build-types/ha-server.nix / modules/ha/cluster-config.nix).
|
||||
haServerDrbd = 7789; # DRBD replication (TCP)
|
||||
haServerIscsi = 3260; # iSCSI target (TCP)
|
||||
haServerCorosync1 = 5404; # Corosync totem ring (UDP)
|
||||
haServerCorosync2 = 5405; # Corosync totem ring (UDP)
|
||||
haServerCorosyncCrypto = 5407; # Corosync crypto sync (UDP)
|
||||
haServerPacemakerRemoted = 3121; # pacemaker-remoted (TCP)
|
||||
haServerPcsd = 2224; # pcsd cluster daemon (TCP)
|
||||
|
||||
# Opened on the docker host's firewall for the Traefik-fronted
|
||||
# container stack (docker-compose config lives in the separate
|
||||
# /home/debian/docker repo, not here): 80/443 are Traefik's own
|
||||
|
||||
Reference in New Issue
Block a user