This repository has been archived on 2026-07-30. You can view files and clone it. You cannot open issues or pull requests or push a commit.
Files
nixos/variables.nix
beatzaplentyandClaude Sonnet 4.6 c96752e5d0 Add Docker Swarm HA cluster: ha-docker-1 and ha-docker-2
Two new NixOS Proxmox VMs (VMIDs 202/203) forming a dual-manager Docker
Swarm on dedicated vmbr3 (192.168.30.0/24, VLAN 30) for gossip and VXLAN,
with NFS via the storage-client network (vmbr2) from the existing HA cluster.

- nixos/variables.nix: add ha-docker IP/interface/port vars and swarm CIDR
- nixos/modules/build-types/ha-docker.nix: new build type — Docker 29,
  NFS mounts, beszel-agent, health monitoring, swarm firewall rules with
  checkReversePath = "loose" for VXLAN routing mesh
- nixos/hosts/ha-docker-{1,2}/host.nix: per-host identity — three NICs
  (LAN, storage, swarm), IPA dyndns pinned to LAN interface
- nixos/flake.nix: add proxmox-ha-docker-{1,2} targets; build-validated
  with nix build --dry-run (169 derivations, no errors)
- nixos/docs/ip-addressing.md: document VLAN 30 / swarm.home zone,
  ha-docker IP allocations across all three subnets
- nixos/scripts/docker-swarm/deploy.sh: 10-phase lifecycle script
  (bridge, keys, IPA, VMs, swarm init, DNS, verify); modelled on
  scripts/ha/deploy.sh with --destroy mode
- nixos/docs/internal/docker-swarm-cutover.md: service-by-service
  migration guide covering Traefik log rotation, Nextcloud cron sidecar,
  docker-health-to-gotify swarm awareness updates, Passbolt/Gitea steps,
  DNS cutover, and CT 105 decommission checklist

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01DASH15okNvWeY1rVJmyJoJ
2026-07-30 19:01:17 +10:00

344 lines
18 KiB
Nix
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
rec {
# ── Gitea / flake remote ──────────────────────────────────────────────────
# External Gitea/DDNS domain — used only for the remote flake URL in
# Switch-nix / Test-nix aliases (modules/common/configuration.nix).
giteaDomain = "gitea.lan.ddnsgeek.com";
# Org/repo path within Gitea, combined with giteaDomain to form the
# git+https:// URL used by Switch-nix / Test-nix.
giteaRepoPath = "beatzaplenty/infrastructure";
# internal repo path to flake
giteaRepoFlakePath = "nixos";
# ── Network ───────────────────────────────────────────────────────────────
# Base LAN domain for service subdomains (pve., docker., nix-cache., …)
homeDomain = "sweet.home";
# Tailscale MagicDNS suffix for this tailnet
tailnetDomain = "tail13f623.ts.net";
lanCidr = "192.168.2.0/24";
lanGateway = "192.168.2.254";
lanPrefixLength = 24;
# NIC names inside guests — determined by the hypervisor/platform, not the OS.
lxcLanInterface = "eth0"; # LAN NIC in LXC containers (Proxmox --net0 name=eth0)
lxcStorageInterface = "eth1"; # storage-client NIC in LXC containers (vmbr2, --net1)
vmLanInterface = "ens18"; # LAN NIC in Proxmox VMs (virtio, first NIC)
vmStorageInterface = "ens19"; # cluster-internal NIC in HA VMs (vmbr1 — DRBD + Corosync)
vmStorageClientInterface = "ens20"; # storage-client NIC in HA VMs (vmbr2 — iSCSI/NFS VIP)
# ── Host IPs ──────────────────────────────────────────────────────────────
pxeServerIp = "192.168.2.223"; # pxe-boot LXC container LAN IP
nixCacheIp = "192.168.2.224"; # nix-cache LXC container LAN IP
tailscaleRouterIp = "192.168.2.222"; # tailscale-router LXC container LAN IP
torRelayIp = "192.168.2.221"; # tor-relay LXC container LAN IP
dockerIp = "192.168.2.225"; # docker Proxmox VM LAN IP
pbsIp = "192.168.2.244"; # Proxmox Backup Server LAN IP (not NixOS-managed)
domainControllerIp = "192.168.2.253"; # FreeIPA — authoritative DNS for sweet.home (not NixOS-managed)
# FreeIPA server FQDN used by security.ipa and Kerberos. Must be a
# resolvable name (not an IP); resolves to domainControllerIp.
ipaServer = "domain-controller.${homeDomain}";
# ── Cross-host references ─────────────────────────────────────────────────
nixCacheHost = "nix-cache"; # substituter/remote-builder hostname
dockerHost = "docker"; # docker-compose stack host
# Raspberry Pi's own Tailscale hostname (not fronted by any server — it
# exports its own NFS share directly). Resolved as
# "${raspberryPiHost}.${tailnetDomain}" in modules/raspi/mount-data.nix.
raspberryPiHost = "raspberrypi";
remoteBuilderUser = "nixremote";
# Tailscale's internal "Quad100" DNS resolver, reachable from any Tailscale
# node via tailscale0. Used by modules/tailscale/ts-dns-forwarder.nix to
# forward *.tailnetDomain queries on behalf of FreeIPA's conditional
# forwarder zone.
tailscaleResolverIp = "100.100.100.100";
# ── SSH keys ──────────────────────────────────────────────────────────────
# nix-cache's SSH host public key (not a secret — private half never leaves
# the host). Wired into every client's programs.ssh.knownHosts by
# modules/nix-cache/remote-builder-client.nix so distributed builds don't
# hit "Host key verification failed" on a fresh client. Update if nix-cache
# is ever rebuilt with a new host key.
nixCacheHostKey = "ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAICuHUxGNH6ei3BZD+EfZs3l4X8uJNcjQiOsM/G4yo4O/ lxc-nix-cache";
# Beszel hub's SSH public key — used by every agent to authenticate the
# hub's incoming connection. Update if the docker host is ever rebuilt and
# the hub generates a new keypair.
beszelHubKey = "ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIFPR9kwtC4TAeTRu46A7+opZsYpxqkRJ+x/ZyB2GWCeG";
# Public keys authorized to SSH in as remoteBuilderUser on the nix-cache
# host (modules/nix-cache/server.nix) — one per client host allowed to use
# it as a distributed builder.
remoteBuilderAuthorizedKeys = [
"ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIK+ioWPhHixlgCB9KIQ0QTHTz6A+Oo2F3uKiINLip5rO root@docker"
"ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAICMJhrfFayLBG+gWtO6oAvgambw5nWWgztiTFEaaaVRH debian@surface"
"ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAII/rLceRhnDobVXQYiPceuhDHHvVjFQ1pc9A6un/eUlA root@server"
"ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIHj11bLPpRzH2oslnwFzEvY9cSgfEFtSZbLQaDm4nZMK root@pxe-boot"
"ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIH4/Sesm8NYpj73R0cbGhI0Ubvz73vIVWAnbEDTlBTdh root@tor-relay"
"ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIIBCFUUtePndW7pqtlawft1QCdHmBVs3O/c8EJO+RcXV root@tailscale-router"
"ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIBZ9WKKAlP9Z7GQdgaZ1Xgw9C+vja2lqEZO5rJFpVqYN root@ha-server-1"
"ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIEGNKlaaMckd8nLWNGz4B2QokXjnnIvM+rEUv+R6h0sp root@ha-server-2"
"ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIK+XeMco7OxUpjrjZm54HogMs9QB5xlcKmElASRvrmlW root@nixos"
"ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIMseQwPpmaa6cgV5U8KhUsiVSYARG85zGa9rho0LJWks wayne@pve1"
];
# Primary admin SSH public key, authorized on the primary user of every
# host and the installer image's nixos/root users.
adminSshKey = "ssh-rsa AAAAB3NzaC1yc2EAAAADAQABAAABgQCq/Q5LvIXlZwO2kdeAN5nLGZ59nZB7JHYMEszHxmNtGMzv1lM31jiPNsr0z2EKVZhE7OOfa2IF9rhWYD7JUA9G0yzdZ4WTXFNGVVOJoOVH6vAF3XCxoVilOEwTc7h2Wiy+rzd0B28/3spffzQQWJhY6GRQVa8j+6xAGF60Fcvl1vLosYT9Bn2ZbK4TCWOwAn2jqXIieGpZdn/UNZbGOeKRiCvhktDfMAzuQzN/9jMu/oF4pkPn2X1UrsQdNlvp0Ci8md612MozIpncQJyAF1ADhunr3sMx0isUXiqD29R5DS4TftpekqLNLak+zcxFa8N7DcRNp3DcKfJvyTkwQrR4r+b7lFLYOLHLagSso9CzeW/paAS2q9I5SBm/2DtE1diLLg2jZikYcstsu/G5RgvbzbKqjiaMwTdXC3AMvDxQrs7U5pDRZFzoofG3cpODbTm+uy3m0kP70z0M1K45UbDG0p+itnTu9x40JbQEgefbx38AItNvAIx1A8HO4I1VX28= wayne@stream";
# Additional SSH keys granted access alongside adminSshKey on every host
# (modules/common/configuration.nix) and on HA cluster root
# (modules/ha/cluster-config.nix). Single definition here prevents the
# two modules from drifting out of sync.
extraAdminSshKeys = [
"ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAICMJhrfFayLBG+gWtO6oAvgambw5nWWgztiTFEaaaVRH debian@surface"
"ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIGygkCljN6uKpdJbHTOQtn8ZnH+wKXDLAwrDFbLrE/65 nixos@nixos"
];
# ── Wifi ──────────────────────────────────────────────────────────────────
# Prestaged wifi SSID for the gui host's NetworkManager profile
# (modules/networking/wifi.nix). Password is sops-encrypted in
# secrets/gui.yaml (wifi-password) — not stored here.
wifiSsid = "nbn-fttp-net-5G";
# ── Bare-metal GUI host ───────────────────────────────────────────────────
# Two disks for the ZFS RAID0 (striped) root pool on baremetal-gui
# (modules/disko/baremetal.nix). Only referenced at disko-format time;
# afterward the pool imports by-partlabel/by-id paths regardless.
guiRootDisk1 = "/dev/sda";
guiRootDisk2 = "/dev/sdb";
# ── System / users ────────────────────────────────────────────────────────
timeZone = "Australia/Brisbane";
# Main interactive user on every host. Modules that grant this user a
# group, home directory, or tmpfiles ownership reference this so a rename
# is a one-line change here.
primaryUser = "nixos";
# Primary IPA/domain user. Home Manager is configured for this user on
# every IPA-enrolled host (modules/ipa/client.nix).
ipaUser = "wayne";
# GID of the IPA "docker-access" group (GID 50010 on the IPA server). The
# local "docker" group is pinned to this GID on every Docker host so IPA
# group membership alone grants socket access — no per-host
# users.groups.docker.members entries needed.
dockerAccessGid = 50010;
# ── HA file-server cluster ────────────────────────────────────────────────
#
# Three network segments, all internal to pve1:
# LAN VLAN 2 / vmbr0 / 192.168.2.x — management only
# Cluster VLAN 10 / vmbr1 / 192.168.10.x — DRBD replication + Corosync ring0
# Storage-client VLAN 20 / vmbr2 / 192.168.20.x — iSCSI + NFS client access
#
# The host octet is consistent across subnets: node1 = .228, node2 = .227,
# VIP = .229 everywhere.
#
# Protocol separation (firewall-enforced on HA nodes):
# NFS — both subnets; LAN VIP for pxe-boot/LAN clients, storage VIP for docker
# iSCSI — storage-client subnet only
haServer1Host = "ha-server-1";
haServer2Host = "ha-server-2";
haServer1Ip = "192.168.2.228"; # LAN IP, node 1 (vmbr0 / ens18)
haServer2Ip = "192.168.2.227"; # LAN IP, node 2 (vmbr0 / ens18)
haServer1StorageIp = "192.168.10.228"; # cluster-net IP, node 1 (vmbr1 / ens19, VLAN 10)
haServer2StorageIp = "192.168.10.227"; # cluster-net IP, node 2 (vmbr1 / ens19, VLAN 10)
haStorageCidr = "192.168.10.224/29"; # cluster subnet — VLAN 10, internal to pve1
haStoragePrefixLength = 29;
haServer1ClientIp = "192.168.20.228"; # storage-client IP, node 1 (vmbr2 / ens20, VLAN 20)
haServer2ClientIp = "192.168.20.227"; # storage-client IP, node 2 (vmbr2 / ens20, VLAN 20)
haServerVip = "192.168.20.229"; # storage-client floating VIP (Pacemaker vip-storage, VLAN 20)
haServerLanVip = "192.168.2.229"; # LAN floating VIP (Pacemaker vip-lan) — NFS for LAN clients
dockerStorageIp = "192.168.20.225"; # docker CT storage-client IP (vmbr2 / eth1, VLAN 20)
haClientCidr = "192.168.20.0/24"; # storage-client subnet — VLAN 20, internal to pve1
haClientPrefixLength = 24;
haStorageRoot = "/srv/ha-data"; # XFS-over-DRBD mount point on the Active node
# NFS VIP FQDNs — use these in fileSystems device strings so mounts
# survive a future VIP renumber via a DNS-only update, not a NixOS rebuild.
haStorageNfsFqdn = "nfs.storage.home"; # storage-client VIP (VLAN 20) — docker + future swarm
haLanNfsFqdn = "ha-vip-lan.${homeDomain}"; # LAN VIP (VLAN 2) — pxe-boot + other LAN clients
haIscsiIqn = "iqn.2026-01.home.sweet:ha-storage";
# DRBD backing disk — identified by SCSI controller path so it resolves to
# the correct block device regardless of OS-level naming (sda vs sdb can
# differ between VMs depending on disk-add order). drive-scsi1 is always
# the data disk; drive-scsi0 is the OS disk.
haServerDrbdDisk = "/dev/disk/by-id/scsi-0QEMU_QEMU_HARDDISK_drive-scsi1";
# ── Docker Swarm cluster ──────────────────────────────────────────────────
#
# Three network segments, all internal to pve1:
# LAN vmbr0 192.168.2.0/24 — management; SSH + external service traffic
# Storage-client vmbr2 192.168.20.0/24 — NFS from HA cluster VIP (shared with HA nodes)
# Swarm cluster vmbr3 192.168.30.0/24 — Docker Swarm gossip + VXLAN overlay
#
# Host octet consistent across subnets: node1 = .230, node2 = .231.
# IPs from the .230.239 expansion buffer documented in docs/ip-addressing.md.
#
# Docker Swarm uses --advertise-addr and --data-path-addr on the swarm NIC
# (ens20/vmbr3) so all inter-node cluster traffic stays on the isolated
# internal bridge and never crosses the LAN.
#
# When expanding to a second Proxmox node, vmbr3 (VLAN 30) and vmbr1
# (VLAN 10) share the same inter-node trunk NIC via VLAN tagging — same
# physical wire, different VLAN IDs.
haDocker1Host = "ha-docker-1";
haDocker2Host = "ha-docker-2";
haDocker1Ip = "192.168.2.230"; # LAN management NIC (ens18, vmbr0)
haDocker2Ip = "192.168.2.231";
haDocker1StorageIp = "192.168.20.230"; # storage-client NIC (ens19, vmbr2)
haDocker2StorageIp = "192.168.20.231";
haDocker1SwarmIp = "192.168.30.230"; # swarm cluster NIC (ens20, vmbr3)
haDocker2SwarmIp = "192.168.30.231";
haDockerSwarmCidr = "192.168.30.0/24";
haDockerSwarmPrefixLength = 24;
# NIC names for ha-docker VMs. ens19/ens20 occupy the same guest bus
# positions as vmStorageInterface/vmStorageClientInterface on ha-server VMs
# but are attached to different bridges — storage (vmbr2) and swarm (vmbr3)
# respectively. Kept as named variables to avoid bare literals in modules.
haDockerStorageInterface = "ens19"; # vmbr2 — NFS client
haDockerSwarmInterface = "ens20"; # vmbr3 — Docker Swarm gossip + VXLAN
# ── Storage / NFS ─────────────────────────────────────────────────────────
# NFS share definitions — used by ha-server.nix (exports), docker/mount-data.nix,
# and pxe-boot/mount-pxe-images.nix (mounts). `subpath` is relative to
# haStorageRoot; `mountpoint` is the absolute local path on each client.
# Renaming a share only requires changing it here — exports and all client
# mounts follow automatically.
nfsShares = {
options = "(rw,sync,no_subtree_check,no_root_squash)";
dockerConfig = { subpath = "docker/config"; mountpoint = "/mnt/docker/config"; };
dockerDatabases = { subpath = "docker/databases"; mountpoint = "/mnt/docker/databases"; };
dockerVolumes = { subpath = "docker/volumes"; mountpoint = "/mnt/docker/volumes"; };
nextcloudData = { subpath = "docker/nextcloud-data"; mountpoint = "/mnt/nextcloud-data"; };
raspiVolumes = { subpath = "raspi/volumes"; mountpoint = "/mnt/raspi-backup"; };
proxmoxIsos = { subpath = "proxmox/iso"; mountpoint = "/mnt/iso"; };
proxmoxLxcImages = { subpath = "proxmox/lxc"; mountpoint = "/mnt/lxc"; };
pxebootImages = { subpath = "pxe-boot/images"; mountpoint = "/mnt/pxe-images"; };
};
# The Raspberry Pi's own NFS export — not under haStorageRoot, served
# directly by the Pi over Tailscale (see raspberryPiHost) and mounted by
# modules/raspi/mount-data.nix.
raspiNfsPath = "/home/raspi/raspi";
raspiMountpoint = "/mnt/raspi";
# ── Ports ─────────────────────────────────────────────────────────────────
#
# Every literal port referenced from modules/ or hosts/, grouped by the
# service that opens or connects to it. Kept as separate entries even where
# two share a number today (e.g. nixCacheHttp and pxeBootHttp are both 80)
# so changing one service's port never silently changes another.
ports = {
# nix-cache's nginx reverse proxy in front of nix-serve
# (modules/nix-cache/server.nix)
nixCacheHttp = 80;
# pxe-boot's nginx asset server; also used to build pxeBaseUrl
# (modules/build-types/pxe-boot.nix)
pxeBootHttp = 80;
# pxe-boot's atftpd TFTP server — UDP (modules/build-types/pxe-boot.nix)
pxeBootTftp = 69;
# DHCP proxy port opened by dnsmasq on the pxe-boot host
# (modules/build-types/pxe-boot.nix)
dhcp = 67;
# DNS port opened on tailscale-router for FreeIPA's conditional forwarder
# (modules/tailscale/ts-dns-forwarder.nix)
dns = 53;
# NFS stack: portmapper (rpcbind), NFS data, and mountd RPC service.
# Mountd is pinned to a fixed port so the firewall can whitelist it
# without opening rpcbind's full dynamic range. All three need TCP + UDP
# (modules/build-types/ha-server.nix).
nfsRpcbind = 111;
nfsd = 2049;
nfsMountd = 20048;
# HA cluster ports (modules/ha/cluster-config.nix)
haServerDrbd = 7789; # DRBD replication (TCP)
haServerIscsi = 3260; # iSCSI target (TCP)
haServerCorosync1 = 5404; # Corosync totem ring (UDP)
haServerCorosync2 = 5405; # Corosync totem ring (UDP)
haServerCorosyncCrypto = 5407; # Corosync crypto sync (UDP)
haServerPacemakerRemoted = 3121; # pacemaker-remoted (TCP)
haServerPcsd = 2224; # pcsd cluster daemon (TCP)
# Docker host — Traefik HTTP/HTTPS listeners plus one additional exposed
# service (modules/build-types/docker.nix)
dockerHttp = 80;
dockerHttps = 443;
dockerExtra = 8080;
# Docker Swarm inter-node ports (modules/build-types/ha-docker.nix).
# Firewalled to haDockerSwarmCidr only — vmbr3 is an isolated bridge
# with no physical uplink, so these ports are unreachable from LAN.
dockerSwarmMgmt = 2377; # TCP — Raft consensus + cluster management
dockerSwarmDisc = 7946; # TCP+UDP — Serf gossip (container network discovery)
dockerSwarmVxlan = 4789; # UDP — VXLAN overlay data path
# Beszel monitoring hub on docker.sweet.home, reached by every agent
# (modules/beszel/enable-agent.nix, hosts/nixos/home.nix)
beszelHub = 8090;
# Proxmox VE and PBS web UIs — desktop shortcuts on the gui build type
# (hosts/nixos/home.nix, modules/build-types/gui.nix)
pveWeb = 8006;
pbsWeb = 8007;
# Tor relay's ORPort (modules/tor/enable-relay.nix). Opened via
# services.tor.openFirewall rather than allowedTCPPorts directly, but
# kept here so it's not a bare literal if ever referenced elsewhere.
torRelayOrPort = 9001;
};
# ── Build / image settings ────────────────────────────────────────────────
# .raw disk image size for every proxmox-* host's standalone Disko image
# build (modules/disko/proxmox.nix — see docs/proxmox-images.md).
proxmoxImageSize = "50G";
# nix-cache Nix store GC retention (modules/nix-cache/server.nix)
nixCacheGcMaxAge = "30d";
# Traefik access log rotation, watched on the docker host at
# nfsShares.dockerVolumes.mountpoint (modules/traefik/rotate-logs.nix)
traefikLogRotate = {
maxSize = "100M"; # rotate once a log file exceeds this size
keep = 20; # number of rotated logs to retain
};
}