Compare commits

...
Author SHA1 Message Date
beatzaplentyandClaude Sonnet 4.6 539bdf9833 fix(ha): resolve data disk device via by-id symlink even in dry-run
Check NixOS configurations / eval-hosts (pull_request) Successful in 10m21s
The by-id lookup is read-only so it's safe to run in dry-run mode.
Previously it was gated behind `if ! $DRY_RUN`, which always triggered
the sdb fallback warning in dry-run — making it look like the device
path wasn't reliable when the symlink actually exists on both servers.

Now the lookup always runs and the script errors out with a clear message
if the by-id symlink is genuinely missing, instead of silently falling
back to a guessed /dev/sd* name.

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-07-29 13:58:53 +10:00
beatzaplentyandClaude Sonnet 4.6 ced1657407 fix(ha): prefix qm commands with sudo for non-root Proxmox SSH user
Check NixOS configurations / eval-hosts (pull_request) Successful in 10m25s
qm lives at /usr/sbin/qm, which is not in the default PATH for
non-interactive SSH sessions as a non-root user.  Add PVE_SUDO (set to
"sudo" when PVE_SSH_USER != root, matching create-proxmox-resource.sh's
own sudo_prefix pattern) and prepend it to all three qm invocations in
the script.

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-07-29 13:53:41 +10:00
beatzaplenty fdf41c659c Merge pull request 'Worktree crispy churning kernighan' (#102) from worktree-crispy-churning-kernighan into main
Check NixOS configurations / eval-hosts (push) Successful in 10m50s
Reviewed-on: #102
2026-07-29 03:46:05 +00:00
beatzaplentyandClaude Sonnet 4.6 0ef8259225 fix(gc-hosts): source nix-daemon profile on pve1 before running gc
Check NixOS configurations / eval-hosts (pull_request) Successful in 10m34s
BatchMode SSH sessions don't source /etc/profile on non-NixOS hosts, so
nix-collect-garbage isn't on PATH for the wayne user. Source the nix-daemon
profile script explicitly, matching the pattern in scripts/lib/nix-bootstrap.sh.

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-07-29 13:45:48 +10:00
beatzaplentyandClaude Sonnet 4.6 629c1457a9 refactor(gc-hosts): discover running pve1 guests dynamically each run
Replace the static host list with dynamic discovery: workstation (nixos)
and pve1 are hard-wired first and second; remaining hosts are discovered
on every run by SSHing to pve1, listing running VMs/containers via
pct/qm list, and resolving their NixOS hostnames from a single flake eval.

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-07-29 13:44:26 +10:00
beatzaplenty 7ba603c005 Merge pull request 'feat(scripts): add gc-hosts.sh for parallel nix gc across all live hosts' (#101) from worktree-crispy-churning-kernighan into main
Check NixOS configurations / eval-hosts (push) Successful in 10m32s
Reviewed-on: #101
2026-07-29 03:36:09 +00:00
beatzaplentyandClaude Sonnet 4.6 decf3ddff9 feat(scripts): add gc-hosts.sh for parallel nix gc across all live hosts
Check NixOS configurations / eval-hosts (pull_request) Successful in 10m29s
Runs nix-collect-garbage -d on all deployed NixOS hosts and pve1 in
parallel, skipping nix-cache to avoid evicting shared cache paths.

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-07-29 13:35:24 +10:00
beatzaplenty 4490dfab7d Merge pull request 'feat(nix-cache): authorize pve1's wayne key as remote builder client' (#100) from worktree-pve1-nix-cache-builder into main
Check NixOS configurations / eval-hosts (push) Successful in 10m32s
Reviewed-on: #100
2026-07-29 03:16:53 +00:00
2 changed files with 237 additions and 12 deletions
+227
View File
@@ -0,0 +1,227 @@
#!/usr/bin/env bash
# gc-hosts.sh — Run nix-collect-garbage -d on all live NixOS hosts.
#
# The host list is rebuilt on every run:
# 1. This workstation (nixos) — always first
# 2. pve1 — always second (non-NixOS Proxmox node with Nix installed)
# 3. Every NixOS guest currently running on pve1 (discovered via pct/qm list)
#
# nix-cache is excluded: gc-ing the shared binary cache evicts store paths
# that other hosts depend on for substitution.
#
# NixOS hosts: tries "sudo -n nix-collect-garbage -d" first (works when
# wheelNeedsPassword = false, e.g. the HA cluster). Falls back to user-level
# "nix-collect-garbage -d" if sudo needs a password — still collects
# unreferenced store paths and old nixos-user profile generations, but leaves
# old system generations in place.
# pve1: runs "nix-collect-garbage -d" as the login user (no system generations
# on a non-NixOS host).
#
# Usage (from repo root):
# bash scripts/gc-hosts.sh [--dry-run]
set -euo pipefail
cd "$(dirname "$0")/.."
source scripts/env.sh 2>/dev/null || true
source scripts/lib/nix-eval.sh 2>/dev/null || true
# ── config ────────────────────────────────────────────────────────────────────
: "${MAX_JOBS:=8}"
: "${NIXOS_USER:=nixos}"
: "${PVE1_SSH_USER:=${PROXMOX_SSH_USER:-wayne}}"
SSH_OPTS=(-o StrictHostKeyChecking=no -o BatchMode=yes -o ConnectTimeout=10)
DRY_RUN=0
for arg in "$@"; do
case "$arg" in
--dry-run) DRY_RUN=1 ;;
*) echo "Unknown option: $arg" >&2; exit 1 ;;
esac
done
# ── build the host list ───────────────────────────────────────────────────────
# ORDERED_HOSTS: names in display/execution order.
# HOST_TARGET[name]: SSH target string (user@host).
# HOST_TYPE[name]: "nixos" (try sudo gc, fallback user) | "nix" (user gc only).
declare -a ORDERED_HOSTS=()
declare -A HOST_TARGET=()
declare -A HOST_TYPE=()
declare -A _SEEN_HOSTNAMES=() # dedup tracker
_add_host() {
local name="$1" target="$2" type="$3"
if [[ -n "${_SEEN_HOSTNAMES[$name]+_}" ]]; then return; fi
_SEEN_HOSTNAMES[$name]=1
ORDERED_HOSTS+=("$name")
HOST_TARGET[$name]="$target"
HOST_TYPE[$name]="$type"
}
# 1. Workstation (hard-wired first)
_add_host "nixos" "${NIXOS_USER}@nixos" "nixos"
# 2. pve1 (hard-wired second; non-NixOS, no system generations)
_add_host "pve1" "${PVE1_SSH_USER}@${PVE1_HOST}" "nix"
# 3. Dynamically discover running NixOS guests on pve1
echo "Discovering running guests on ${PVE1_HOST}..."
# Evaluate the full flake hostname map in one shot.
hostname_map="{}"
if ! hostname_map="$(
nix eval --json "${NIX_EVAL_FLAGS[@]}" .#nixosConfigurations \
--apply 'cfgs: builtins.mapAttrs (_: cfg: cfg.config.networking.hostName) cfgs' \
2>/dev/null
)"; then
echo " warning: flake eval failed — skipping dynamic host discovery" >&2
fi
# Get names of all currently running guests from pve1.
if ssh "${SSH_OPTS[@]}" "${PVE1_SSH_USER}@${PVE1_HOST}" "true" 2>/dev/null; then
running_guests="$(
ssh "${SSH_OPTS[@]}" "${PVE1_SSH_USER}@${PVE1_HOST}" bash <<'REMOTE'
{ sudo pct list 2>/dev/null | awk 'NR>1 && $2=="running" { print $NF }';
sudo qm list 2>/dev/null | awk 'NR>1 && $3=="running" { print $2 }'; } | sort -u
REMOTE
)" || running_guests=""
while IFS= read -r guest; do
[[ -z "$guest" ]] && continue
# Resolve flake target name → NixOS hostname.
hostname="$(printf '%s' "$hostname_map" \
| jq -r --arg g "$guest" '.[$g] // empty' 2>/dev/null || true)"
[[ -z "$hostname" ]] && continue
# Exclude nix-cache and any target whose hostname is already in our list.
case "$hostname" in nix-cache) continue ;; esac
if [[ -n "${_SEEN_HOSTNAMES[$hostname]+_}" ]]; then continue; fi
echo " + $guest$hostname"
_add_host "$hostname" "${NIXOS_USER}@${hostname}" "nixos"
done <<< "$running_guests"
else
echo " warning: ${PVE1_HOST} unreachable — skipping dynamic host discovery" >&2
fi
echo ""
echo "Hosts: ${ORDERED_HOSTS[*]}"
echo ""
# ── dry-run ───────────────────────────────────────────────────────────────────
if [[ "$DRY_RUN" -eq 1 ]]; then
echo "[dry-run] commands that would run:"
for host in "${ORDERED_HOSTS[@]}"; do
target="${HOST_TARGET[$host]}"
type="${HOST_TYPE[$host]}"
if [[ "$type" == "nixos" ]]; then
echo " ssh ${SSH_OPTS[*]} $target 'sudo -n nix-collect-garbage -d'"
echo " # fallback: ssh ... $target 'nix-collect-garbage -d'"
else
echo " ssh ${SSH_OPTS[*]} $target '. /nix/var/nix/profiles/default/etc/profile.d/nix-daemon.sh && nix-collect-garbage -d'"
fi
done
exit 0
fi
# ── gc worker ─────────────────────────────────────────────────────────────────
gc_one() {
local host="$1" target="${HOST_TARGET[$1]}" type="${HOST_TYPE[$1]}" logfile="$2"
if ! ssh "${SSH_OPTS[@]}" "$target" "true" 2>>"$logfile"; then
echo "unreachable"; return
fi
if [[ "$type" == "nixos" ]]; then
if ssh "${SSH_OPTS[@]}" "$target" "sudo -n nix-collect-garbage -d" \
>>"$logfile" 2>>"$logfile"; then
echo "ok(sudo)"; return
fi
echo "[sudo needs password — falling back to user-level gc]" >>"$logfile"
if ssh "${SSH_OPTS[@]}" "$target" "nix-collect-garbage -d" \
>>"$logfile" 2>>"$logfile"; then
echo "ok(user)"; return
fi
else
# Non-NixOS node: BatchMode SSH doesn't source the Nix daemon profile, so
# nix-collect-garbage won't be on PATH unless we source it explicitly.
local nix_profile='. /nix/var/nix/profiles/default/etc/profile.d/nix-daemon.sh 2>/dev/null || true'
if ssh "${SSH_OPTS[@]}" "$target" "$nix_profile && nix-collect-garbage -d" \
>>"$logfile" 2>>"$logfile"; then
echo "ok"; return
fi
fi
echo "failed:$?"
}
# ── parallel execution ────────────────────────────────────────────────────────
echo "Running gc on ${#ORDERED_HOSTS[@]} hosts (up to ${MAX_JOBS} parallel)..."
echo ""
TMPDIR_GC="$(mktemp -d)"
trap 'rm -rf "$TMPDIR_GC"' EXIT
declare -A LOGS=()
job_count=0
for host in "${ORDERED_HOSTS[@]}"; do
logfile="${TMPDIR_GC}/${host}.log"
resultfile="${TMPDIR_GC}/${host}.result"
LOGS[$host]="$logfile"
: > "$logfile"
( result="$(gc_one "$host" "$logfile")"; echo "$result" > "$resultfile" ) &
(( job_count++ )) || true
if [[ "$job_count" -ge "$MAX_JOBS" ]]; then
wait -n 2>/dev/null || wait
(( job_count-- )) || true
fi
done
wait
# ── summary ───────────────────────────────────────────────────────────────────
echo "Results:"
echo "──────────────────────────────"
ok_hosts=()
warn_hosts=()
fail_hosts=()
for host in "${ORDERED_HOSTS[@]}"; do
result="$(cat "${TMPDIR_GC}/${host}.result" 2>/dev/null || echo "failed:missing")"
case "$result" in
ok|"ok(sudo)"|"ok(user)")
printf " %-22s %s\n" "$host" "$result"
ok_hosts+=("$host") ;;
unreachable)
printf " %-22s UNREACHABLE\n" "$host"
warn_hosts+=("$host") ;;
*)
printf " %-22s FAILED (%s)\n" "$host" "$result"
fail_hosts+=("$host") ;;
esac
done
echo ""
echo " ${#ok_hosts[@]} succeeded, ${#warn_hosts[@]} unreachable, ${#fail_hosts[@]} failed"
for host in "${warn_hosts[@]+"${warn_hosts[@]}"}" "${fail_hosts[@]+"${fail_hosts[@]}"}"; do
logfile="${LOGS[$host]}"
if [[ -s "$logfile" ]]; then
echo ""
echo "── $host ──"
cat "$logfile"
fi
done
echo ""
[[ "${#fail_hosts[@]}" -eq 0 ]]
+9 -11
View File
@@ -30,6 +30,8 @@ DATA_DISK_SLOT="${DATA_DISK_SLOT:-scsi1}" # Proxmox disk name (scsi1 = data di
HA_USER="${HA_USER:-nixos}" HA_USER="${HA_USER:-nixos}"
PVE_HOST="${PVE_HOST:-${PVE1_HOST}}" PVE_HOST="${PVE_HOST:-${PVE1_HOST}}"
PVE_SSH_USER="${PVE_SSH_USER:-${PROXMOX_SSH_USER}}" PVE_SSH_USER="${PVE_SSH_USER:-${PROXMOX_SSH_USER}}"
PVE_SUDO=""
[[ "$PVE_SSH_USER" != "root" ]] && PVE_SUDO="sudo"
# By-id symlink for the data disk; basename resolves to the raw block device. # By-id symlink for the data disk; basename resolves to the raw block device.
# matches variables.nix's haServerDrbdDisk. # matches variables.nix's haServerDrbdDisk.
DATA_DISK_BYID="${DATA_DISK_BYID:-scsi-0QEMU_QEMU_HARDDISK_drive-${DATA_DISK_SLOT}}" DATA_DISK_BYID="${DATA_DISK_BYID:-scsi-0QEMU_QEMU_HARDDISK_drive-${DATA_DISK_SLOT}}"
@@ -140,7 +142,7 @@ fi
echo "" echo ""
echo "Looking up VM IDs on ${PVE_HOST}..." echo "Looking up VM IDs on ${PVE_HOST}..."
QM_LIST=$(pve "qm list 2>/dev/null" || true) QM_LIST=$(pve "$PVE_SUDO qm list 2>/dev/null" || true)
VMID1=$(echo "$QM_LIST" | awk -v name="$NODE1" '$0 ~ name {print $1}' | head -1) VMID1=$(echo "$QM_LIST" | awk -v name="$NODE1" '$0 ~ name {print $1}' | head -1)
VMID2=$(echo "$QM_LIST" | awk -v name="$NODE2" '$0 ~ name {print $1}' | head -1) VMID2=$(echo "$QM_LIST" | awk -v name="$NODE2" '$0 ~ name {print $1}' | head -1)
@@ -182,12 +184,12 @@ echo "── Phase 1 — Proxmox disk resize (${DATA_DISK_SLOT} ${SIZE} on both
echo " ${DRY_PREFIX}qm resize $VMID1 ${DATA_DISK_SLOT} ${SIZE} ($NODE1 on ${PVE_HOST})" echo " ${DRY_PREFIX}qm resize $VMID1 ${DATA_DISK_SLOT} ${SIZE} ($NODE1 on ${PVE_HOST})"
if ! $DRY_RUN; then if ! $DRY_RUN; then
pve "qm resize $VMID1 ${DATA_DISK_SLOT} ${SIZE}" pve "$PVE_SUDO qm resize $VMID1 ${DATA_DISK_SLOT} ${SIZE}"
fi fi
echo " ${DRY_PREFIX}qm resize $VMID2 ${DATA_DISK_SLOT} ${SIZE} ($NODE2 on ${PVE_HOST})" echo " ${DRY_PREFIX}qm resize $VMID2 ${DATA_DISK_SLOT} ${SIZE} ($NODE2 on ${PVE_HOST})"
if ! $DRY_RUN; then if ! $DRY_RUN; then
pve "qm resize $VMID2 ${DATA_DISK_SLOT} ${SIZE}" pve "$PVE_SUDO qm resize $VMID2 ${DATA_DISK_SLOT} ${SIZE}"
fi fi
echo " Phase 1 done." echo " Phase 1 done."
@@ -202,17 +204,13 @@ rescan_node() {
local node_name=$1 run_fn=$2 local node_name=$1 run_fn=$2
# Resolve block device name from the stable by-id symlink on the guest. # Resolve block device name from the stable by-id symlink on the guest.
# Read-only lookup — safe to run even in dry-run so we show the real device.
local blk_dev="" local blk_dev=""
if ! $DRY_RUN; then
blk_dev=$($run_fn "bash -c 'basename \$(readlink -f /dev/disk/by-id/${DATA_DISK_BYID})'" 2>/dev/null || true) blk_dev=$($run_fn "bash -c 'basename \$(readlink -f /dev/disk/by-id/${DATA_DISK_BYID})'" 2>/dev/null || true)
fi
if [[ -z "$blk_dev" ]]; then if [[ -z "$blk_dev" ]]; then
# Fall back: scsi1 → index 1 → sdb, scsi2 → sdc, etc. echo " ERROR: /dev/disk/by-id/${DATA_DISK_BYID} not found on $node_name" >&2
local slot_idx echo " Check DATA_DISK_BYID or DATA_DISK_SLOT configuration." >&2
slot_idx=$(echo "$DATA_DISK_SLOT" | grep -oE '[0-9]+$' || echo "1") exit 1
blk_dev=$(printf "sd%s" "$(echo "abcdefghij" | cut -c$((slot_idx + 1)))")
echo " WARNING: could not resolve /dev/disk/by-id/${DATA_DISK_BYID} on $node_name;" \
"falling back to /dev/${blk_dev}" >&2
fi fi
echo " ${DRY_PREFIX}Rescanning /dev/${blk_dev} on ${node_name}..." echo " ${DRY_PREFIX}Rescanning /dev/${blk_dev} on ${node_name}..."