Archived
Check NixOS configurations / eval-hosts (pull_request) Failing after 30m6s
scripts/ had grown to 10 top-level scripts covering three distinct concerns (sops/age + SSH host-key management, Proxmox deployment, and repo-wide bootstrap/CI) with no grouping. Move the key-management scripts (backup-admin-key.sh, rotate-admin-key.sh, prepare-host-key.sh, sync-host-keys.sh) into scripts/secrets/, and the Proxmox scripts (create-proxmox-resource.sh, configure-nix-cache-client.sh) into scripts/proxmox/; leave env.sh, codex-setup.sh, codex-maintenance.sh, and bump-nixpkgs-release.sh at the top level (frequently hand-typed or pure shared config) and scripts/lib/ as-is. Updates every cross-reference: each moved script's repo_root computation (now one directory deeper), shellcheck source= directives, inter-script paths (create-proxmox-resource.sh's call into sync-host-keys.sh and its remote bootstrap of configure-nix-cache-client.sh on the Proxmox node), and every doc/module mention (CLAUDE.md's Scripts section reorganized to match, README.md, docs/auto-installer.md, docs/proxmox-images.md, modules/installer/common.nix, modules/platforms/lxc.nix). CI workflows need no change -- they only invoke codex-maintenance.sh, which didn't move. Verified via bash -n, shellcheck (no new warnings beyond the pre-existing SC1091/SC2029/SC2095 baseline), and live dry-runs of sync-host-keys.sh --all and create-proxmox-resource.sh --list from their new paths. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
165 lines
7.8 KiB
Bash
Executable File
165 lines
7.8 KiB
Bash
Executable File
#!/usr/bin/env bash
|
|
# Shared config for scripts/*.sh. Source this instead of hardcoding a
|
|
# second copy of these values in every script:
|
|
# source "$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)/env.sh"
|
|
# Every variable can still be overridden per-invocation via the
|
|
# environment (e.g. PROXMOX_STORAGE=tank-nvme ./scripts/proxmox/create-proxmox-resource.sh ...)
|
|
# since each one only sets a default if unset.
|
|
|
|
# SSH-reachable Proxmox node that scripts/proxmox/create-proxmox-resource.sh runs
|
|
# pct/qm on. Matches the Proxmox web UI hostname already used in
|
|
# hosts/nixos/home.nix's desktop shortcuts (pve.<homeDomain> from
|
|
# variables.nix) -- change this if that's not actually reachable over SSH,
|
|
# or if you're targeting a different node in a multi-node cluster.
|
|
: "${PROXMOX_HOST:=pve.sweet.home}"
|
|
: "${PROXMOX_SSH_USER:=root}"
|
|
|
|
# Where this flake repo lives on the Proxmox node itself.
|
|
# scripts/proxmox/create-proxmox-resource.sh builds images directly on the node
|
|
# instead of transferring them over the network -- it clones the repo here
|
|
# (from this checkout's own `origin` remote) the first time it doesn't
|
|
# find it, installing build tooling via scripts/codex-setup.sh, then
|
|
# `git pull`s it before every subsequent build.
|
|
: "${PROXMOX_REMOTE_REPO_DIR:=/root/nixos}"
|
|
|
|
# Storage pool names -- Proxmox's own stock-install defaults, but this
|
|
# varies a lot by setup (ZFS pool name, custom LVM-thin volume, etc.).
|
|
# Verify with `pvesm status` on the node and correct these if wrong.
|
|
: "${PROXMOX_STORAGE:=local-lvm}" # VM disks / CT rootfs
|
|
: "${PROXMOX_ISO_STORAGE:=local}" # uploaded images/ISOs/CT templates
|
|
|
|
: "${PROXMOX_BRIDGE:=vmbr0}"
|
|
|
|
# Fallback resource sizing when a script doesn't get --cores/--memory.
|
|
: "${PROXMOX_DEFAULT_CORES:=2}"
|
|
: "${PROXMOX_DEFAULT_MEMORY_MB:=2048}"
|
|
|
|
# `pct create` (unlike `pct restore`) requires an explicit rootfs size --
|
|
# no backup metadata to infer it from. Matches Proxmox's own GUI default.
|
|
: "${PROXMOX_DEFAULT_LXC_DISK_GB:=8}"
|
|
|
|
# `pct create --memory` only sets RAM -- swap is a wholly separate
|
|
# parameter that otherwise silently stays at Proxmox's own 512M default
|
|
# regardless of --memory (confirmed: creating with --memory 2048 left
|
|
# swap at 512). create-proxmox-resource.sh defaults --swap to whatever
|
|
# --memory resolves to at runtime rather than a static value here, so it
|
|
# tracks a --memory picked at the CLI too, not just the default above.
|
|
|
|
# Required for a modern (v247+) systemd guest to actually boot as an
|
|
# unprivileged container: systemd's routine use of nested user namespaces
|
|
# and credential mounts (LoadCredential=, DynamicUser=, etc. -- used even
|
|
# by plain getty units) gets denied by AppArmor's default LXC confinement
|
|
# without these. Confirmed live: without them, every getty unit
|
|
# crash-loops on a denied `/run/credentials/*` mount every ~3s (visible
|
|
# as garbage on the console) and core services like nsncd fail the same
|
|
# way on userns_create; system.build.tarball never finishes activating.
|
|
#
|
|
# mount=nfs;nfs4: without it, AppArmor blanket-denies the `nfs`/
|
|
# `rpc_pipefs` mount syscalls any NFS client share needs -- confirmed
|
|
# live on lxc-docker (which mounts several, see modules/docker/mount-data.nix
|
|
# and modules/raspi/mount-data.nix): `mount: /var/lib/nfs/rpc_pipefs:
|
|
# permission denied`. Harmless to grant on lxc targets that don't mount
|
|
# NFS at all -- it only widens what the container is *allowed* to mount,
|
|
# nothing here forces a mount to happen.
|
|
: "${PROXMOX_DEFAULT_LXC_FEATURES:=nesting=1,keyctl=1,mount=nfs;nfs4}"
|
|
|
|
export PROXMOX_HOST PROXMOX_SSH_USER PROXMOX_STORAGE PROXMOX_ISO_STORAGE \
|
|
PROXMOX_BRIDGE PROXMOX_DEFAULT_CORES PROXMOX_DEFAULT_MEMORY_MB \
|
|
PROXMOX_DEFAULT_LXC_DISK_GB PROXMOX_DEFAULT_LXC_FEATURES \
|
|
PROXMOX_REMOTE_REPO_DIR
|
|
|
|
# Matches variables.nix's nixCacheHost -- update both if it ever changes.
|
|
: "${NIX_CACHE_HOST:=nix-cache}"
|
|
export NIX_CACHE_HOST
|
|
|
|
# nix_extra_opts: call as a plain statement (NOT inside $(...)/<(...) --
|
|
# that forks a subshell, and the whole point is exporting a decision back
|
|
# into *this* shell) to populate the global NIX_OPTS array with whatever
|
|
# extra `nix`/`nix-shell` CLI options are needed to avoid nix-cache when
|
|
# it's unreachable:
|
|
# nix_extra_opts
|
|
# nix build "${NIX_OPTS[@]}" ...
|
|
#
|
|
# Without this, every single `nix eval`/`nix build` call retries each
|
|
# store path against a dead substituter up to 5 times with backoff
|
|
# (confirmed: ~15s+ per lookup even with a short connect-timeout, because
|
|
# nix's own retry count isn't controllable that way), and separately
|
|
# tries it as a remote builder too -- both fail independently, so both
|
|
# are checked.
|
|
#
|
|
# Checked with a single fast `curl`/TCP probe (bypassing nix's retry logic
|
|
# entirely) the first time this is called in a given process, and the
|
|
# result is exported as NIX_EXTRA_OPTS so a script that shells out to
|
|
# another script in this repo (e.g. create-proxmox-resource.sh calling
|
|
# sync-host-keys.sh) reuses the same decision instead of probing twice.
|
|
declare -a NIX_OPTS=()
|
|
|
|
nix_extra_opts() {
|
|
if [[ -n "${NIX_EXTRA_OPTS_DECIDED:-}" ]]; then
|
|
if [[ -n "${NIX_EXTRA_OPTS:-}" ]]; then
|
|
eval "NIX_OPTS=(${NIX_EXTRA_OPTS})"
|
|
else
|
|
NIX_OPTS=()
|
|
fi
|
|
return
|
|
fi
|
|
export NIX_EXTRA_OPTS_DECIDED=1
|
|
NIX_OPTS=()
|
|
|
|
# Retry a couple of times, 1s apart, before believing either check --
|
|
# belt-and-suspenders against a genuine multi-second blip (nix-cache
|
|
# restarting), on top of the fix below. Worst case (~11s total, host
|
|
# genuinely gone) is still nowhere near the 15s+ *per lookup* nix's own
|
|
# substituter retries would cost if this check didn't exist at all.
|
|
local attempt cache_up=0 builder_up=0
|
|
for attempt in 1 2 3; do
|
|
if curl --silent --fail --max-time 3 "http://${NIX_CACHE_HOST}/nix-cache-info" >/dev/null 2>&1; then
|
|
cache_up=1
|
|
break
|
|
fi
|
|
[[ "$attempt" -lt 3 ]] && sleep 1
|
|
done
|
|
|
|
if [[ "$cache_up" -eq 0 ]]; then
|
|
echo "nix-cache (http://${NIX_CACHE_HOST}) is unreachable -- skipping it (substituter + remote builder) for the rest of this run." >&2
|
|
NIX_OPTS=(--option substituters "https://cache.nixos.org/" --builders "")
|
|
else
|
|
for attempt in 1 2 3; do
|
|
# `exec 3<>/dev/tcp/...` just opens the fd and returns -- it does NOT
|
|
# read from it. Confirmed live this is load-bearing, not stylistic:
|
|
# the previous `cat < /dev/tcp/.../22` blocked forever and always hit
|
|
# the timeout even against a perfectly healthy nix-cache, because
|
|
# sshd sends its banner and then holds the connection open waiting
|
|
# for the client to speak next -- `cat` never sees EOF, so this
|
|
# check reported "unreachable" unconditionally, 100% of the time,
|
|
# regardless of whether the remote builder was actually up.
|
|
if timeout 3 bash -c "exec 3<>/dev/tcp/${NIX_CACHE_HOST}/22" 2>/dev/null; then
|
|
builder_up=1
|
|
break
|
|
fi
|
|
[[ "$attempt" -lt 3 ]] && sleep 1
|
|
done
|
|
if [[ "$builder_up" -eq 0 ]]; then
|
|
echo "nix-cache's SSH remote builder (nixremote@${NIX_CACHE_HOST}:22) is unreachable -- disabling remote builds for the rest of this run." >&2
|
|
NIX_OPTS=(--builders "")
|
|
fi
|
|
fi
|
|
|
|
# `printf '%q '` with a genuinely empty NIX_OPTS still runs one format
|
|
# pass over a missing argument and yields the literal `'' ` rather than
|
|
# an empty string (confirmed live) -- a subprocess that later does
|
|
# `eval "NIX_OPTS=(${NIX_EXTRA_OPTS})"` (the branch above, for e.g.
|
|
# sync-host-keys.sh reusing this process's decision) would then rebuild
|
|
# a 1-element array holding an empty string instead of a 0-element
|
|
# array, and `nix-shell "${NIX_OPTS[@]}" -p <pkg>` chokes on that stray
|
|
# element as a bogus positional argument. Guard the empty case
|
|
# explicitly so nix-cache being reachable (NIX_OPTS legitimately empty)
|
|
# round-trips as truly empty instead.
|
|
if [[ ${#NIX_OPTS[@]} -gt 0 ]]; then
|
|
printf -v NIX_EXTRA_OPTS '%q ' "${NIX_OPTS[@]}"
|
|
else
|
|
NIX_EXTRA_OPTS=""
|
|
fi
|
|
export NIX_EXTRA_OPTS
|
|
}
|