Archived
Check NixOS configurations / eval-hosts (pull_request) Successful in 10m40s
Documentation fixes:
- README/AGENTS: rename tailscale-exit-node → tailscale-router, add ha-server
build type and proxmox-ha-server-{1,2} host table rows, add baremetal to
platform list, remove references to non-existent flake-target-refactor-spec.md
and remove-sensetive-info-refactor.md
- docs/auto-installer.md: fix lxc-tailscale-exit-node → lxc-tailscale-router,
add pxe-minimal to the flake outputs list
- variables.nix: fix domainControllerIp comment — IPA is the authoritative DNS
at .253 (Pi-hole is gone), not a forwarding intermediary
Code deduplication:
- Extract duplicate SSH host-key preservation activation scripts from
modules/platforms/lxc.nix and modules/platforms/proxmox.nix into a shared
modules/common/preserve-ssh-host-key.nix; both platforms now import it
- Replace 8-line hand-enumerated NFS export lists in server.nix and ha-server.nix
with a mkNfsExports helper that generates exports from vars.nfsShares — adding
a share to variables.nix now propagates to both exporters automatically
Dead code removal:
- modules/common/configuration.nix: remove leftover NixOS skeleton comments
(hardware-configuration import, grub lines) that were never used
- modules/docker/enable-service.nix: remove commented-out listenOptions and
daemon.settings blocks
- hosts/server/host.nix, hosts/nix-cache/host.nix: remove #DOCKER_HOST comments
Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
209 lines
10 KiB
Nix
209 lines
10 KiB
Nix
{ config, lib, modulesPath, flakeTarget, ... }:
|
|
|
|
let
|
|
# Bakes this exact flake target's pre-generated SSH host key straight
|
|
# into /etc/ssh/ -- mirrors modules/installer/host-keys.nix's
|
|
# builtins.getEnv pattern (impure and empty under normal `nix
|
|
# build`/`nix eval`, so this is a no-op unless explicitly opted into
|
|
# with NIXOS_HOST_KEYS_DIR=... --impure), but places the key directly
|
|
# rather than staging it under /etc/host-keys/ for a later manual copy
|
|
# -- this is the whole system for a `lxc-*` host, built straight to a
|
|
# pct-restorable tarball with no install step, so there's no later copy
|
|
# step to stage for.
|
|
#
|
|
# Without this, config.system.build.tarball's built-in system just
|
|
# generates a fresh host key at first boot like any other host would --
|
|
# but sops-nix derives its decryption key from *this* file, and
|
|
# .sops.yaml only trusts whatever key scripts/secrets/sync-host-keys.sh already
|
|
# registered for this exact target name. A freshly-generated key can
|
|
# never match that, so every secret (including this host's own login)
|
|
# permanently fails to decrypt. Confirmed live: sops-install-secrets
|
|
# errored with "Error getting data key: 0 successful groups required,
|
|
# got 0" -- the container's actual host key's age fingerprint didn't
|
|
# match the one registered in .sops.yaml at all.
|
|
hostKeysDirStr = builtins.getEnv "NIXOS_HOST_KEYS_DIR";
|
|
hasHostKeysDir = hostKeysDirStr != "" && builtins.pathExists hostKeysDirStr;
|
|
hostKeysDir = /. + hostKeysDirStr;
|
|
|
|
# flakeTarget ("${platform}-${buildType}") comes in via specialArgs from
|
|
# flake.nix's mkTarget -- exactly the name scripts/secrets/sync-host-keys.sh
|
|
# registers keys under. Deliberately not read back from
|
|
# config.environment.etc."flake-target" (which is set to the same value)
|
|
# -- this module also *contributes* to environment.etc below, and a
|
|
# module reading the merged value of an option it's still defining is a
|
|
# circular dependency (confirmed: "infinite recursion encountered").
|
|
privKeyFile = hostKeysDir + "/${flakeTarget}_ssh_host_ed25519_key";
|
|
pubKeyFile = hostKeysDir + "/${flakeTarget}_ssh_host_ed25519_key.pub";
|
|
hasKeyForThisTarget =
|
|
hasHostKeysDir
|
|
&& builtins.pathExists privKeyFile
|
|
&& builtins.pathExists pubKeyFile;
|
|
in
|
|
{
|
|
# LXC containers share the host kernel — Proxmox starts them by exec'ing
|
|
# /sbin/init directly, no bootloader/initrd involved — and Proxmox has its
|
|
# own container hostname/network provisioning outside Nix. nixpkgs' own
|
|
# virtualisation/proxmox-lxc.nix module already handles all of this
|
|
# correctly (boot.isContainer, loader.initScript, systemd-networkd) and,
|
|
# critically, provides config.system.build.tarball — a directly
|
|
# `pct restore`-able container image, no nixos-install/bind-mount needed
|
|
# (nixos-install refuses to touch the filesystem it's currently running
|
|
# on, which is exactly what bind-mounting / onto /mnt for an installer
|
|
# LXC container does).
|
|
imports = [
|
|
(modulesPath + "/virtualisation/proxmox-lxc.nix")
|
|
../common/preserve-ssh-host-key.nix
|
|
];
|
|
|
|
proxmoxLXC = {
|
|
# host.nix declares each host's real hostname (networking.hostName);
|
|
# keep that instead of letting Proxmox's ambient container config win.
|
|
manageHostName = true;
|
|
# Unprivileged by default -- matches how these containers are actually
|
|
# created (scripts/proxmox/create-proxmox-resource.sh reads this value
|
|
# back to decide `pct create`'s --unprivileged flag, so the two stay
|
|
# in sync).
|
|
#
|
|
# Any lxc-* host with an NFS fileSystem must be privileged: the kernel's
|
|
# NFS client doesn't set FS_USERNS_MOUNT, so mounting NFS from inside
|
|
# *any* non-init user namespace -- which is exactly what an unprivileged
|
|
# container's UID-mapped root runs in -- is rejected at the VFS layer
|
|
# with EPERM, no matter what Proxmox's own `mount=nfs;nfs4` container
|
|
# feature allows at the AppArmor layer (confirmed live: TCP to the NFS
|
|
# server succeeds, the server's export table matches the container's IP,
|
|
# and `mount.nfs: Operation not permitted` still fires immediately with
|
|
# no corresponding denial anywhere in the server's logs -- a kernel-level
|
|
# rejection, not a network or export-permission one). Deriving this from
|
|
# fileSystems rather than a per-host override keeps it self-consistent:
|
|
# any new lxc-* host that declares an NFS mount automatically gets the
|
|
# privilege level it needs without a separate manual flag.
|
|
privileged = builtins.any
|
|
(fs: fs.fsType == "nfs" || fs.fsType == "nfs4")
|
|
(builtins.attrValues config.fileSystems);
|
|
};
|
|
|
|
boot.loader = {
|
|
grub.enable = false;
|
|
systemd-boot.enable = false;
|
|
};
|
|
|
|
# NetworkManager depends on a running udevd to enumerate/classify devices,
|
|
# which boot.isContainer disables (see nixpkgs' container-config.nix) —
|
|
# that's what broke DHCP-hostname registration in Pi-hole. The imported
|
|
# proxmox-lxc.nix module already switches networking to systemd-networkd
|
|
# for the same reason; it just doesn't disable NetworkManager itself,
|
|
# which modules/common/configuration.nix enables for every host.
|
|
networking.networkmanager.enable = lib.mkForce false;
|
|
|
|
environment.etc = lib.mkIf hasKeyForThisTarget {
|
|
"ssh/ssh_host_ed25519_key" = {
|
|
source = privKeyFile;
|
|
mode = "0600";
|
|
};
|
|
"ssh/ssh_host_ed25519_key.pub" = {
|
|
source = pubKeyFile;
|
|
mode = "0644";
|
|
};
|
|
};
|
|
|
|
# virtualisation/proxmox-lxc.nix (imported above) registers the Nix
|
|
# store DB via a systemd service (register-nix-paths) -- it never runs
|
|
# an activation script at all. Confirmed live this means neither
|
|
# sops-nix's "for users" secrets (password hashes -- installed by the
|
|
# activation script itself, not a systemd service, since they need to
|
|
# exist *before* user creation) nor the user-creation step that
|
|
# consumes them ever run on a real lxc-* boot. In this config sops-nix
|
|
# does NOT generate its own boot-time service (confirmed live: no
|
|
# sops-nix.service in systemctl list-unit-files on a deployed
|
|
# lxc-tor-relay container); /run/secrets is a tmpfs cleared on every
|
|
# reboot, so secrets must be reinstalled on each non-first boot by
|
|
# nixos-lxc-sops-reinstall (below).
|
|
#
|
|
# A systemd service, not boot.postBootCommands: tried that first (it's
|
|
# a genuine, generally-invoked hook -- nixos/modules/system/boot/stage-2-init.sh,
|
|
# which becomes this container's actual /sbin/init, unconditionally
|
|
# runs it) but switch-to-configuration behaves differently that early in
|
|
# boot (raw stage-2-init.sh, before systemd itself has even started) --
|
|
# confirmed live it silently failed to rewrite /etc/shadow from there
|
|
# even in "test" mode, despite the exact same command working reliably
|
|
# every time when run post-boot (i.e. as a normal systemd service, which
|
|
# is what this is). Not fully root-caused why the early context
|
|
# specifically breaks it; a real systemd service sidesteps needing to.
|
|
#
|
|
# /etc/shadow already has PLACEHOLDER entries for every declared user
|
|
# baked in at build time (part of constructing the system closure).
|
|
# update-users-groups.pl deliberately never overwrites an *existing*
|
|
# shadow entry -- a correct safety property in general (don't clobber a
|
|
# real user's real password on a config rebuild) -- but on a genuine
|
|
# first boot that only means the real hashedPasswordFile-derived hash
|
|
# never gets the chance to be applied either, since the placeholder is
|
|
# already "seen". Safe to clear here specifically: there is no real
|
|
# password yet to protect on a first boot.
|
|
#
|
|
# "test" mode, not "boot": confirmed live "boot" mode aborts partway
|
|
# through (before rewriting /etc/shadow) on a warning that "/boot" is on
|
|
# a different filesystem -- a real check for a host with a bootloader to
|
|
# update, meaningless for a container that has none
|
|
# (boot.loader.{grub,systemd-boot}.enable are both false above), but it
|
|
# still aborts the script. "test" runs every activation step without
|
|
# touching boot-loader state at all.
|
|
#
|
|
# ConditionPathExists (systemd-native, not a bash-level check) means
|
|
# this only ever runs once, on the genuine first boot -- systemd itself
|
|
# skips even starting it on every later boot once the marker exists.
|
|
# switch-to-configuration is otherwise the operator's call per this
|
|
# repo's own safety rules, not something to run on every boot.
|
|
systemd.services.nixos-lxc-first-boot-activate = {
|
|
description = "Complete first-boot NixOS activation (users, secrets) for this LXC container";
|
|
wantedBy = [ "multi-user.target" ];
|
|
unitConfig.ConditionPathExists = "!/var/lib/nixos-lxc-first-boot-activated";
|
|
serviceConfig = {
|
|
Type = "oneshot";
|
|
RemainAfterExit = true;
|
|
};
|
|
script = ''
|
|
rm -f /etc/shadow
|
|
/run/current-system/bin/switch-to-configuration test
|
|
mkdir -p /var/lib
|
|
touch /var/lib/nixos-lxc-first-boot-activated
|
|
'';
|
|
};
|
|
|
|
# Reinstalls sops secrets on every non-first boot. /run/secrets is a
|
|
# tmpfs that is cleared on each reboot; without this service, secrets
|
|
# are permanently absent after the first boot and every service that
|
|
# reads from /run/secrets fails on start.
|
|
#
|
|
# wantedBy/before network.target: switch-to-configuration test requires
|
|
# D-Bus to restart systemd targets after running activation scripts. D-Bus
|
|
# is available once basic.target completes (the default After=basic.target
|
|
# that DefaultDependencies would otherwise add). Placing the service before
|
|
# network.target ensures secrets are ready before any network-dependent
|
|
# service (including beszel-agent and nix-serve) starts, while running late
|
|
# enough that D-Bus is already up.
|
|
#
|
|
# ConditionPathExists=... skips this service on the genuine first boot
|
|
# (the marker doesn't exist yet); nixos-lxc-first-boot-activate handles
|
|
# that case. On every subsequent boot the condition passes and secrets
|
|
# are reinstalled before user services start.
|
|
#
|
|
# SuccessExitStatus=11: switch-to-configuration exits 11 when it cannot
|
|
# acquire the activation lock (another switch is already in progress).
|
|
# During a nixos-rebuild switch the activation already installs secrets, so
|
|
# treating the lock-held case as success is correct.
|
|
systemd.services.nixos-lxc-sops-reinstall = {
|
|
description = "Reinstall sops secrets on each non-first boot (LXC, /run is tmpfs)";
|
|
wantedBy = [ "network.target" ];
|
|
before = [ "network.target" ];
|
|
unitConfig.ConditionPathExists = "/var/lib/nixos-lxc-first-boot-activated";
|
|
serviceConfig = {
|
|
Type = "oneshot";
|
|
RemainAfterExit = true;
|
|
SuccessExitStatus = "11";
|
|
};
|
|
script = ''
|
|
/run/current-system/bin/switch-to-configuration test
|
|
'';
|
|
};
|
|
}
|