Compare commits

..
Author SHA1 Message Date
beatzaplenty 6e601ce85c Merge branch 'main' into worktree-claude-md-pve-guardrails
Check NixOS configurations / eval-hosts (pull_request) Successful in 10m51s
2026-07-21 20:08:56 +00:00
beatzaplenty cbdb288bc3 Merge branch 'main' into worktree-claude-md-pve-guardrails
Check NixOS configurations / eval-hosts (pull_request) Successful in 10m18s
2026-07-21 08:50:26 +00:00
150 changed files with 1245 additions and 9768 deletions
-2
View File
@@ -23,5 +23,3 @@ host-keys/
# Temporary Milestone 1 audit checklist (remove-sensetive-info-refactor.md) # Temporary Milestone 1 audit checklist (remove-sensetive-info-refactor.md)
# - working notes only, never committed, deleted once every row is rotated. # - working notes only, never committed, deleted once every row is rotated.
secrets-inventory.md secrets-inventory.md
.claude/worktrees/
.claude/settings.local.json
+28 -192
View File
@@ -1,28 +1,17 @@
keys: keys:
- &admin age1njap586hc0q43kr03g6c8eqhdsmk8zcafkl3f83xwlc2gqhlmfgs4tmwad - &admin age1njap586hc0q43kr03g6c8eqhdsmk8zcafkl3f83xwlc2gqhlmfgs4tmwad
- &proxmox-minimal age19m0m7vdfg86yqy8l5mmle5jdd0unrn3f55t232w8h5ey42cqw34sfpt32n - &docker age19gfn2yedg76dmztm4hncr7vf3r3c9j0qpt4rap7y7gersjk4m3ks2lhd0e
- &lxc-gui age1rrxqea6q6pn39sw8y5te63h2py8jgjl9v0jyper86w3ggtn67upqg3ah39 - &server age1ll6hj5ggruetgjwjfnplpn5xtq35uhlcdflksx3xmnjm6s3uad9sz70jkf
- &baremetal-gui age1adur9g330gua4l6ndk8cqjg35qc8yxwgme6wrl2hpylcc7vxm38q05ejuy - &nix-cache age120le4a5l8dh3lyfgvmj3d9ksmej6ajs5mer5y7r0vfg3x9fn69dqf8xgzu
- &linode-docker age17e89ty6p0fw24daanen57wg8uald9s025t3wwxsw269svwpmgvrshfvfvt - &nix-minimal age120whqj96g26lsgy4udvgsn8dc9lumh8jeu3a564fx79rjr5lxffqmrljuu
- &linode-gui age1hrx8qj02fj2ea6d4g9vqhyj9hl7fppkjqfdx2l37py3h6pdkr95s8n8rvs - &proxmox-minimal age10at8862478urh0eeuwh8hzln6ck78jgwtztgxatwqlzwagg77y5snm4xzg
- &linode-minimal age1e7l8dusgmgfzd2cxrrzwepzjxt69hzqj4epee0cs27u6yg4kxcuqm34ncx - &lxc-nix-cache age1xjst4frdh0th6q8m7p7u9g5af7ty5jqeum0p6z8a52a9q7st7ewqw8yl9j
- &linode-nix-cache age1jcx3yajjhghn8qh8za3yeu8nxykzlg3p4nrv03vnfvzl0mzayg2qmg940e - &lxc-docker age1ezk9x53zt8kcnscdm80jcyf0xq97vndv7jsn3rl8cc0cwm2jmpmq372dzs
- &linode-tailscale-router age1f7usptjx9rv4rxauasve200gxtdt9jkqhhdqstlf20wvlm7u75rsjfw50m - &lxc-minimal age1jy444f9d9stygj4p3w9kh54cqcfr654tvr75tdvee5cxsgtdtc9q3v60ep
- &lxc-docker age17jqc66x9yeshfgd9v78mj483r4zzarqdtuxtrkxe4x5mw679gphshd94th - &lxc-pxe-boot age1fxxzpnfse8nd9wz78ht3m0plrmraacf4cpga0pe8fm2tdnqcgy8q7qsyvp
- &lxc-minimal age1px0h5l9zp2dww0m8fncrc82kfdmzplsfv2ltat7sna28xpg09pqqcl3s2k - &lxc-gui age190htw7prp4vln076dxjx3gxxaq06h0zl0te7cqgpx79vl3lhkaes8suy05
- &lxc-nix-cache age1ufg390ydrmma849t9xfkxxl5xvdkk6mngnlzhmy7mvuaje8sgcmsmnq6l7 - &proxmox-server age1ukpqxzl44mnjpy5r96sfuc5sqzm47u4k8ujjh5qdgy6jvl9uqgpspymqfk
- &lxc-pxe-boot age16j42pdc5dr6wnj7xayhkqdj2rny9u68fcqejs50hqq42scssh4gsnrrnlt - &vm-server age15kh7akxlx7zn00tey79rq2g8lgs4j5y77rcnyfxrxap8ckfu0a9sqvtdhh
- &lxc-tailscale-router age1k7d2du5mejsmv5rzavm4xwgpthqvcfsehduquv28nzs53zppa3kqngfxq2
- &lxc-tor-relay age16kqfmvz4e23hmdlqresnyw69ej604s320mmd49h4hm3fhqchtgyqrws0k2
- &proxmox-docker age1arhf2q45zw6wf2uevju4savp575x3m2tfvved5zzq3ay92ynua9s3cm92c
- &proxmox-gui age19mn8zrxl8zpps9yvrh4euquvygpp4fp8queg7xc6qhtnl4ng8c9qx02qwn
- &proxmox-nix-cache age1jlltcv5jcnm40z5k0q6hv053k2rqpqvemtuecdwn527uw8uqz4es3x7m68
- &proxmox-pxe-boot age1ug787sgt6st6k82fgkrug2lzltw4qsukrrqqs3w27ewwqj8rg4hsxcmylz
- &proxmox-tailscale-router age1zhfyuzlq40reuqlr34gf77852nhs3t6mqfzrqmas8z6sxk7tcfhsungrm0
- &proxmox-ha-server-1 age1k73g8x47hs93wcv7qh92n3htz8pl295g49hyvlrf3570mts0hgys5g04d6
- &proxmox-ha-server-2 age1fefy6dk8zn5c3edwmrs9vwx79quftnt784m628t9e34q3ft3cehqz8u72r
- &proxmox-ha-docker-1 age16y0vfts5v2k20e2rj23qa5lc5gm96gm64uwsqsr0y688xl0ne46qcfr0sm
- &proxmox-ha-docker-2 age1tm3zj5lp3elw2j832az4nj9xxmhqd66ag3cqcv3vskvmd7qdmqmsdpdcn0
creation_rules: creation_rules:
# Shared across every currently-deployed host: root/nixos password hash, # Shared across every currently-deployed host: root/nixos password hash,
@@ -33,189 +22,36 @@ creation_rules:
key_groups: key_groups:
- age: - age:
- *admin - *admin
- *proxmox-minimal - *docker
- *lxc-gui - *server
- *baremetal-gui - *nix-cache
- *linode-docker
- *linode-gui
- *linode-minimal
- *linode-nix-cache
- *linode-tailscale-router
- *lxc-docker
- *lxc-minimal - *lxc-minimal
- *nix-minimal
- *lxc-nix-cache - *lxc-nix-cache
- *proxmox-minimal
- *lxc-docker
- *lxc-pxe-boot - *lxc-pxe-boot
- *lxc-tailscale-router - *lxc-gui
- *lxc-tor-relay - *proxmox-server
- *proxmox-docker - *vm-server
- *proxmox-gui
- *proxmox-nix-cache
- *proxmox-pxe-boot
- *proxmox-tailscale-router
- *proxmox-ha-server-1
- *proxmox-ha-server-2
- *proxmox-ha-docker-1
- *proxmox-ha-docker-2
- path_regex: secrets/nix-cache\.yaml$ - path_regex: secrets/nix-cache\.yaml$
key_groups: key_groups:
- age: - age:
- *admin - *admin
- *linode-nix-cache - *nix-cache
- *lxc-nix-cache - *lxc-nix-cache
- *proxmox-nix-cache
- path_regex: secrets/tor-relay\.yaml$ - path_regex: secrets/server\.yaml$
key_groups: key_groups:
- age: - age:
- *admin - *admin
- *lxc-tor-relay - *server
- *proxmox-server
- *vm-server
- path_regex: secrets/tailscale-router\.yaml$ - path_regex: secrets/docker\.yaml$
key_groups: key_groups:
- age: - age:
- *admin - *admin
- *linode-tailscale-router - *docker
- *lxc-tailscale-router
- *proxmox-tailscale-router
# HA file server per-node secrets (beszel-token).
# proxmox-ha-server-1 / proxmox-ha-server-2 keys are added automatically
# by scripts/secrets/sync-host-keys.sh once the hosts are provisioned;
# until then only the admin key can decrypt these files.
- path_regex: secrets/ha-server-1\.yaml$
key_groups:
- age:
- *admin
- *proxmox-ha-server-1
# proxmox-ha-server-1 added by sync-host-keys.sh
- path_regex: secrets/ha-server-2\.yaml$
key_groups:
- age:
- *admin
- *proxmox-ha-server-2
# proxmox-ha-server-2 added by sync-host-keys.sh
# Shared HA cluster corosync authkey (binary sops file).
# Encrypted for both HA nodes so either can decrypt on boot.
# Both host keys added by sync-host-keys.sh; admin key allows initial creation.
- path_regex: secrets/ha-corosync-authkey$
key_groups:
- age:
- *admin
- *proxmox-ha-server-1
- *proxmox-ha-server-2
# proxmox-ha-server-1 added by sync-host-keys.sh
# proxmox-ha-server-2 added by sync-host-keys.sh
# gui-host-specific secrets (currently: wifi-password, see
# modules/networking/wifi.nix). Only *lxc-gui has a registered key today
# -- proxmox-gui/linode-gui/baremetal-gui haven't been provisioned via
# scripts/secrets/sync-host-keys.sh yet, so whichever variant is actually
# deployed next needs its recipient added here (and `sops updatekeys` rerun)
# before it can decrypt this.
- path_regex: secrets/gui\.yaml$
key_groups:
- age:
- *admin
- *lxc-gui
- *baremetal-gui
- *linode-gui
- *proxmox-gui
# IPA host keytabs (binary sops files).
# Each keytab is encrypted for all platform variants of that host so any
# deployed variant can decrypt it at boot. Run
# scripts/ipa/create-nixos-ipa-host-account.sh <hostname> to enroll a new
# host and produce the keytab; this section is updated by that script.
- path_regex: secrets/nix-cache\.keytab$
key_groups:
- age:
- *admin
- *linode-nix-cache
- *lxc-nix-cache
- *proxmox-nix-cache
- path_regex: secrets/tailscale-router\.keytab$
key_groups:
- age:
- *admin
- *linode-tailscale-router
- *lxc-tailscale-router
- *proxmox-tailscale-router
- path_regex: secrets/pxe-boot\.keytab$
key_groups:
- age:
- *admin
- *lxc-pxe-boot
- *proxmox-pxe-boot
# nixos = the workstation (hosts/nixos/host.nix). All gui platform variants
# share the hostname "nixos" and must be able to decrypt at boot.
- path_regex: secrets/nixos\.keytab$
key_groups:
- age:
- *admin
- *baremetal-gui
- *lxc-gui
- *proxmox-gui
- *linode-gui
- path_regex: secrets/docker\.keytab$
key_groups:
- age:
- *admin
- *linode-docker
- *lxc-docker
- *proxmox-docker
- path_regex: secrets/tor-relay\.keytab$
key_groups:
- age:
- *admin
- *lxc-tor-relay
- path_regex: secrets/nix-minimal\.keytab$
key_groups:
- age:
- *admin
- *lxc-minimal
- *proxmox-minimal
- *linode-minimal
# Host keytab for ha-server-1 FreeIPA enrollment (binary sops file).
# Generated by scripts/ipa/create-nixos-ipa-host-account.sh.
- path_regex: secrets/ha-server-1\.keytab$
key_groups:
- age:
- *admin
- *proxmox-ha-server-1
# proxmox-ha-server-1 added by sync-host-keys.sh
# Host keytab for ha-server-2 FreeIPA enrollment (binary sops file).
# Generated by scripts/ipa/create-nixos-ipa-host-account.sh.
- path_regex: secrets/ha-server-2\.keytab$
key_groups:
- age:
- *admin
- *proxmox-ha-server-2
# proxmox-ha-server-2 added by sync-host-keys.sh
# Host keytab for ha-docker-1 FreeIPA enrollment (binary sops file).
# Generated by scripts/ipa/create-nixos-ipa-host-account.sh.
- path_regex: secrets/ha-docker-1\.keytab$
key_groups:
- age:
- *admin
- *proxmox-ha-docker-1
# Host keytab for ha-docker-2 FreeIPA enrollment (binary sops file).
# Generated by scripts/ipa/create-nixos-ipa-host-account.sh.
- path_regex: secrets/ha-docker-2\.keytab$
key_groups:
- age:
- *admin
- *proxmox-ha-docker-2
+6 -8
View File
@@ -6,14 +6,12 @@ This repository contains flake-based NixOS configurations for Wayne's LAN
servers and workstation. servers and workstation.
The flake exposes NixOS configurations named `<platform>-<buildtype>` The flake exposes NixOS configurations named `<platform>-<buildtype>`
(platforms: `linode`, `proxmox`, `lxc`, `baremetal`; build types: `minimal`, (platforms: `linode`, `proxmox`, `lxc`; build types: `minimal`, `nix-cache`,
`nix-cache`, `docker`, `gui`, `pxe-boot`, `tailscale-router`, `server`, `docker`, `gui`, `pxe-boot`, `tailscale-exit-node`, `tor-relay`), generated from `modules/platforms/*`
`tor-relay`, `ha-server`), generated from `modules/platforms/*` and and `modules/build-types/*` by the `mkTarget` function in `flake.nix`. Not
`modules/build-types/*` by the `mkTarget` function in `flake.nix`. Not every every combination is built — `pxe-boot` has no `linode` variant. See
combination is built — `pxe-boot` has no `linode` variant, `ha-server` only `README.md` for the full current target list; treat `flake.nix` as the
exists on `proxmox`, and `tor-relay` only exists on `lxc`. See `README.md` source of truth since this list can drift.
for the full current target list; treat `flake.nix` as the source of truth
since this list can drift.
Do not deploy, switch, reboot, repartition, format disks, or run destructive Do not deploy, switch, reboot, repartition, format disks, or run destructive
install commands from this repository unless explicitly asked. install commands from this repository unless explicitly asked.
+150
View File
@@ -0,0 +1,150 @@
# Flake End-to-End Audit Report
**Date:** 2026-07-21
**Scope:** Full static lint/eval sweep + live build/deploy/interrogate/destroy testing of every `lxc-*` and `proxmox-*` flake target against `pve.sweet.home`, plus an audit of the operator's ability to manage the flake/secrets tooling.
**Branch:** `worktree-flake-e2e-audit` (this session's isolated worktree)
## Executive Summary
The flake itself is in good shape: `nixpkgs-fmt`, `statix`, and a full eval + dry-run build of every host and package are all clean. Every `lxc-*`/`proxmox-*` target's NixOS configuration builds successfully — no target has a broken derivation graph.
The issues found are **operational, not code-level**:
1. **pve.sweet.home is critically low on disk space** (91-95% full during this session) and cannot currently build the two largest closures (`gui`, `pxe-boot`) to completion — this actively blocks deploying/redeploying those hosts via the documented workflow.
2. **A real, reproducible secrets-decryption failure** was caught live: a stale cached container image (built before a same-day sops-key fix) boots with sshd never starting and every secret failing to decrypt. This is a **general hazard in `create-proxmox-resource.sh`'s "reuse the cached image if present" default**, not a one-off.
3. **sops key/anchor drift**: `proxmox-minimal` has a `.sops.yaml` recipient anchor with no corresponding private key anywhere in this environment; several `lxc-*`/`proxmox-*` targets have no sops registration at all yet.
4. One concrete script bug was found and **fixed in this session**: `create-proxmox-resource.sh` never enabled the QEMU guest agent channel on VMs it creates, despite the guest OS already running it.
5. A management-surface audit (of the operator's ability to run this repo day to day) found 5 process gaps, detailed below.
Nothing here required or received a `nixos-rebuild switch/boot/test`, `nixos-install`, or any disk-formatting command — all validation was `nix build`/`nix eval`, plus disposable `pct`/`qm` create-then-destroy cycles via the repo's own `create-proxmox-resource.sh`.
---
## 1. Static Analysis Results — all clean
`bash scripts/codex-maintenance.sh --full-check --dry-run` (whole-tree sweep, not just changed files):
| Check | Result |
|---|---|
| Secret grep | Clean — only the documented exceptions (installer's own hashed passwords, `access-tokens` comment references) |
| `nixpkgs-fmt --check` | 0/53 files would be reformatted |
| `statix` | No lint warnings |
| nix-cache host key drift check | Up to date |
| Full eval of every host's `system.build.toplevel` | All 19 `nixosConfigurations` targets evaluate cleanly |
| Dry-run build of every host + package | All succeed, no derivation errors |
No drift, no formatting issues, no lint findings anywhere in the tree.
---
## 2. Per-Target Test Results
Legend: **LIVE** = built on pve, `pct`/`qm` create → interrogated → destroyed. **BUILD-ONLY** = `nix build` validated the config (mostly `.config.system.build.toplevel`, occasionally `.tarball`), no resource created on pve.
| Target | Test type | Result | Notes |
|---|---|---|---|
| `lxc-docker` | BUILD-ONLY | ✅ PASS | Live redeploy skipped — CT105 is already running this identity in production; `--allow-duplicate-host` would have destroyed it. |
| `lxc-minimal` | **LIVE** | ✅ PASS (after retry) | First attempt reused a stale cached tarball predating a same-day sops-key commit → activation failed, sshd never started (see Finding #2). Redeployed with `--force-rebuild`: clean boot, `systemctl is-system-running` = `running`, secrets decrypted, sshd listening, users correct. |
| `lxc-nix-cache` | BUILD-ONLY | ✅ PASS (after retry) | Live redeploy skipped — CT101 is already running this identity. First local build attempt appeared to hang on a remote-builder handoff to nix-cache; killed and retried with `--builders ""` (local-only), succeeded. |
| `lxc-gui` | **LIVE (attempted)** | ⚠️ BLOCKED by pve disk space | Registered a fresh sops key (no prior registration existed), built successfully through the full NixOS system closure, then **failed packaging the tarball**: `No space left on device` on pve's root filesystem. Not a flake defect. |
| `lxc-pxe-boot` | **LIVE (attempted)** | ⚠️ BLOCKED by pve disk space | Same failure as `lxc-gui` — this target additionally builds a full nested installer/netboot image (`stage-installer-artifacts.nix`), making it similarly large. Failed with the same `No space left on device` error, immediately after the gui attempt had already consumed pve's remaining headroom. |
| `lxc-server` | BUILD-ONLY | ✅ PASS | No sops key registered yet; live deploy also would have hit `boot.zfs.extraPools` trying to import a real ZFS pool that doesn't exist in an isolated test container — an expected limitation of testing this build type outside its real hardware, not a bug. |
| `lxc-tailscale-exit-node` | BUILD-ONLY | ✅ PASS | No sops key registered yet. |
| `lxc-tor-relay` | BUILD-ONLY | ✅ PASS | Live redeploy skipped — CT106 already holds this identity in production. |
| `proxmox-docker` | BUILD-ONLY | ✅ PASS (after retry) | Live redeploy skipped — both CT105 *and* VM103 already hold `docker` identities. Combined `toplevel` + `diskoImagesScript` build crashed with a **Nix-internal assertion failure** (`worker.cc:360`) under this session's memory pressure (see Finding #6) — not a flake bug. Retried with `toplevel` alone: clean. |
| `proxmox-minimal` | **LIVE (attempted)** | ⚠️ BLOCKED by key drift → BUILD-ONLY | `.sops.yaml` has a registered `&proxmox-minimal` anchor but **no corresponding private key exists anywhere in this environment** — the script correctly refused to generate a mismatched replacement. Fell back to `toplevel` build: ✅ PASS. |
| `proxmox-nix-cache` | BUILD-ONLY | ✅ PASS | No sops key registered yet. |
| `proxmox-gui` | BUILD-ONLY | ⚠️ Killed after ~40min (resource-limited) | This session's local build machine has only 2GB RAM; swap filled completely (2.0/2.0GB) and the build stalled, so it was killed rather than risk destabilizing the session further. **Not a flake defect** — the equivalent `gui` NixOS configuration already proved fully buildable during the `lxc-gui` live attempt above (it built the entire system closure successfully and only failed at the pve-side tarball-packaging step due to disk space, not the config). |
| `proxmox-pxe-boot` | BUILD-ONLY | ⚠️ Killed after ~35min (resource-limited) | Was deep into building the nested installer's kernel initrd (this build type bundles a full netboot installer image via `stage-installer-artifacts.nix`) when killed to keep the audit moving. **Not a flake defect** — this target's own module logic was already effectively validated via the earlier *live* pve deploy attempt (`lxc-pxe-boot` above), which built the complete image and only failed at the final tarball-packaging step due to pve's disk space (Finding 1). |
| `proxmox-server` | BUILD-ONLY | ✅ PASS | No sops key registered yet; same ZFS-pool caveat as `lxc-server` would apply to a live deploy. |
| `proxmox-tailscale-exit-node` | BUILD-ONLY | ✅ PASS | No sops key registered yet. |
**Not tested at all:** `linode-*` targets (not deployable to Proxmox) and `installer` (not a normal host) — both were still covered by the static eval/dry-run-build sweep above.
---
## 3. Findings, Ranked by Severity
### Finding 1 — pve.sweet.home is critically low on disk space (blocks real deployments)
At session start: `/dev/mapper/pve-root` was **95% full, 5.3GB free** (of 94GB). After two failed large builds it recovered slightly to **91% full, 8.2GB free** (nix cleans up its own failed-build scratch space). `/nix/store` alone is 26GB; `nix-store --gc --print-dead` reports **zero** reclaimable garbage — everything currently in the store is a live GC root, so `nix-collect-garbage` won't help without first removing old roots.
**Why it matters:** `create-proxmox-resource.sh` builds every VM/CT image **directly on pve**, not on a build machine and transferred over. With <10GB headroom, any closure approaching a few GB (the `gui` build type: full Cinnamon desktop + Firefox + LibreOffice + GIMP + VS Code + xrdp; the `pxe-boot` build type: nginx/atftpd *plus* an entire nested installer/netboot image) cannot currently be built there at all. Both `lxc-gui` and `lxc-pxe-boot` failed live with `No space left on device` during this audit.
**Recommended action:** Expand `pve-root`'s LV, or free space by pruning old container templates in `/var/lib/vz/template/cache` (1.5GB) / old backups in `/var/lib/vz/dump` (306MB) / auditing what's pinning 26GB of `/nix/store` as live GC roots (likely `result-*` symlinks — see below). This is real production disk state; **not something this session touched or fixed** — it needs the operator's judgment on what's safe to remove.
**Secondary, smaller finding:** every `create-proxmox-resource.sh` run leaves a `result-<target>` symlink in the node's repo checkout as a permanent GC root (`ls /root/nixos/result-*` on pve showed 3 from this session alone: `lxc-docker`, `lxc-minimal`, `lxc-nix-cache`). These accumulate forever and pin their entire closures in the store. Consider having the script clean up its own `result-*` link after staging the built artifact (or use a temp `--out-link` under `/tmp`), so `nix-collect-garbage` can actually reclaim old build outputs.
### Finding 2 — Stale cached images can silently ship broken secrets (reproduced live)
`create-proxmox-resource.sh`'s default behavior is: if the node already has `<target>.tar.xz`/`.raw` staged, **reuse it** — only `--force-rebuild` forces a fresh build. This session hit exactly the failure mode `docs/auto-installer.md` already warns about: `lxc-minimal`'s cached tarball (built 2026-07-20T15:57Z) predated a same-day sops-key fix commit (2026-07-20T17:49Z, "clean up in ailse 3"). The deployed container booted with:
```
sops-install-secrets: failed to decrypt '.../common.yaml': Error getting data key: 0 successful groups required, got 0
Activation script snippet 'setupSecrets' failed (1)
```
— every secret permanently failed to decrypt, `sshd` never started (though the container otherwise looked "running"). This was **not a code bug**: the currently-committed `secrets/common.yaml` decrypts fine for that host's key when checked independently; the *cached artifact on pve* simply reflected an older commit's ciphertext. Redeploying with `--force-rebuild` fixed it immediately.
**Why it matters:** this is silent and easy to trigger by accident — any operator who redeploys a host without remembering `--force-rebuild` after a secrets change gets a container that looks like it started (`pct start` succeeds, `pct status` = running) but is completely inaccessible.
**Recommended action:** Have `create-proxmox-resource.sh` compare the cached image's build timestamp (or embed the source commit hash in the staged filename) against current HEAD, and warn (or refuse without `--force-rebuild`) if they differ — rather than silently trusting presence alone.
### Finding 3 — sops key/anchor drift
Two concrete instances hit live during this session:
- **`proxmox-minimal`**: `.sops.yaml` already has a registered `&proxmox-minimal` age recipient, but this environment's `host-keys/` directory has no corresponding private key file. `sync-host-keys.sh` correctly refused to generate a replacement (it would silently mismatch whatever's already registered/deployed) — but this means **no environment currently has this host's private key**, unless it exists on some other machine that was never backed up here.
- **`lxc-gui`**, and by the same logic `lxc-server`/`lxc-tailscale-exit-node`/most `proxmox-*` targets, have **no sops registration at all yet** — expected for undeployed hosts per `docs/auto-installer.md`, but this session's live-testing needed to register `lxc-gui`'s key on the fly, which immediately hit **Finding 3b**: registering a key locally does nothing for pve's build until it's pushed to `origin/main` (pve builds via `git pull`, not from this uncommitted worktree). This is exactly gap #4 the management-surface audit (below) already flagged in the abstract — this session hit it concretely.
**Recommended action:** for `proxmox-minimal`, decide whether to regenerate its key (destroying old-key decrypt access, if anything still holds it) or track down wherever the original private key lives and back it up here. For the general pattern, see the management-surface audit's recommendation to pre-flight-check key registration before building.
### Finding 4 — QEMU guest agent never wired up (found and fixed this session)
`modules/common/configuration.nix:44` sets `services.qemuGuest.enable = true` on every host — the guest-side agent daemon is correctly enabled everywhere. But `scripts/proxmox/create-proxmox-resource.sh`'s `qm create` call never passed `--agent 1`, so **Proxmox never created the virtio-serial channel** the agent needs. Every `proxmox-*` VM this script ever created was silently missing `qm guest exec`/IP-address reporting in the Proxmox UI, despite the guest daemon actually running.
**Status: fixed in this session's worktree** (`scripts/proxmox/create-proxmox-resource.sh`, `qm create` now includes `--agent enabled=1`) — see the diff, included in the PR from this session.
### Finding 5 — Orphaned container on pve (CT102)
`pve.sweet.home` has a stopped LXC container, **VMID 102**, with an essentially empty config (`lock: create` and nothing else — no hostname, no rootfs, no network) — the leftover of a `pct create` that started and never finished. It predates this session (not created by any of this audit's activity) and wasn't touched. **Recommend the operator confirm it's abandoned and remove it** (`pct destroy 102 --purge 1`) — left as-is it may be someone's genuine in-progress work, so it wasn't assumed safe to delete autonomously.
### Finding 6 — Nix-internal crash under memory pressure (tooling, not flake)
Building `proxmox-docker`'s `toplevel` and `diskoImagesScript` together crashed with a Nix-internal assertion failure (`Assertion '!awake.empty()' failed ... worker.cc:360`, a known class of bug in Nix's multi-goal build scheduler) while this session's 2GB-RAM build container was under heavy swap pressure (1.8-2.0/2GB swap in use) from a separate concurrent build. Retrying the same target alone (no concurrency) succeeded cleanly. **Not a flake defect** — purely an artifact of this session's constrained build environment; noted for completeness since it looked alarming in isolation.
### Finding 7 — Management-surface audit: 5 operability gaps
A focused audit of "can the operator actually run this repo day to day" (flake, home-manager, sops, related scripts) found:
1. **No documented recovery path if the `&admin` sops age key is lost without a backup.** `scripts/secrets/backup-admin-key.sh` exists and works but is referenced nowhere in `README.md`/`docs/` — no forcing function ensures a backup was ever taken. `rotate-admin-key.sh` requires the *old* key to re-key; there's no bootstrap-from-nothing path documented (the real fallback — deriving an age identity from any still-live host's own SSH key — isn't written down anywhere).
2. **home-manager has no standalone iteration path.** It's wired only inside `nixosConfigurations` (`flake.nix`) — no `homeConfigurations` output. The fastest real shortcut (`nix build .#nixosConfigurations.<target>.config.home-manager.users.nixos.home.activationPackage`) isn't documented anywhere, so the practical workflow is a full host rebuild to test one HM tweak.
3. **Gitea's flake-lock-update workflow pushes straight to `main` with no pre-merge validation.** `.gitea/workflows/update-flake-lock.yml` commits and pushes `nix flake update`'s result directly; `codex-maintenance.sh` only runs *after*, on the resulting push — a genuinely broken lockfile bump lands on `main` before anything catches it. (The GitHub-side workflow is safer — PR-based — but has the opposite gap: nothing alerts if the PR sits unmerged.)
4. **No pre-flight check that a build target has a registered sops key before building it.** `docs/auto-installer.md` documents the failure mode (silent, total secrets-decrypt failure) but nothing in `create-proxmox-resource.sh` refuses to proceed when it's about to build a target with no `.sops.yaml` anchor — it's on the operator to remember. This session's `lxc-gui` test hit close to this exact gap (needed the key added on the fly, mid-session).
5. **`vars.remoteBuilderAuthorizedKeys` has the same drift risk as `vars.nixCacheHostKey`, but no checker script.** `sync-nix-cache-host-key.sh --check` guards the latter; the former (and `vars.pxeServerIp`/`vars.pbsIp`) has no equivalent — a rotated/revoked client key just silently stops working with no diagnostic pointing back here.
---
## 4. Action Plan (priority order)
1. **Free up disk space on pve.sweet.home** (or expand `pve-root`). Blocking: `lxc-gui`, `proxmox-gui`, `lxc-pxe-boot`, `proxmox-pxe-boot` cannot currently be built/redeployed on this node at all.
2. **Decide on `proxmox-minimal`'s orphaned sops key**: locate the original private key and back it up here, or accept regenerating it (breaks decrypt access for whoever/whatever currently holds the old one).
3. **Merge this session's PR** (see below) to get the `--agent 1` fix and `lxc-gui`'s new sops registration onto `main` — required before `lxc-gui` can be live-redeployed with working secrets.
4. **Add a staleness guard to `create-proxmox-resource.sh`'s cache-reuse path** (Finding 2) — highest-leverage fix, since it silently produces a broken-but-"running" host.
5. **Add a pre-flight sops-anchor check to `create-proxmox-resource.sh`** (management-surface gap #4) — same root cause class as #4 above, catch it before building instead of at first boot.
6. Investigate/clean up **CT102** on pve (Finding 5) — confirm abandoned, then remove.
7. Document `backup-admin-key.sh` in `README.md`'s Security Notes and add the live-host-key bootstrap-recovery procedure to `docs/` (management-surface gap #1).
8. Add pre-push validation to the Gitea flake-lock-update workflow (management-surface gap #3).
9. Lower-priority: document the home-manager `activationPackage` shortcut (gap #2); extend `sync-nix-cache-host-key.sh`'s drift-check pattern to `remoteBuilderAuthorizedKeys` (gap #5).
10. Follow-up session: finish build-validating `proxmox-gui` and `proxmox-pxe-boot` (both killed here after 35-40min on this session's 2GB-RAM machine — not failures, just unfinished) once pve has headroom (item 1) — ideally from a machine with more RAM. `proxmox-server` and `proxmox-tailscale-exit-node` already passed build-only validation in this session, no follow-up needed.
---
## 5. Uncommitted Changes From This Session
This worktree (`worktree-flake-e2e-audit`) currently has:
- `scripts/proxmox/create-proxmox-resource.sh` — the `--agent enabled=1` fix (Finding 4).
- `.sops.yaml` / `secrets/common.yaml``lxc-gui`'s new age key registered as a recipient (generated live during this session's testing).
Per this session's standard workflow, these will be committed, pushed, and opened as a draft PR rather than pushed to `main` directly — merging it is the operator's call, and is also **prerequisite to live-redeploying `lxc-gui` successfully** (its build will keep hitting the sops-staleness failure from Finding 2 on pve until this registration is on `origin/main`).
+38 -178
View File
@@ -21,17 +21,15 @@ machines when deployed.
`modules/installer/common.nix` (the auto-installer's own root/nixos login — `modules/installer/common.nix` (the auto-installer's own root/nixos login —
a deliberate, documented choice, see `docs/auto-installer.md`, not a deliberate, documented choice, see `docs/auto-installer.md`, not
accidental tech debt) and **SSH public keys** in `variables.nix` accidental tech debt) and **SSH public keys** in `variables.nix`
(`vars.adminSshKey`, `vars.remoteBuilderAuthorizedKeys`, `vars.beszelHubKey`). Don't use the installer's hardcoded hash as a (`vars.adminSshKey`, `vars.remoteBuilderAuthorizedKeys`) plus a couple of
per-host `KEY` values for beszel-agent auth (`hosts/server/host.nix`,
`hosts/nix-cache/host.nix`). Don't use the installer's hardcoded hash as a
template for a *real* host — every other host uses sops-nix template for a *real* host — every other host uses sops-nix
(`hashedPasswordFile`, see "Security Notes" in `README.md`). Flag any *new* (`hashedPasswordFile`, see "Security Notes" in `README.md`). Flag any *new*
secret-like string you encounter instead of committing it. secret-like string you encounter instead of committing it.
- `host-keys/` is gitignored — used only by the auto-installer's own - `host-keys/` is gitignored — locally-generated *private* SSH host keys for
environment for pre-seeding non-LXC host keys before first boot (see the auto-installer (see `docs/auto-installer.md`). Never commit its
`docs/auto-installer.md`). Never commit its contents; if `git status` contents; if `git status` ever shows it as trackable, something is wrong.
ever shows it as trackable, something is wrong. All deployed hosts use
clan vars (`vars/per-machine/<target>/openssh/`, committed and
sops-encrypted) for their SSH host keys — those ARE tracked by git and
belong in the repo.
### Two Proxmox nodes: `pve1.sweet.home` (production) and `pve-test.sweet.home` (sandbox) ### Two Proxmox nodes: `pve1.sweet.home` (production) and `pve-test.sweet.home` (sandbox)
@@ -166,50 +164,21 @@ before committing.
Beyond `codex-setup.sh`/`codex-maintenance.sh` above, `scripts/` is Beyond `codex-setup.sh`/`codex-maintenance.sh` above, `scripts/` is
organized by purpose: `scripts/secrets/` (sops/age + SSH host-key organized by purpose: `scripts/secrets/` (sops/age + SSH host-key
management), `scripts/proxmox/` (Proxmox deployment), `scripts/installer/` management), `scripts/proxmox/` (Proxmox deployment), `scripts/lib/`
(the auto-installer's own shell script, templated into the image — see (shared helpers, sourced by the scripts below — not run directly), and a
below), `scripts/lib/` (shared helpers, sourced by the scripts below — not handful of repo-wide scripts left at the top level (`env.sh`,
run directly), and a handful of repo-wide scripts left at the top level `bump-nixpkgs-release.sh`, plus `codex-setup.sh`/`codex-maintenance.sh`
(`env.sh`, `bump-nixpkgs-release.sh`, plus `codex-setup.sh`/ above). When adding a new script, put it in the matching subfolder rather
`codex-maintenance.sh` above). When adding a new script, put it in the than the top level, and if it duplicates logic another script already has,
matching subfolder rather than the top level, and if it duplicates logic lift the shared part into `scripts/lib/` instead of copying it.
another script already has, lift the shared part into `scripts/lib/`
instead of copying it.
### `scripts/installer/`
- `scripts/installer/auto-install.sh` — the interactive install script
baked into the auto-installer image (see `docs/auto-installer.md`), kept
as a real, version-controlled shell file rather than inline in
`modules/installer/common.nix`'s Nix. It sources `scripts/env.sh` itself
for `LAN_DOMAIN` (`export LAN_DOMAIN`/`: "${LAN_DOMAIN:=...}"`, matching
`variables.nix`'s `lanDomain` — manually kept in sync, same pattern as
`NIX_CACHE_HOST` mirroring `nixCacheHost`), rather than Nix-level string
substitution — that's what makes it work identically whether run
straight from a git checkout or from inside the built installer image.
`common.nix` bakes `scripts/env.sh` in alongside it at a matching
relative path (`/etc/nixos-installer/env.sh` next to
`/etc/nixos-installer/installer/auto-install.sh`) so the script's own
`source "$(dirname ...)/../env.sh"` line resolves the same way in both
contexts — this is also why it's invoked from
`/etc/nixos-installer/installer/auto-install.sh` rather than a flat
`/etc/auto-install.sh`. `#!/usr/bin/env bash`, not
`#!/run/current-system/sw/bin/bash`: the latter only resolves on an
already-activated NixOS system, breaking the checked-out-file case
entirely (confirmed live: "cannot execute: required file not found" on
a non-NixOS box); `/usr/bin/env` is reliably present on both NixOS
(`environment.usrbinenv`'s own default) and any normal Linux distro.
### `scripts/secrets/` ### `scripts/secrets/`
- `scripts/secrets/sync-host-keys.sh` — generates/registers SSH host keys - `scripts/secrets/sync-host-keys.sh` — generates/registers SSH host keys
and their `.sops.yaml`/`secrets/*.yaml` recipients for flake targets, and their `.sops.yaml`/`secrets/*.yaml` recipients for flake targets,
idempotently (`--all`, `<target>`, `--remove`, `--regenerate-all-keys`, idempotently (`--all`, `<target>`, `--remove`, `--regenerate-all-keys`,
all with `--dry-run`). Stores keys as clan vars all with `--dry-run`). The primary tool for provisioning a new host's
(`vars/per-machine/<target>/openssh/`, committed and sops-encrypted) for secrets access — see "Creating a new machine" in `docs/auto-installer.md`.
all flake targets. The primary tool for provisioning a new host's
secrets access — see "Creating a new machine" in
`docs/auto-installer.md`.
- `scripts/secrets/prepare-host-key.sh` — narrower predecessor: generates a - `scripts/secrets/prepare-host-key.sh` — narrower predecessor: generates a
key by an arbitrary name without touching `.sops.yaml`. Still useful to key by an arbitrary name without touching `.sops.yaml`. Still useful to
pre-generate a key before its flake target exists yet, since pre-generate a key before its flake target exists yet, since
@@ -252,14 +221,6 @@ instead of copying it.
silent skip rather than a failure) only reports drift; the no-flags form silent skip rather than a failure) only reports drift; the no-flags form
updates both files in place. Declarative clients still need a rebuild to updates both files in place. Declarative clients still need a rebuild to
pick up the fix. pick up the fix.
- `scripts/secrets/push-host-keys.sh [--all | <target>] [--dry-run]
[--skip-git-check]` — pushes newly-generated SSH host keys from
`host-keys/` to already-running NixOS hosts, so they can decrypt sops
secrets after a rebuild following `sync-host-keys.sh
--regenerate-all-keys`. Verifies that `.sops.yaml` and `secrets/*.yaml`
are committed and pushed to the remote first (hosts rebuild from the
remote Gitea flake, so recipient changes must land there before any key
push).
### `scripts/proxmox/` ### `scripts/proxmox/`
@@ -281,16 +242,6 @@ instead of copying it.
failure just falls back to building from source / `cache.nixos.org`) so failure just falls back to building from source / `cache.nixos.org`) so
the node substitutes from and can offload builds to nix-cache on every the node substitutes from and can offload builds to nix-cache on every
subsequent run, not just this one. subsequent run, not just this one.
- `scripts/proxmox/clone-pve1-to-pve-test.sh <vmid> [--new-vmid <id>]
[--mode snapshot|suspend|stop] [--dry-run]` — ad-hoc clone of a single
VM or CT from pve1 (production) to pve-test (sandbox) via vzdump +
qmrestore/pct restore. Streams the archive directly between nodes (no
local staging copy). Always restores with `--unique 1` (fresh MAC
addresses) since the original is still running on the LAN. Cleans up
the vzdump archive from both nodes after a successful restore. The
script's own default is pve1 → pve-test, matching CLAUDE.md's policy
(unlike `create-proxmox-resource.sh`, which defaults to production for
the operator's own unqualified use).
- `scripts/proxmox/configure-nix-cache-client.sh [--dry-run] - `scripts/proxmox/configure-nix-cache-client.sh [--dry-run]
[--no-remote-builder] [--no-restart]` — the non-NixOS equivalent of [--no-remote-builder] [--no-restart]` — the non-NixOS equivalent of
`modules/nix-cache/client.nix`/`remote-builder-client.nix`, for a plain `modules/nix-cache/client.nix`/`remote-builder-client.nix`, for a plain
@@ -306,50 +257,6 @@ instead of copying it.
marked block rather than duplicating it); restarts `nix-daemon` by marked block rather than duplicating it); restarts `nix-daemon` by
default so the change takes effect immediately. default so the change takes effect immediately.
### `scripts/ha/`
HA cluster lifecycle and operational scripts. All mutate real cluster state
when run for real — always run against pve-test first unless the operator
explicitly targets pve1.
- `scripts/ha/deploy.sh [--skip-*] [--destroy] [--dry-run]` — full
lifecycle manager: phases through bridge creation, key sync, VM creation
(via `create-proxmox-resource.sh`), NIC/disk attachment, and cluster
initialisation. `--destroy` tears it back down. Safe to rerun
idempotently; each phase can be individually skipped.
- `scripts/ha/cluster-init.sh` — one-time cluster bootstrap run **as root
on ha-server-1** after both VMs are booted. Generates/distributes the
Corosync authkey, initialises DRBD metadata, creates XFS on `/dev/drbd0`,
configures LIO iSCSI, and registers all Pacemaker resources (DRBD → XFS
→ iSCSI → NFS → VIPs).
- `scripts/ha/health.sh` — read-only cluster health snapshot: SSH
reachability, quorum, DRBD state, Pacemaker resources, and VIP port
reachability. Safe to run from the workstation at any time.
- `scripts/ha/failover.sh [--to node1|node2] [--force] [--timeout <s>]
[--dry-run]` — graceful failover by putting the active node into
Pacemaker standby and waiting for resources to appear on the target.
- `scripts/ha/acceptance-tests.sh` — T1T7 acceptance tests (failover,
NFS/iSCSI connectivity, DRBD sync, etc.) that must all pass before the
cluster is considered production-ready.
- `scripts/ha/resize-data-disk.sh --size +NNg [--force] [--dry-run]` —
online data-disk resize: `qm resize` on both VMs, guest block-device
rescan, `drbdadm resize`, `xfs_growfs`. No downtime required.
- `scripts/ha/cluster-enable-stonith.sh` — enables the `fence_pve_ssh`
STONITH resource after the fence SSH key is deployed to both nodes and
authorised on the Proxmox host. Run once after `cluster-init.sh`.
- `scripts/ha/fence-pve-ssh.py` — Python STONITH fence agent for Pacemaker.
Deploy to `/etc/pacemaker/fence_pve_ssh` on both HA nodes (`chmod +x`).
SSHes to the Proxmox host and runs `qm stop/start <vmid>`.
### `scripts/ipa/`
- `scripts/ipa/create-nixos-ipa-host-account.sh [options] <hostname>` —
adds a NixOS host to the FreeIPA domain and produces a sops-encrypted
keytab at `secrets/<hostname>.keytab`, ready for `modules/ipa/client.nix`.
Replaces three error-prone manual steps: `ipa host-add`, `ipa-getkeytab`
(run on the DC, SCP'd back), and `sops encrypt` in the correct location
(must be at `secrets/<hostname>.keytab` for the creation rule to match).
### `scripts/lib/` ### `scripts/lib/`
Sourced by the scripts above, never run directly: Sourced by the scripts above, never run directly:
@@ -359,15 +266,6 @@ Sourced by the scripts above, never run directly:
`create-proxmox-resource.sh` runs over SSH. `create-proxmox-resource.sh` runs over SSH.
- `nix-eval.sh` — `NIX_EVAL_FLAGS` plus `list_flake_targets`/ - `nix-eval.sh` — `NIX_EVAL_FLAGS` plus `list_flake_targets`/
`flake_target_hostname` flake-introspection helpers. `flake_target_hostname` flake-introspection helpers.
- `nix-parallel.sh` — `run_nix_parallel`: fans out independent `nix eval`/
`nix build --dry-run` calls across up to `NIX_PARALLEL_JOBS` processes,
capped by available memory (~1 GB/job) rather than raw `nproc` to avoid
OOM on constrained CI runners. Used by `codex-maintenance.sh`.
- `clan-vars.sh` — helpers for reading/writing SSH host keys stored as clan
vars (`vars/per-machine/<target>/openssh/`, sops-encrypted) instead of
the gitignored `host-keys/` directory. Sourced by
`create-proxmox-resource.sh` and `sync-host-keys.sh`; depends on
`sops-age.sh` and `ssh-host-keys.sh` being sourced first.
- `ssh-host-keys.sh` — `generate_host_ed25519_key`/`ssh_pubkey_to_age`, - `ssh-host-keys.sh` — `generate_host_ed25519_key`/`ssh_pubkey_to_age`,
shared by `sync-host-keys.sh` and `prepare-host-key.sh`. shared by `sync-host-keys.sh` and `prepare-host-key.sh`.
- `sops-age.sh` — `age_pubkey_from_identity_file`/`sops_yaml_admin_pubkey`/ - `sops-age.sh` — `age_pubkey_from_identity_file`/`sops_yaml_admin_pubkey`/
@@ -384,20 +282,8 @@ Sourced by the scripts above, never run directly:
### Top level ### Top level
- `scripts/env.sh` — shared config (`PROXMOX_HOST`, storage pool, bridge, - `scripts/env.sh` — shared config (`PROXMOX_HOST`, storage pool, bridge,
default cores/memory, `NIX_CACHE_HOST`, `LAN_DOMAIN`) sourced by default cores/memory) sourced by `create-proxmox-resource.sh`. Add new
`create-proxmox-resource.sh` and `scripts/installer/auto-install.sh`. Add cross-script config here instead of duplicating it per-script.
new cross-script config here instead of duplicating it per-script.
- `scripts/recover-hosts.sh [<hostname> ...]` — fixes sops/SSH-key/GitHub-token
issues on deployed NixOS hosts and triggers a `Switch-nix` rebuild on each.
With no args discovers every known hostname; with args checks only those.
Fixes applied automatically (prompts before rebuilding): SSH host key drift
(restores the registered key) and stale GitHub access tokens (empties the
rendered `nix-github-token.conf` so Nix falls back to unauthenticated requests
until sops-nix re-renders the correct token after the next successful rebuild).
- `scripts/gc-hosts.sh [--dry-run]` — runs `nix-collect-garbage -d` on all live
NixOS hosts (workstation first, then pve1, then all Proxmox guests). Excludes
`nix-cache` (gc-ing the shared binary cache evicts store paths other hosts
depend on). Uses passwordless sudo where available; falls back to user-level gc.
- `scripts/bump-nixpkgs-release.sh` — bumps `flake.nix`'s `nixpkgs.url`/ - `scripts/bump-nixpkgs-release.sh` — bumps `flake.nix`'s `nixpkgs.url`/
`home-manager.url` in place. Exists because flake input URLs can't `home-manager.url` in place. Exists because flake input URLs can't
reference `variables.nix` (confirmed empirically — `nix flake metadata` reference `variables.nix` (confirmed empirically — `nix flake metadata`
@@ -435,15 +321,11 @@ nixosSystem {
} }
``` ```
Platforms: `linode`, `proxmox`, `lxc`, `baremetal`. Build types: `minimal`, Platforms: `linode`, `proxmox`, `lxc`. Build types: `minimal`, `nix-cache`,
`nix-cache`, `docker`, `gui`, `pxe-boot`, `tailscale-router`, `tor-relay`, `server`, `docker`, `gui`, `pxe-boot`, `tailscale-exit-node`, `tor-relay`. Not
`ha-server`. Not every combination is built — e.g. `pxe-boot` has no `linode` every combination is built — e.g. `pxe-boot` has no `linode` variant
variant (PXE/DHCP/TFTP need LAN L2 adjacency a Linode VPS doesn't have), (PXE/DHCP/TFTP need LAN L2 adjacency a Linode VPS doesn't have), and
`tor-relay` only exists as `lxc-tor-relay`, `ha-server` only exists as `tor-relay` currently only exists as `lxc-tor-relay`. Treat `flake.nix`'s
`proxmox-ha-server-{1,2}`, and `baremetal` only exists as `baremetal-gui`
(the real gui-host hardware — see `hosts/nixos/host.nix` and
`modules/platforms/baremetal.nix`). Treat
`flake.nix`'s
`generatedTargets` as the source `generatedTargets` as the source
of truth for which hosts exist — `README.md`, `AGENTS.md`, of truth for which hosts exist — `README.md`, `AGENTS.md`,
`docs/flake-lock-automation.md`, and the CI eval workflows `docs/flake-lock-automation.md`, and the CI eval workflows
@@ -456,25 +338,22 @@ removing a host.
- `hosts/<name>/host.nix` — per-machine identity **only**: hostname, hostId, - `hosts/<name>/host.nix` — per-machine identity **only**: hostname, hostId,
per-machine secrets, `system.stateVersion`. These files carry no `imports` per-machine secrets, `system.stateVersion`. These files carry no `imports`
of their own — all shared behavior comes from the platform/build-type modules of their own beyond narrow parameterized helpers (see
composed in `flake.nix`, not from the host file. `modules/beszel/host-token.nix` below) — all shared behavior comes from the
- `modules/platforms/{linode,proxmox,lxc,baremetal}.nix` — platform-specific platform/build-type modules composed in `flake.nix`, not from the host file.
config: boot method, guest tooling, and the hardware config, imported - `modules/platforms/{linode,proxmox,lxc}.nix` — platform-specific config:
directly by the platform module itself — **not** wired in from boot method, guest tooling, and (for linode/proxmox) the hypervisor-specific
`flake.nix`. VM platforms use `../hardware-configuration/vm/{proxmox,linode}.nix`; hardware config, imported directly by the platform module itself
`baremetal.nix` uses `../hardware-configuration/baremetal.nix` (adapted (`../hardware-configuration/vm/{proxmox,linode}.nix`) — **not** wired in
from a real `nixos-generate-config` run on the actual hardware, not a from `flake.nix`. `lxc.nix` has no hardware-configuration counterpart since
vm/ file, since it isn't a VM) plus `hardware.enableRedistributableFirmware containers share the host kernel; instead it imports nixpkgs' own
= true` for real wifi/GPU/microcode firmware that VMs never needed.
`lxc.nix` has no hardware-configuration counterpart since containers
share the host kernel; instead it imports nixpkgs' own
`virtualisation/proxmox-lxc.nix`, which gives every `lxc-*` host a `virtualisation/proxmox-lxc.nix`, which gives every `lxc-*` host a
`config.system.build.tarball` output — a plain rootfs tarball, used as a `config.system.build.tarball` output — a plain rootfs tarball, used as a
`pct create ... vztmpl` CT template (**not** `pct restore`, which expects `pct create ... vztmpl` CT template (**not** `pct restore`, which expects
`vzdump` backup-archive metadata this doesn't have), no install step — `vzdump` backup-archive metadata this doesn't have), no install step —
see `docs/auto-installer.md`. see `docs/auto-installer.md`.
- `modules/build-types/*.nix` — what a system is for: - `modules/build-types/*.nix` — what a system is for:
minimal/docker/gui/pxe-boot/nix-cache/tailscale-router/tor-relay/ha-server. minimal/server/docker/gui/pxe-boot/nix-cache/tailscale-exit-node/tor-relay.
- `modules/common/configuration.nix` — base NixOS config imported by every - `modules/common/configuration.nix` — base NixOS config imported by every
host: locale, users, nix settings, git. host: locale, users, nix settings, git.
- `modules/common/home.nix` / `hosts/nixos/home.nix` — Home Manager config for - `modules/common/home.nix` / `hosts/nixos/home.nix` — Home Manager config for
@@ -491,16 +370,6 @@ removing a host.
boots, so this declares them with `destroy = false` (disko never wipes boots, so this declares them with `destroy = false` (disko never wipes
them) and a bare `filesystem`/`swap` content type instead of a partition them) and a bare `filesystem`/`swap` content type instead of a partition
table — idempotent against an already-provisioned disk, never destructive. table — idempotent against an already-provisioned disk, never destructive.
- `modules/disko/baremetal.nix` — `baremetal-gui`'s disko config: a ZFS
RAID0 (striped, no redundancy — disko's zpool `mode` defaults to `""`,
which is a plain stripe rather than `"mirror"`/`"raidz"`) root pool
across two disks, ESP + systemd-boot on the first. Device paths
(`vars.guiRootDisk1`/`guiRootDisk2`) are placeholders — fill in stable
`/dev/disk/by-id/...` paths before running disko for real.
`modules/platforms/baremetal.nix` also imports
`modules/services/zfs/enable-service.nix` for this (the `zfs_unstable`
package, autoScrub/autoSnapshot/trim) — the only other importer today is
`ha-server`'s NFS data pool, an unrelated non-root ZFS use.
- `modules/boot/efi.nix` — systemd-boot + EFI vars, paired with the disko module. - `modules/boot/efi.nix` — systemd-boot + EFI vars, paired with the disko module.
- `modules/installer/` — the auto-installer environment (ISO, also served as - `modules/installer/` — the auto-installer environment (ISO, also served as
PXE netboot): `common.nix` (shared config + the generated PXE netboot): `common.nix` (shared config + the generated
@@ -514,21 +383,14 @@ removing a host.
substituter + SSH remote-builder wiring; see `docs/nix-cache.md` for the substituter + SSH remote-builder wiring; see `docs/nix-cache.md` for the
full design (per-host local stores, no shared `/nix/store`, and how the full design (per-host local stores, no shared `/nix/store`, and how the
`nixremote` signing/SSH keys fit together). `nixremote` signing/SSH keys fit together).
- `modules/ha/` — HA cluster NixOS modules: `cluster-config.nix` (DRBD, - `modules/beszel/host-token.nix` — parameterized helper module
Corosync, Pacemaker, firewall rules, cluster-wide NFS/iSCSI port (`{ name, sopsFile }`) that wires a host's beszel-agent sops secret/template
authorisation — shared by both ha-server nodes), `pacemaker-stack.nix` and `environmentFile`; used by `hosts/server/host.nix` and
(Pacemaker + Corosync service enablement), and supporting modules. See `hosts/nix-cache/host.nix` to avoid duplicating that boilerplate.
`docs/ha.md` for the cluster operational guide.
- `modules/ipa/client.nix` — FreeIPA client enrollment: sssd, Kerberos keytab,
and IPA host registration; imported by every real host via
`modules/common/configuration.nix`.
- `modules/beszel/enable-agent.nix` — enables beszel-agent, sets `HUB_URL`,
fixes the upstream `StateDirectory` bug, and wires the universal
`beszel-token` sops secret (from `secrets/common.yaml`) into the agent's
`environmentFile`; see `docs/beszel.md` for the full setup guide.
- `modules/tailscale/`, `modules/docker/`, `modules/networking/`, - `modules/tailscale/`, `modules/docker/`, `modules/networking/`,
`modules/traefik/`, `modules/tor/`, `modules/services/*` — single-purpose, `modules/traefik/`, `modules/tor/`, `modules/services/*` — single-purpose,
single-host feature modules (e.g. `docker/enable-service.nix`, single-host
feature modules (e.g. `docker/enable-service.nix`,
`services/zfs/enable-service.nix`). Grep `modules/build-types/*.nix` for `services/zfs/enable-service.nix`). Grep `modules/build-types/*.nix` for
each build type's `imports` list to see which modules apply where. each build type's `imports` list to see which modules apply where.
@@ -552,5 +414,3 @@ duplicating config.
- `docs/flake-lock-automation.md` — how `flake.lock` updates flow through CI - `docs/flake-lock-automation.md` — how `flake.lock` updates flow through CI
(scheduled `nix flake update` PR + host-eval-on-PR workflow) and why hosts (scheduled `nix flake update` PR + host-eval-on-PR workflow) and why hosts
should track the committed lock file rather than `nixos-rebuild --upgrade-all`. should track the committed lock file rather than `nixos-rebuild --upgrade-all`.
- `docs/ha.md` — HA file-server cluster: DRBD + XFS + LIO iSCSI + NFS managed
by Corosync + Pacemaker; network topology; lifecycle scripts in `scripts/ha/`.
+16 -20
View File
@@ -8,15 +8,13 @@ workstation.
Targets are named `<platform>-<buildtype>`, generated from two orthogonal Targets are named `<platform>-<buildtype>`, generated from two orthogonal
pieces composed in `flake.nix`: pieces composed in `flake.nix`:
- **Platforms** (what it runs on): `linode`, `proxmox`, `lxc`, `baremetal` - **Platforms** (what it runs on): `linode`, `proxmox`, `lxc`
- **Build types** (what it's for): `minimal`, `nix-cache`, `docker`, `gui`, - **Build types** (what it's for): `minimal`, `nix-cache`, `server`, `docker`,
`pxe-boot`, `tailscale-router`, `tor-relay`, `ha-server` `gui`, `pxe-boot`, `tailscale-exit-node`, `tor-relay`
Not every combination exists — `pxe-boot` has no `linode` variant, since Not every combination exists — `pxe-boot` has no `linode` variant, since
PXE/DHCP/TFTP need LAN L2 adjacency that a Linode VPS doesn't have, PXE/DHCP/TFTP need LAN L2 adjacency that a Linode VPS doesn't have, and
`tor-relay` and `ha-server` currently only exist on `lxc`/`proxmox`, and `tor-relay` currently only exists as `lxc-tor-relay`. The full list:
`baremetal` currently only exists as `baremetal-gui` (the real gui-host
hardware). The full list:
| Target | Purpose | | Target | Purpose |
| --- | --- | | --- | --- |
@@ -24,13 +22,12 @@ hardware). The full list:
| `proxmox-minimal` | Minimal NixOS host profile on Proxmox — previously the flat `nix-minimal` target | | `proxmox-minimal` | Minimal NixOS host profile on Proxmox — previously the flat `nix-minimal` target |
| `lxc-minimal` | Minimal NixOS host profile in a Proxmox LXC container | | `lxc-minimal` | Minimal NixOS host profile in a Proxmox LXC container |
| `linode-nix-cache` / `proxmox-nix-cache` / `lxc-nix-cache` | Local Nix binary cache and remote builder — previously the flat `nix-cache` target | | `linode-nix-cache` / `proxmox-nix-cache` / `lxc-nix-cache` | Local Nix binary cache and remote builder — previously the flat `nix-cache` target |
| `linode-server` / `proxmox-server` / `lxc-server` | Storage, NFS, backup, and monitoring exporter host — previously the flat `server` target |
| `linode-docker` / `proxmox-docker` / `lxc-docker` | Docker host for the main container stack — previously the flat `docker` target | | `linode-docker` / `proxmox-docker` / `lxc-docker` | Docker host for the main container stack — previously the flat `docker` target |
| `linode-gui` / `proxmox-gui` / `lxc-gui` | Cinnamon desktop workstation — previously the flat `nixos` target | | `linode-gui` / `proxmox-gui` / `lxc-gui` | Cinnamon desktop workstation — previously the flat `nixos` target |
| `baremetal-gui` | Same Cinnamon desktop workstation, on the real gui-host hardware — ZFS RAID0 root, systemd-boot |
| `proxmox-pxe-boot` / `lxc-pxe-boot` | HTTP/iPXE boot asset host — previously the flat `pxe-boot` target | | `proxmox-pxe-boot` / `lxc-pxe-boot` | HTTP/iPXE boot asset host — previously the flat `pxe-boot` target |
| `linode-tailscale-router` / `proxmox-tailscale-router` / `lxc-tailscale-router` | Tailscale subnet router + MagicDNS forwarder for the LAN | | `linode-tailscale-exit-node` / `proxmox-tailscale-exit-node` / `lxc-tailscale-exit-node` | Tailscale exit node |
| `lxc-tor-relay` | Tor middle relay | | `lxc-tor-relay` | Tor middle relay |
| `proxmox-ha-server-1` / `proxmox-ha-server-2` | HA file-server cluster nodes — DRBD + XFS + iSCSI + NFS, managed by Corosync + Pacemaker |
Which variant of a given buildtype is actually deployed isn't tracked Which variant of a given buildtype is actually deployed isn't tracked
anywhere in this repo — that's live infrastructure state, not something a anywhere in this repo — that's live infrastructure state, not something a
@@ -47,7 +44,8 @@ section for which is which.
Each buildtype's `hosts/<name>/host.nix` carries the per-machine identity Each buildtype's `hosts/<name>/host.nix` carries the per-machine identity
(hostname, hostId, per-machine secrets, `system.stateVersion`) that must stay (hostname, hostId, per-machine secrets, `system.stateVersion`) that must stay
fixed regardless of which platform it's built for. Every deployed host fixed regardless of which platform it's built for — see
`flake-target-refactor-spec.md` for the full rationale. Every deployed host
stamps its own active target name into `/etc/flake-target` at build time, so stamps its own active target name into `/etc/flake-target` at build time, so
`nixos-rebuild switch --flake .#$(cat /etc/flake-target)` always picks up the `nixos-rebuild switch --flake .#$(cat /etc/flake-target)` always picks up the
right one even after a platform migration changes the flake attribute name. right one even after a platform migration changes the flake attribute name.
@@ -66,13 +64,12 @@ nix eval --json .#nixosConfigurations --apply builtins.attrNames | jq -r '.[]'
| `variables.nix` | Single source of truth for shared values (LAN domain/CIDR, hostnames, timezone, primary username, storage root, NFS share subpaths/mountpoints, service ports, ...) — passed to every module and Home Manager config as the `vars` argument via `specialArgs`/`extraSpecialArgs` | | `variables.nix` | Single source of truth for shared values (LAN domain/CIDR, hostnames, timezone, primary username, storage root, NFS share subpaths/mountpoints, service ports, ...) — passed to every module and Home Manager config as the `vars` argument via `specialArgs`/`extraSpecialArgs` |
| `hosts/<name>/host.nix` | Per-machine identity: hostname, hostId, per-machine secrets, `system.stateVersion` | | `hosts/<name>/host.nix` | Per-machine identity: hostname, hostId, per-machine secrets, `system.stateVersion` |
| `hosts/nixos/home.nix` | Workstation-specific Home Manager config (used by the `gui` build type) | | `hosts/nixos/home.nix` | Workstation-specific Home Manager config (used by the `gui` build type) |
| `modules/platforms/` | Platform-specific config: virtualisation guest tools, boot method, hardware config (`linode.nix`, `proxmox.nix`, `lxc.nix`, `baremetal.nix`) | | `modules/platforms/` | Platform-specific config: virtualisation guest tools, boot method, hardware config (`linode.nix`, `proxmox.nix`, `lxc.nix`) |
| `modules/build-types/` | Build-type-specific config: what makes a system minimal/server/docker/gui/pxe-boot/nix-cache | | `modules/build-types/` | Build-type-specific config: what makes a system minimal/server/docker/gui/pxe-boot/nix-cache |
| `modules/common/` | Shared NixOS config, Home Manager, aliases imported by every host | | `modules/common/` | Shared NixOS config, Home Manager, aliases imported by every host |
| `modules/nix-cache/` | Binary cache and remote builder client/server modules | | `modules/nix-cache/` | Binary cache and remote builder client/server modules |
| `modules/installer/` | Auto-installer environment (ISO, also served as PXE netboot) — see `docs/auto-installer.md` | | `modules/installer/` | Auto-installer environment (ISO, also served as PXE netboot) — see `docs/auto-installer.md` |
| `host-keys/` | Gitignored; only used by the auto-installer environment for pre-seeding SSH host keys before first boot — see `docs/auto-installer.md`. All deployed hosts use clan vars (`vars/per-machine/<target>/openssh/`) instead | | `host-keys/` | Gitignored, locally-generated SSH host keys for the auto-installer — see `docs/auto-installer.md` |
| `vars/per-machine/` | Clan vars: committed, sops-encrypted SSH host keys for all deployed hosts; read by `create-proxmox-resource.sh` at deploy time |
| `docs/` | Operational notes for cache, builders, lock updates, boot services, the auto-installer, and Proxmox image builds | | `docs/` | Operational notes for cache, builders, lock updates, boot services, the auto-installer, and Proxmox image builds |
| `scripts/` | Codex setup, validation, host-key, release-bump, and Proxmox resource helpers | | `scripts/` | Codex setup, validation, host-key, release-bump, and Proxmox resource helpers |
@@ -162,10 +159,9 @@ sops-nix-everywhere: it has a hardcoded login password instead (no stable
per-boot host key for sops-nix to derive from on ephemeral media) — see per-boot host key for sops-nix to derive from on ephemeral media) — see
"Host keys" in `docs/auto-installer.md` for why, and how the private keys it "Host keys" in `docs/auto-installer.md` for why, and how the private keys it
*does* pre-seed for target hosts stay out of git via the gitignored *does* pre-seed for target hosts stay out of git via the gitignored
`host-keys/` directory. All deployed hosts use clan vars `host-keys/` directory.
(`vars/per-machine/<target>/openssh/`, committed and sops-encrypted) for
their SSH host keys.
This repository's git *history* still contains secrets committed before the This repository's git *history* still contains secrets committed before this
sops-nix migration — those are being scrubbed and rotated separately; don't migration (see `remove-sensetive-info-refactor.md`) — those are being
treat the repo as safe to make public until that's finished. scrubbed and rotated separately; don't treat the repo as safe to make public
until that's finished.
-25
View File
@@ -1,25 +0,0 @@
-----BEGIN CERTIFICATE-----
MIIESDCCArCgAwIBAgIBATANBgkqhkiG9w0BAQsFADA1MRMwEQYDVQQKDApTV0VF
VC5IT01FMR4wHAYDVQQDDBVDZXJ0aWZpY2F0ZSBBdXRob3JpdHkwHhcNMjYwNzI2
MjExMzQxWhcNNDYwNzI2MjExMzQxWjA1MRMwEQYDVQQKDApTV0VFVC5IT01FMR4w
HAYDVQQDDBVDZXJ0aWZpY2F0ZSBBdXRob3JpdHkwggGiMA0GCSqGSIb3DQEBAQUA
A4IBjwAwggGKAoIBgQCzljYktbHdMGVJ6Wq0XQJuHLN6dkCSOgtoIzQtriPQkkNI
uo28LwobaiQQ8sX4kGRH/BTKnH8QlId/jug4Uc+sDHnABYu++AiOhPbBX8gCpRQ0
hebBjZiktHSBUEJR31siWOVdBoKBDJEoxehx7XUXvcxIJcaRN+LHYjO86nJN55HB
VwFU2JcYDk98c+144dFJxXdr++MjWe4Z/oVVU8JHIOtNtKhVhvij6oOSWxcYoJO/
S80LRj1vx/o6o/3G6bYug7PjY7JjZk/Oj61whijZkcsoO1MXSYI6UywJZGflv+ZB
7HyufdYAsK3WhE8O2FX3/kq64Ol83HNtoR8Dt68rTg1xpW6K45jS6iDPKueYGkb0
oSx7e++90VAW2PDhj6QQ3JJ4O5VQwrrecekJzUrAean0FOEbmgyi4PsEp1Vk6LDQ
SsIn1x0euyxVivQMlzNX2XrZL3urn1BNPqAdntXQMkR0Wl8sbUiJPe0kxG52CGXs
6yfNEXbPmVGcC0TBdGECAwEAAaNjMGEwHQYDVR0OBBYEFLh5QbI1UWMH0WR4z8bG
lhrOX3X5MB8GA1UdIwQYMBaAFLh5QbI1UWMH0WR4z8bGlhrOX3X5MA8GA1UdEwEB
/wQFMAMBAf8wDgYDVR0PAQH/BAQDAgHGMA0GCSqGSIb3DQEBCwUAA4IBgQCVodVN
owwo53OQe02QhtEbIur2PL7zIfvhvCTRD4J8gwpbMIqT7JQK0tV6Mvsg2L8yTb2O
KjrWeLKHGWaZZlhGSPTbkMFdb/Ls8M9FSnkc2bwcdWW3Z1lOiCjBYYqwLCG6JhvB
5SXVwWNJwXeasL2m7oFTSwhsqPpARJ2t25u2N35o+tqIoCjijKwkmEOT66N9EAbu
2VQjtYZWPkBtP4YCe0Ey6u4oy7sy8ThNAjOylZok+J4JW7QEFjK4Q/emhA4aQq5H
gg9qgMuG+5oi6D1g2Wy+fMTRBaukJtLYZbBpQMQhMYWg44uPp/2bbNPTID/nV1KB
GcPyHaskcVxPdYWxAPMwk3AeJXWyOq7atAPTF5sbk0kQQf2m+vyOqcli5CxRMUgV
rcyi9l6+dZW4U+38Q0ET5M3OuxNI4hA7kVY2cfTakXWNqh97+TIHnstblDhAxECK
6ZLMJQYUy7LqJTX84H27CBWLexEMjXwdr5HCV88Fj6mAK0fRufnIw5FeneA=
-----END CERTIFICATE-----
+9 -36
View File
@@ -8,13 +8,10 @@ lives here.
The installer provides a small NixOS install environment (ISO, or the same The installer provides a small NixOS install environment (ISO, or the same
image netbooted via PXE) with SSH access, Git support, and an interactive image netbooted via PXE) with SSH access, Git support, and an interactive
installation script. installation script.
Logging in as any user (root or `nixos`) runs Logging in as any user (root or `nixos`) runs `/etc/auto-install.sh`,
`/etc/nixos-installer/installer/auto-install.sh` (the same file as discovers available hosts from this same flake, lets the operator choose a
`scripts/installer/auto-install.sh` in this repo — see "Installer process" target, applies that host's Disko storage configuration, installs NixOS, and
below for why it's baked in at that path rather than a flat reboots.
`/etc/auto-install.sh`), discovers available hosts from this same flake,
lets the operator choose a target, applies that host's Disko storage
configuration, installs NixOS, and reboots.
**This applies to every `nixosConfigurations` target except `lxc-*` hosts — **This applies to every `nixosConfigurations` target except `lxc-*` hosts —
see "LXC hosts" immediately below for why those are different.** see "LXC hosts" immediately below for why those are different.**
@@ -22,7 +19,7 @@ see "LXC hosts" immediately below for why those are different.**
## LXC hosts ## LXC hosts
`lxc-*` targets (`lxc-minimal`, `lxc-nix-cache`, `lxc-server`, `lxc-docker`, `lxc-*` targets (`lxc-minimal`, `lxc-nix-cache`, `lxc-server`, `lxc-docker`,
`lxc-gui`, `lxc-pxe-boot`, `lxc-tailscale-router`, `lxc-tor-relay`) are **not** installed via `auto-install.sh` — the `lxc-gui`, `lxc-pxe-boot`, `lxc-tailscale-exit-node`, `lxc-tor-relay`) are **not** installed via `auto-install.sh` — the
interactive menu deliberately excludes them. Don't try to select one there; interactive menu deliberately excludes them. Don't try to select one there;
`nixos-install` would bind-mount `/` onto `/mnt` (LXC containers have no raw `nixos-install` would bind-mount `/` onto `/mnt` (LXC containers have no raw
disk to partition) and then refuse to touch the filesystem it's currently disk to partition) and then refuse to touch the filesystem it's currently
@@ -133,15 +130,13 @@ Flake outputs:
```nix ```nix
nixosConfigurations.installer # ISO/netboot installer image nixosConfigurations.installer # ISO/netboot installer image
packages.x86_64-linux.iso # installer ISO/netboot image packages.x86_64-linux.iso # installer ISO/netboot image
packages.x86_64-linux.pxe # auto-installer netboot bundle (kernel + initrd + ipxe script) packages.x86_64-linux.pxe # netboot-ipxe + netboot-initrd + netboot-kernel, bundled
packages.x86_64-linux.pxe-minimal # vanilla NixOS minimal netboot bundle (no installer wiring)
``` ```
```sh ```sh
nix build .#iso nix build .#iso
nix build .#pxe nix build .#pxe
nix build .#pxe-minimal
``` ```
There's no `nixosConfigurations.proxmox-lxc` (installer-boots-as-an-LXC- There's no `nixosConfigurations.proxmox-lxc` (installer-boots-as-an-LXC-
@@ -155,11 +150,7 @@ use case.
The `pxe` variant is also built automatically as part of the `pxe-boot` host The `pxe` variant is also built automatically as part of the `pxe-boot` host
itself (`modules/pxe-boot/stage-installer-artifacts.nix`) and served over itself (`modules/pxe-boot/stage-installer-artifacts.nix`) and served over
iPXE as the menu's "NixOS Auto-Installer" entry — see `docs/pxe-boot.md`. iPXE — see `docs/pxe-boot.md`.
That same host also builds and serves `packages.x86_64-linux.pxe-minimal`,
a vanilla NixOS minimal netboot image with none of this auto-installer's
wiring, as a separate "NixOS Minimal" menu entry — also documented in
`docs/pxe-boot.md`, not covered further here since it's not this installer.
## Host keys ## Host keys
@@ -197,10 +188,6 @@ default.
`auto-install.sh` still supports the older manual path as a fallback: if a `auto-install.sh` still supports the older manual path as a fallback: if a
host's key isn't baked in (`/etc/host-keys`), it checks `/root/host-keys` host's key isn't baked in (`/etc/host-keys`), it checks `/root/host-keys`
next, where you can `scp` a key in after boot, same as before this migration. next, where you can `scp` a key in after boot, same as before this migration.
If neither has it and the script is running interactively (an actual
operator at the other end of stdin, not an unattended run), it prompts for
an arbitrary directory to check (a mounted USB stick, another filesystem,
etc.) and copies the key pair into `/root/host-keys` from there if found.
## Storage ## Storage
@@ -226,21 +213,7 @@ entirely (see "LXC hosts" above), so it never reaches this code path.
## Installer process ## Installer process
`scripts/installer/auto-install.sh` is a real, version-controlled shell `/etc/auto-install.sh`:
script — not an inline Nix string. It sources `scripts/env.sh` for
`LAN_DOMAIN` itself (same as every other script in `scripts/`), so it
behaves identically whether it's run straight from a git checkout (e.g.
manually, from a stock NixOS ISO that isn't this repo's own installer
image) or from inside the built installer image. That's also why it's
baked in at `/etc/nixos-installer/installer/auto-install.sh` rather than a
flat `/etc/auto-install.sh``modules/installer/common.nix` bakes
`scripts/env.sh` in alongside it at `/etc/nixos-installer/env.sh`,
preserving the same relative layout (`installer/auto-install.sh` ->
`../env.sh`) the checked-out repo has, so the script's own
`source ".../env.sh"` line resolves correctly in both places without any
Nix-level templating.
Once running, it:
1. Queries `nixosConfigurations` from this flake over the network (`git+https://<lanDomain>/beatzaplenty/nixos.git`) — this happens at *install* time, not build time, so a generic installer image always sees whatever hosts are currently committed, without needing a rebuild. 1. Queries `nixosConfigurations` from this flake over the network (`git+https://<lanDomain>/beatzaplenty/nixos.git`) — this happens at *install* time, not build time, so a generic installer image always sees whatever hosts are currently committed, without needing a rebuild.
2. Presents them as a menu; confirms the choice. 2. Presents them as a menu; confirms the choice.
-104
View File
@@ -1,104 +0,0 @@
# Beszel agent
[Beszel](https://github.com/henrygd/beszel) is the monitoring dashboard used
in this LAN. The hub runs as a Docker container on `docker.sweet.home` (port
`vars.ports.beszelHub`, 8090). Each monitored NixOS host runs a
`beszel-agent` that connects back to the hub.
---
## How it works
Everything is handled by a single module:
**`modules/beszel/enable-agent.nix`** — imported by a build type. It:
- Enables `beszel-agent`
- Sets `HUB_URL` to `docker.sweet.home:8090`
- Sets `KEY` from `vars.beszelHubKey` (`variables.nix`) — the hub's SSH
public key, shared by every agent. Update `beszelHubKey` if the docker
host is ever rebuilt and the hub generates a new keypair.
- Reads the universal `beszel-token` from `secrets/common.yaml` via sops
and passes it to the agent as `TOKEN` in an env file
- Fixes an upstream bug where the agent couldn't persist its hub-pairing
fingerprint across restarts (adds a real `StateDirectory`)
A host file needs no beszel configuration at all — just import the module
in the build type and add the system in the hub UI.
---
## Adding beszel to a new build type
Add `../beszel/enable-agent.nix` to the `imports` list in
`modules/build-types/<type>.nix`:
```nix
imports = [
../beszel/enable-agent.nix
# ... other imports
];
```
That's the only change required. The host file needs nothing.
---
## Adding a new system to the hub
1. Rebuild and deploy the host with its build type importing `enable-agent.nix`.
2. Open the beszel hub (`http://docker.sweet.home:8090`).
3. Go to **Systems → Add system**, enter the host's IP and the default port
(45876). The agent will connect and the system will appear as active.
---
## One-time setup: add the token to `secrets/common.yaml`
The universal token is stored once in the common secrets file, shared by all
agents. Only needed once, not per-host:
```sh
sops secrets/common.yaml
```
Add:
```yaml
beszel-token: <token from the beszel hub Settings → Keys>
```
`secrets/common.yaml` is already a sops recipient for every host via their
SSH host keys, so no additional sops recipient setup is needed.
---
## Optional: monitoring extra filesystems
To report disk usage for a mount beyond the root filesystem, add
`EXTRA_FILESYSTEMS` in the host file:
```nix
services.beszel.agent.environment = {
EXTRA_FILESYSTEMS = "/mnt/data"; # colon-separated for multiple paths
};
```
---
## Optional: monitoring Docker containers
`enable-agent.nix` has a commented-out line for Docker monitoring:
```nix
#DOCKER_HOST = "tcp://docker-socket-proxy:2375";
```
Uncomment it if the host runs docker-socket-proxy and you want per-container
stats. Hosts without Docker should leave it commented out.
---
## If the hub key changes
If the docker host is ever rebuilt and beszel generates a new SSH keypair,
update `beszelHubKey` in `variables.nix` and rebuild all beszel-enabled hosts.
The new key is visible in the beszel hub under **Settings → Keys**.
-160
View File
@@ -1,160 +0,0 @@
# HA File-Server Cluster
Two `proxmox-ha-server-{1,2}` VMs form an active/passive file-server cluster:
DRBD replicates a block device between nodes; Corosync + Pacemaker manage
failover; XFS, LIO iSCSI, and NFS are brought up as a collocated resource
group on whichever node holds the DRBD Primary role.
NixOS modules: `modules/ha/`. Lifecycle scripts: `scripts/ha/`.
Cluster-wide constants: `variables.nix` (`haServer*` vars).
---
## Network layout
Three subnets — all internal to pve1 (`vmbr0`/`vmbr1`/`vmbr2`):
| Subnet | VLAN | CIDR | Bridge | Purpose |
|---|---|---|---|---|
| LAN | 2 | `192.168.2.0/24` | `vmbr0` | Management, LAN NFS |
| Cluster | 10 | `192.168.10.224/29` | `vmbr1` | Corosync ring0 + DRBD replication |
| Storage-client | 20 | `192.168.20.0/24` | `vmbr2` | NFS + iSCSI for docker/swarm |
Each HA VM has three NICs: `ens18` (LAN/vmbr0), `ens19` (cluster/vmbr1),
`ens20` (storage-client/vmbr2). See `docs/ip-addressing.md` for all IPs.
Corosync ring0 uses the cluster NIC; ring1 (backup heartbeat) uses the LAN
NIC. DRBD replicates over the cluster NIC. No storage traffic crosses the LAN.
---
## Pacemaker resources
All resources run collocated on whichever node is Primary, in this order:
```
ms-drbd0 (promotable DRBD clone)
→ xfs-data (XFS mount on /dev/drbd0 → /srv/ha-data)
→ iscsi-target (targetctl)
→ nfs-server (nfs-server.service)
→ vip-lan (192.168.2.229/24 on vmbr0 — NFS for LAN clients)
→ vip-storage (192.168.20.229/24 on vmbr2 — NFS + iSCSI for VLAN 20)
```
`vip-lan` serves pxe-boot and other LAN-only NFS clients.
`vip-storage` serves docker and any future swarm nodes; iSCSI is available on
VLAN 20 but NFS is preferred for multi-host volume sharing.
---
## DRBD fencing
`fencing resource-only` with `crm-fence-peer.sh`/`crm-unfence-peer.sh`
wrappers (`modules/ha/cluster-config.nix`). The DRBD kernel module invokes
these via the User Mode Helper with a minimal PATH; the wrappers prepend
`/run/current-system/sw/bin` before exec-ing the real handlers so Pacemaker
tools (`cibadmin`, `crm_mon`, etc.) are found.
STONITH is initially disabled (`stonith-enabled: false`,
`no-quorum-policy: ignore`). Enable it once the `fence_pve_ssh` fence agent
(`scripts/ha/fence-pve-ssh.py`) is deployed and authorised:
```bash
scripts/ha/cluster-enable-stonith.sh # run as root on ha-server-1
```
---
## Deploying the cluster from scratch
Use `scripts/ha/deploy.sh` — it orchestrates all phases:
```bash
# Against pve-test (safe — Claude's default target):
scripts/ha/deploy.sh --node "$PVE_TEST_HOST" [--dry-run]
# Against pve1 (production — requires explicit operator go-ahead):
scripts/ha/deploy.sh --node "$PVE1_HOST"
```
Phases (each skippable with `--skip-<phase>`):
1. `ensure-bridge` — creates `vmbr1`/`vmbr2` on the Proxmox node if absent
2. `sync-keys` — generates SSH host keys for both nodes; registers sops recipients
3. `create-vms` — builds disk images, creates VMs via `create-proxmox-resource.sh`
4. `add-hardware` — attaches storage NIC and DRBD data disk to each VM
5. `init-cluster` — runs `scripts/ha/cluster-init.sh` on ha-server-1
`--destroy` runs the teardown sequence.
---
## Day-to-day operations
```bash
# Read-only health check (safe from workstation):
scripts/ha/health.sh
# Graceful failover (prompts for confirmation):
scripts/ha/failover.sh [--to node1|node2]
# Online data-disk growth (no downtime):
scripts/ha/resize-data-disk.sh --size +20G
# Acceptance tests (run after any significant change):
scripts/ha/acceptance-tests.sh
```
---
## Adding FreeIPA host accounts
IPA host registration is automated:
```bash
scripts/ipa/create-nixos-ipa-host-account.sh <hostname>
```
This runs `ipa host-add`, fetches a keytab from the domain controller, and
writes a sops-encrypted `secrets/<hostname>.keytab` in one step. The module
`modules/ipa/client.nix` (imported by every host via
`modules/common/configuration.nix`) consumes the keytab via sops-nix.
---
## Storage layout
```
/srv/ha-data/
docker/
config/ NFS → docker:/mnt/docker/config
databases/ NFS → docker:/mnt/docker/databases
volumes/ NFS → docker:/mnt/docker/volumes
nextcloud-data/ NFS → docker:/mnt/docker/nextcloud-data
proxmox/
iso/ NFS → pve1 ISO storage
lxc/ NFS → pve1 CT template storage
pxe-boot/
images/ NFS → pxe-boot:/srv/pxe/http/images (PXE assets)
raspi/
volumes/ NFS → raspi NFS mounts
iscsi-lun.img iSCSI fileio backstore (VLAN 20 only, not in active use)
```
All shares are defined in `variables.nix` (`vars.nfsShares.*`). The NFS
export list lives in `modules/ha/nfs-exports.nix`.
---
## Key variables
| Variable | Description |
|---|---|
| `vars.haServer1Ip` / `vars.haServer2Ip` | LAN management IPs |
| `vars.haServer1StorageIp` / `vars.haServer2StorageIp` | Cluster NIC IPs (DRBD/Corosync ring0) |
| `vars.haServerLanVip` | Pacemaker `vip-lan` — NFS for LAN (192.168.2.229) |
| `vars.haServerVip` | Pacemaker `vip-storage` — NFS + iSCSI for VLAN 20 (192.168.20.229) |
| `vars.haLanNfsFqdn` | FQDN of `vip-lan`: `ha-vip-lan.sweet.home` |
| `vars.haStorageRoot` | XFS mount point: `/srv/ha-data` |
| `vars.haServerDrbdDisk` | Block device for DRBD backing store |
| `vars.haStorageCidr` | Cluster subnet CIDR (`192.168.10.224/29`) |
| `vars.haClientCidr` | Storage-client subnet CIDR (`192.168.20.0/24`) |
-330
View File
@@ -1,330 +0,0 @@
# Docker Swarm Cutover Plan
Migration guide for moving containerised services from the existing single-host
Docker LXC container (CT 105, `docker.sweet.home`, 192.168.2.225) to the new
Docker Swarm cluster (`ha-docker-1` / `ha-docker-2`, 192.168.2.230231).
CT 105 stays running throughout. Services migrate one stack at a time.
Roll back any stack by restarting it on CT 105 if anything goes wrong.
---
## Prerequisites
- Swarm cluster deployed and healthy (`scripts/docker-swarm/deploy.sh`).
- Both nodes show `Ready / Active / Manager` in `docker node ls`.
- NFS mounts healthy on both swarm nodes (`/mnt/docker/config`, `/mnt/docker/databases`, `/mnt/docker/volumes`).
- Access to FreeIPA DNS admin to update A records during cutover.
---
## 1. Traefik — switch to Docker log rotation
**Current state (CT 105):** Traefik writes access logs to the NFS volume at
`/mnt/docker/volumes/traefik-data/logs/`. `modules/traefik/rotate-logs.nix`
rotates those files via `logrotate`.
**Swarm approach:** Remove file-based access logging from Traefik's static
config and rely on Docker's json-file log driver with built-in rotation.
Traefik container logs (including access events) then live under
`/var/lib/docker/containers/<id>/` on the node running Traefik.
### Steps
**1a.** In the Traefik stack definition, add logging config to the service:
```yaml
services:
traefik:
logging:
driver: "json-file"
options:
max-size: "100m"
max-file: "20"
```
**1b.** In `traefik.yml` (Traefik's static config), remove the `accessLog`
file path if present. To keep structured access logs, use Traefik's
`accessLog.format: json` with no `filePath` — logs then go to stdout and are
captured by the json-file driver above.
**1c.** Deploy Traefik to the swarm:
```bash
# On either swarm manager:
docker stack deploy -c /mnt/docker/config/traefik/docker-compose.yml traefik
```
Traefik should be deployed as a **global mode** service so it runs on all
swarm nodes and handles ingress on whichever node a request arrives at:
```yaml
services:
traefik:
deploy:
mode: global
placement:
constraints:
- node.role == manager
```
**1d.** After confirming Traefik works on the swarm, remove
`traefik/rotate-logs.nix` from the `docker` build type in
`modules/build-types/docker.nix` and rebuild CT 105.
**DNS:** Update `docker.sweet.home` and any service FQDNs that point at
192.168.2.225 to a swarm VIP or round-robin A records once Traefik is running
on the swarm. See section 8 (DNS cutover).
---
## 2. Nextcloud — migrate cron job to sidecar container
**Current state (CT 105):** `modules/docker/nextcloud-cron-job.nix` runs a
systemd timer every 5 minutes that calls:
```bash
docker exec nextcloud-webapp php ./cron.php
```
**Swarm problem:** `docker exec` only works against the local daemon. If
Nextcloud is scheduled on the other swarm node, the exec fails silently and
cron never runs.
**Swarm approach:** Add a `nextcloud-cron` sidecar container to the Nextcloud
stack definition, pinned to the same node as the main Nextcloud container via
placement constraints.
### Steps
**2a.** Choose which swarm node will host Nextcloud (e.g. `ha-docker-1`).
Label that node:
```bash
# On either swarm manager:
docker node update --label-add nextcloud=true ha-docker-1
```
**2b.** In the Nextcloud stack compose file, add the sidecar and pin both
services to the labelled node:
```yaml
services:
nextcloud-webapp:
image: nextcloud:production # pin same version as CT 105
deploy:
replicas: 1
placement:
constraints:
- node.labels.nextcloud == true
# ... existing volumes, env, networks ...
nextcloud-cron:
image: nextcloud:production # same image, different entrypoint
entrypoint: /cron.sh
deploy:
replicas: 1
placement:
constraints:
- node.labels.nextcloud == true # must co-locate with webapp
volumes:
# Same data volume as nextcloud-webapp so cron sees the same files.
- nextcloud-data:/var/www/html
# No ports exposed — cron only runs PHP inside the container.
```
`/cron.sh` is Nextcloud's built-in cron entrypoint. It runs
`php -f /var/www/html/cron.php` in a loop, sleeping for 5 minutes between
runs — identical to the current systemd timer.
**2c.** Migrate Nextcloud's data volume to the swarm:
```
/mnt/docker/volumes/nextcloud-data/ → already on NFS, no migration needed
/mnt/docker/databases/nextcloud/ → already on NFS, no migration needed
```
The NFS paths are identical on the swarm nodes (`mount-data.nix` mounts the
same shares from the same VIP). Stop Nextcloud on CT 105, deploy on the
swarm, confirm it starts cleanly.
**2d.** Remove `nextcloud-cron-job.nix` from `modules/build-types/docker.nix`
and rebuild CT 105 after confirming Nextcloud works on the swarm.
---
## 3. docker-health-to-gotify — update for swarm awareness
**Current state (CT 105):** The script at
`/home/nixos/docker/monitoring/gotify/docker-health-to-gotify.sh` runs every
minute, calls `docker ps --filter health=unhealthy`, and notifies Gotify.
**Swarm behaviour:** The same script runs on both swarm nodes independently,
each monitoring its own local Docker daemon. This gives per-node coverage
across the swarm.
**Changes needed in the script** (edit the copy on the NFS volume — it takes
effect on both nodes simultaneously on the next timer fire):
### 3a. Strip the Swarm task suffix from service names
In swarm mode, `docker ps --format '{{.Names}}'` returns names like
`nextcloud-webapp.1.abc123xyz`. The notification should show `nextcloud-webapp`,
not the full task name.
```bash
# Before:
CONTAINER_NAME=$(docker ps --format '{{.Names}}' ...)
# After:
CONTAINER_NAME=$(docker ps --format '{{.Names}}' ... | cut -d. -f1)
```
### 3b. Include the reporting node in the Gotify message
Add `$(hostname)` to the notification payload so you know which swarm node
detected the problem:
```bash
MESSAGE="[$(hostname)] ${CONTAINER_NAME} is unhealthy"
```
### 3c. Extend to catch swarm service replica failures
`docker ps` only shows what's running locally. If a service has zero healthy
replicas (task crash-looping) it may not show up on either node's `docker ps`
at the same moment. Add a swarm-level check:
```bash
# Run only on managers (both ha-docker nodes are managers):
if docker info --format '{{.Swarm.ControlAvailable}}' 2>/dev/null | grep -q true; then
# Find services where running replicas < desired replicas
docker service ls --format '{{.Name}}\t{{.Replicas}}' | \
awk -F'\t' '$2 !~ /^[0-9]+\/[0-9]+$/ || split($2,a,"/") && a[1] < a[2] { print $1, $2 }' | \
while read -r svc_name replicas; do
# Send Gotify notification for degraded service
curl -s -X POST "${GOTIFY_URL}/message" \
-H "X-Gotify-Key: ${GOTIFY_TOKEN}" \
-d "title=Swarm service degraded" \
-d "message=[$(hostname)] ${svc_name}: ${replicas} replicas"
done
fi
```
This catches the case where a service's desired replicas are not running
(e.g. OOM kill, image pull failure) — a failure mode that doesn't produce a
Docker health event on any node.
---
## 4. Passbolt migration
Passbolt has strict data integrity requirements. Migrate with care:
1. **Backup first**`docker exec passbolt-webapp php /usr/share/php/passbolt/bin/cake passbolt export_keys` and a database dump.
2. Database is on NFS (`/mnt/docker/databases/passbolt/`) — no data copy needed.
3. Pin Passbolt to a specific node: `docker node update --label-add passbolt=true ha-docker-1`
4. Add placement constraint `node.labels.passbolt == true` to the Passbolt stack.
5. Stop on CT 105, deploy on swarm, verify login works.
6. Test email delivery and 2FA.
---
## 5. Gitea migration
Gitea's data directory is on NFS (`/mnt/docker/volumes/gitea-data/`).
1. Stop Gitea on CT 105: `docker stop gitea`
2. Deploy to swarm with placement constraint (pin to `ha-docker-1` initially).
3. Verify web UI and SSH clone/push work.
4. Update DNS: `gitea.lan.ddnsgeek.com` → swarm Traefik endpoint.
5. Update the flake remote URL in `variables.nix` (`giteaDomain`) if the address changes.
---
## 6. Other services
Deploy remaining services (Grafana, InfluxDB, NodeRed, Prometheus, etc.)
as swarm stacks. Most have no special migration concern — they use NFS
volumes already on the shared storage.
Services with stateful databases (PostgreSQL, MariaDB) should follow the
pattern: stop on CT 105, confirm NFS database directory is intact, deploy on
swarm, verify.
---
## 7. Monitoring — Beszel
The Beszel hub runs on CT 105 (`docker.sweet.home:8090`). Both swarm nodes
run `beszel-agent` (from `modules/beszel/enable-agent.nix`), pointing at the
existing hub URL.
No migration needed for Beszel itself during the container migration. Once
all services are on the swarm, you may wish to move the Beszel hub too (as a
swarm service with a placement constraint) but this is optional.
---
## 8. DNS cutover
When a service is confirmed working on the swarm, update the FreeIPA DNS
A record from the CT 105 IP (192.168.2.225) to a swarm node IP or, when a
shared Traefik frontend is in place, to a round-robin record across both nodes.
**Recommended approach — Traefik as the single entry point:**
```
service.lan.ddnsgeek.com → Traefik on swarm (global mode)
docker.sweet.home → keep as 192.168.2.225 (CT 105) until fully decommissioned
```
For LAN-only services using `*.sweet.home` names, update FreeIPA directly:
```bash
# On domain-controller (or via SSH):
ipa dnsrecord-mod sweet.home nextcloud --a-rec=192.168.2.230
# Add 192.168.2.231 as a second A record for round-robin (optional):
ipa dnsrecord-add sweet.home nextcloud --a-rec=192.168.2.231
```
Services behind Traefik don't need their own DNS updates — only Traefik's
own entry point IPs need to change.
---
## 9. NixOS cleanup — CT 105
Once all services are migrated:
**Remove from `modules/build-types/docker.nix`:**
- `../docker/nextcloud-cron-job.nix` — replaced by sidecar container
- `../traefik/rotate-logs.nix` — replaced by Docker log driver
**Keep in `modules/build-types/docker.nix` until CT 105 is decommissioned:**
- `../docker/docker-health-to-gotify.nix` — still monitors CT 105's own daemon
- Everything else
**When decommissioning CT 105:**
1. Confirm all NFS volumes are in use only by swarm services (not CT 105).
2. Stop CT 105: `pct stop 105` on pve1.
3. Archive/remove the `lxc-docker` and `proxmox-docker` targets from `flake.nix`.
4. Remove `hosts/docker/`, `modules/build-types/docker.nix`, and `modules/docker/`.
5. Update `variables.nix` to remove `dockerIp`, `dockerStorageIp`, `dockerHost`
(or reassign `dockerHost` to point at a swarm node for Beszel hub resolution).
---
## Rollback
Any stack can be rolled back to CT 105 independently:
```bash
# On CT 105:
docker start <service-name>
# Update DNS A record back to 192.168.2.225
ipa dnsrecord-mod sweet.home <service> --a-rec=192.168.2.225
```
CT 105 remains running throughout the cutover. Only decommission it after
every service is confirmed stable on the swarm and you have run one full
backup cycle from the new hosts.
-228
View File
@@ -1,228 +0,0 @@
# IP Addressing Scheme
## Subnets
| Subnet | VLAN | CIDR | Purpose | Routed? |
|---|---|---|---|---|
| LAN | 2 (native/untagged) | `192.168.2.0/24` | General LAN — clients and infrastructure | Yes (gateway .254) |
| Cluster | 10 | `192.168.10.224/29` | HA file server DRBD replication + Corosync heartbeat | No — internal `vmbr1` only, no uplink |
| Storage client | 20 | `192.168.20.0/24` | HA file server NFS (and iSCSI if needed) — docker and swarm nodes mount from VIP here | No — internal `vmbr2` only, no uplink |
| Swarm cluster | 30 | `192.168.30.0/24` | Docker Swarm gossip (TCP/UDP 7946) + VXLAN overlay (UDP 4789) | No — internal `vmbr3` only, no uplink |
When expanded to a second Proxmox node, VLAN 10 (cluster), VLAN 20 (storage-client), and VLAN 30 (swarm) all share
the same inter-node trunk NIC via 802.1q VLAN tagging — different VLAN IDs, same physical cable.
The cluster and storage-client subnets never leave pve1. `vmbr1` and `vmbr2` are Proxmox Linux
bridges with no physical port attached; traffic between guests on each bridge stays in-kernel.
VLAN IDs match the third octet of each subnet (VLAN 2 → 192.168.**2**.x, VLAN 10 → 192.168.**10**.x,
VLAN 20 → 192.168.**20**.x). The host octet is consistent across all subnets — e.g. ha-node1
is always `.228`: `192.168.2.228` (LAN), `192.168.10.228` (cluster), `192.168.20.228` (storage client).
**Protocol separation** (enforced by firewall on HA nodes):
- NFS (ports 111, 2049, 20048): both subnets, each restricted to its own CIDR
- VLAN 2 only → `vip-lan` (192.168.2.229) — pxe-boot and other LAN clients
- VLAN 20 only → `vip-storage` (192.168.20.229) — docker, future swarm nodes
- iSCSI (port 3260): VLAN 20 only — available but not in active use; NFS is preferred
for multi-host access (shared volumes across a Docker Swarm require a shared filesystem,
not per-host block devices)
---
## DNS Zones
FreeIPA (domain-controller.sweet.home) is authoritative for all zones.
Four zones correspond to the four subnets. All zones are internal only; no external delegation.
### sweet.home — VLAN 2 (192.168.2.x)
General LAN zone. All infrastructure hostnames live here.
| Hostname | A record | Notes |
|---|---|---|
| `domain-controller.sweet.home` | `192.168.2.253` | FreeIPA / KDC / DNS |
| `ha-vip-lan.sweet.home` | `192.168.2.229` | Pacemaker `vip-lan` — NFS for LAN clients |
| `ha-server-1.sweet.home` | `192.168.2.228` | HA node 1 management NIC |
| `ha-server-2.sweet.home` | `192.168.2.227` | HA node 2 management NIC |
| `server.sweet.home` | `192.168.2.226` | Current ZFS/NFS server (retiring) |
| `docker.sweet.home` | `192.168.2.225` | Docker/Traefik host |
| `nix-cache.sweet.home` | `192.168.2.224` | Nix binary cache + remote builder |
| `pxe-boot.sweet.home` | `192.168.2.223` | PXE / TFTP / HTTP netboot |
| `tailscale-router.sweet.home` | `192.168.2.222` | Tailscale exit node |
| `tor-relay.sweet.home` | `192.168.2.221` | Tor relay |
| `pdm.sweet.home` | `192.168.2.220` | Proxmox Deploy Manager |
| `nixos.sweet.home` | `192.168.2.39` | Bare-metal workstation (DHCP) |
| `pve1.sweet.home` | `192.168.2.245` | Proxmox VE hypervisor |
| `pbs.sweet.home` | `192.168.2.244` | Proxmox Backup Server |
PTR records exist for all static hosts. The workstation (`nixos.sweet.home`) is
DHCP-assigned; its PTR is omitted.
### cluster.home — VLAN 10 (192.168.10.x)
Internal only — Corosync ring0 heartbeat and DRBD replication between HA nodes.
No VIP exists on this subnet (DRBD/Corosync endpoints are static per-node IPs).
| Hostname | A record | Notes |
|---|---|---|
| `ha-server-1.cluster.home` | `192.168.10.228` | HA node 1 cluster NIC (ens19 / vmbr1) |
| `ha-server-2.cluster.home` | `192.168.10.227` | HA node 2 cluster NIC (ens19 / vmbr1) |
PTR records exist for both. DNS here is for debugging convenience — DRBD and
Corosync use the IPs from the NixOS config directly, not DNS.
### storage.home — VLAN 20 (192.168.20.x)
Internal only — NFS (and iSCSI) client access to the HA storage VIP. NFS clients
mount from **`nfs.storage.home`** (the Pacemaker floating VIP) so mounts survive
failover transparently without reconfiguration.
| Hostname | A record | Notes |
|---|---|---|
| `nfs.storage.home` | `192.168.20.229` | Pacemaker `vip-storage` — NFS + iSCSI VIP |
| `ha-server-1.storage.home` | `192.168.20.228` | HA node 1 storage-client NIC (ens20 / vmbr2) |
| `ha-server-2.storage.home` | `192.168.20.227` | HA node 2 storage-client NIC (ens20 / vmbr2) |
| `docker.storage.home` | `192.168.20.225` | Docker host storage-client NIC (eth1 / vmbr2) |
| `server.storage.home` | `192.168.20.226` | server VM storage-client NIC (decommissioned — remove DNS record after VM is destroyed) |
PTR records exist for all five. Remove `server.storage.home`, `server.sweet.home`,
and their PTRs from FreeIPA DNS once the server VM is destroyed.
---
## LAN — 192.168.2.0/24
### Address map
| Range | Purpose |
|---|---|
| .1.9 | Reserved, never assign |
| .10.59 | Client DHCP pool (router-assigned) |
| .60.219 | Unallocated buffer |
| .220.229 | Virtual nodes (VMs / LXC containers) |
| .230.239 | Expansion buffer (reserved, unallocated) |
| .240.249 | Physical nodes (bare-metal hosts) |
| .250.253 | Network services |
| .254 | Router / gateway |
### Network services (.250.253)
| IP | Hostname | Role |
|---|---|---|
| `192.168.2.254` | router | Gateway (TP-Link) |
| `192.168.2.253` | domain-controller | FreeIPA — authoritative DNS for `sweet.home`, Kerberos, LDAP |
| `192.168.2.250``.252` | — | Reserved for future network services |
### Physical nodes (.240.249)
| IP | Hostname | Role |
|---|---|---|
| `192.168.2.245` | pve1 | Proxmox VE hypervisor |
| `192.168.2.244` | pbs | Proxmox Backup Server |
| `192.168.2.243` | nixos | Bare-metal workstation (`baremetal-gui`) |
| `192.168.2.246``.249` | — | Reserved — second Proxmox node and associated services |
| `192.168.2.240``.242` | — | Reserved |
pve1 sits mid-range deliberately so a second Proxmox node can slot in on either side.
### Virtual nodes (.220.229)
All VMs and LXC containers run on pve1.
| IP | Hostname | Role | Status |
|---|---|---|---|
| `192.168.2.229` | ha-vip-lan | HA file server LAN floating VIP (Pacemaker `vip-lan`) — LAN iSCSI + NFS | Active |
| `192.168.2.228` | ha-node1 | HA file server node 1 — management NIC | Active |
| `192.168.2.227` | ha-node2 | HA file server node 2 — management NIC | Active |
| `192.168.2.226` | server | Former NFS/ZFS file server — decommissioned | Removed from flake |
| `192.168.2.225` | docker | Docker / Traefik stack (CT 105 — existing single-host) | Active |
| `192.168.2.224` | nix-cache | Nix binary cache + remote builder | Active |
| `192.168.2.223` | pxe-boot | PXE / TFTP / HTTP netboot server | Active |
| `192.168.2.222` | tailscale-router | Tailscale exit node / router | Active |
| `192.168.2.221` | tor-relay | Tor relay | Active |
| `192.168.2.220` | pdm | Proxmox Deploy Manager | Active |
| `192.168.2.231` | ha-docker-2 | Docker Swarm node 2 — management NIC | Active |
| `192.168.2.230` | ha-docker-1 | Docker Swarm node 1 — management NIC | Active |
### Client DHCP pool (.10.59)
Assigned by the router. DNS option points to `192.168.2.253` (domain-controller).
Devices in this range: phones, laptops, IoT, Canon printer, any non-infrastructure host.
No static reservations for infrastructure hosts — all infra uses static IP configuration
on the guest itself (not DHCP reservations), so IPs survive VM recreation regardless of
MAC address churn.
---
## Cluster network — VLAN 10 — 192.168.10.224/29
Internal to pve1 only. Proxmox bridge `vmbr1`, no physical NIC attached.
| IP | Hostname | Interface role |
|---|---|---|
| `192.168.10.228` | ha-node1 | DRBD replication + Corosync ring0 (primary heartbeat) |
| `192.168.10.227` | ha-node2 | DRBD replication + Corosync ring0 (primary heartbeat) |
| — | no gateway | Isolated — not routed to LAN or internet |
Corosync ring1 (backup heartbeat only) uses the LAN IPs (`192.168.2.228` / `192.168.2.227`)
over `vmbr0` — no additional bridge needed, and DRBD traffic never crosses ring1.
---
## Storage-client network — VLAN 20 — 192.168.20.0/24
Internal to pve1 only. Proxmox bridge `vmbr2`, no physical NIC attached.
| IP | Hostname | Interface / role |
|---|---|---|
| `192.168.20.229` | ha-vip-storage | Pacemaker floating VIP — NFS + iSCSI endpoint |
| `192.168.20.228` | ha-node1 | Storage-client NIC (ens20 / vmbr2) |
| `192.168.20.227` | ha-node2 | Storage-client NIC (ens20 / vmbr2) |
| `192.168.20.226` | server | Storage-client NIC (ens19 / vmbr2) — decommissioned |
| `192.168.20.225` | docker | Storage-client NIC (eth1 / vmbr2) — NFS client (CT 105) |
| `192.168.20.231` | ha-docker-2 | Storage-client NIC (ens19 / vmbr2) — NFS client |
| `192.168.20.230` | ha-docker-1 | Storage-client NIC (ens19 / vmbr2) — NFS client |
| — | no gateway | Isolated — not routed to LAN or internet |
NFS clients mount from `192.168.20.229` (surviving failover transparently via the VIP).
Firewall on each HA node restricts NFS and iSCSI ports to `192.168.20.0/24` — LAN hosts
cannot reach either service on this VIP. The `vip-storage` endpoint is not reachable
from the workstation directly (internal bridge only); health checks proxy through the
active HA node.
---
## Swarm cluster network — VLAN 30 — 192.168.30.0/24
Internal to pve1 only. Proxmox bridge `vmbr3`, no physical NIC attached.
Carries Docker Swarm inter-node traffic only: Raft consensus (TCP 2377),
Serf gossip (TCP/UDP 7946), and VXLAN overlay data path (UDP 4789).
Docker Swarm is initialised with `--advertise-addr` and `--data-path-addr`
both pointing to this subnet so all cluster traffic stays on `vmbr3` and
never crosses the LAN.
| IP | Hostname | Interface / role |
|---|---|---|
| `192.168.30.231` | ha-docker-2 | Swarm cluster NIC (ens20 / vmbr3) |
| `192.168.30.230` | ha-docker-1 | Swarm cluster NIC (ens20 / vmbr3) |
| — | no gateway | Isolated — not routed to LAN or internet |
### DNS zone: `swarm.home` — VLAN 30 (192.168.30.x)
| Hostname | A record | Notes |
|---|---|---|
| `ha-docker-1.swarm.home` | `192.168.30.230` | Swarm NIC — debugging only |
| `ha-docker-2.swarm.home` | `192.168.30.231` | Swarm NIC — debugging only |
Operators reach the Docker API on the LAN IPs (`192.168.2.230`/`.231`), not these addresses.
The `swarm.home` records exist for diagnostic convenience (e.g. confirming `vmbr3` routing).
### Multi-node Proxmox expansion
When a second Proxmox node (pve2) is added, VLAN 10 (cluster), VLAN 20 (storage-client),
and VLAN 30 (swarm) all extend to pve2 via 802.1q VLAN tagging on the inter-node trunk
link. All three internal networks share the same physical NIC between hypervisors —
VLAN tags provide the logical separation.
+22 -131
View File
@@ -1,10 +1,9 @@
# pxe-boot # pxe-boot
The `pxe-boot` host serves HTTP boot assets for iPXE clients — including The `pxe-boot` host serves HTTP boot assets for iPXE clients — including a
self-staged copies of both this flake's own auto-installer netboot image self-staged copy of this flake's own auto-installer netboot image, see
(see `docs/auto-installer.md` for what that image actually is and does once `docs/auto-installer.md` for what that image actually is and does once
booted) and a vanilla, unmodified NixOS minimal netboot image for plain booted.
rescue/inspection use.
## Host Role ## Host Role
@@ -15,7 +14,6 @@ rescue/inspection use.
- TFTP root for first-stage bootloaders: `/srv/pxe/tftp` - TFTP root for first-stage bootloaders: `/srv/pxe/tftp`
- iPXE entry script: `/srv/pxe/http/boot.ipxe` - iPXE entry script: `/srv/pxe/http/boot.ipxe`
- Generated iPXE menu: `/srv/pxe/http/menu.ipxe` - Generated iPXE menu: `/srv/pxe/http/menu.ipxe`
- Debian Minimal iPXE script: `/srv/pxe/http/debian.ipxe`
- SystemRescue iPXE script: `/srv/pxe/http/systemrescue.ipxe` - SystemRescue iPXE script: `/srv/pxe/http/systemrescue.ipxe`
- TFTP fallback script: `/srv/pxe/tftp/autoexec.ipxe` - TFTP fallback script: `/srv/pxe/tftp/autoexec.ipxe`
- Boot binaries copied from the Nix `ipxe` package: - Boot binaries copied from the Nix `ipxe` package:
@@ -29,140 +27,43 @@ The host creates these directories with systemd tmpfiles:
```text ```text
/srv/pxe /srv/pxe
/srv/pxe/http /srv/pxe/http
/srv/pxe/http/images -> /mnt/pxe-images (symlink to NFS share) /srv/pxe/http/images
/srv/pxe/http/auto-installer /srv/pxe/http/nixos
/srv/pxe/http/nixos-minimal
/srv/pxe/http/debian
/srv/pxe/http/systemrescue /srv/pxe/http/systemrescue
/srv/pxe/http/ubuntu /srv/pxe/http/ubuntu
/srv/pxe/http/rescue /srv/pxe/http/rescue
/srv/pxe/tftp /srv/pxe/tftp
``` ```
`/srv/pxe/http/images` is a symlink to `/mnt/pxe-images`, which is an NFS Mount shared image storage under `/srv/pxe/http`, preferably
mount of `server.sweet.home:/tank/pxe-boot/images` `/srv/pxe/http/images` unless a menu entry expects files in a specific
(`modules/pxe-boot/mount-pxe-images.nix`). Place large images there (ISOs, directory such as `/srv/pxe/http/nixos`.
disk images) rather than on the pxe-boot host's own root disk. For an LXC
pxe-boot container the mount uses NFSv3+nolock with `nofail` (eager,
non-blocking on server unavailability); for a Proxmox VM it uses NFSv4.2
with `x-systemd.automount` (lazy, triggered on first access).
When running as `lxc-pxe-boot`, the Proxmox container must have
`features: nesting=1,mount=nfs` (at minimum) in its Proxmox config. `nesting=1`
is required by systemd 260+ for credential isolation (user namespace creation
and internal move-mounts); without it, AppArmor denies both, and every
systemd service that uses `PrivateUsers`, `PrivateDevices`, or credential
passing fails on boot. `mount=nfs` allows the NFSv3 mount. Both are set
automatically by `scripts/proxmox/create-proxmox-resource.sh` (via
`PROXMOX_DEFAULT_LXC_FEATURES` in `scripts/env.sh` which defaults to
`nesting=1,keyctl=1,mount=nfs;nfs4`). If you ever change these features
manually via `pct set`, be sure to include both — `pct set` replaces the
entire features string, it does not append to it.
The HTTP iPXE chain is: The HTTP iPXE chain is:
```text ```text
undionly.kpxe or ipxe.efi undionly.kpxe or ipxe.efi
-> autoexec.ipxe from the TFTP root, when iPXE requests it -> autoexec.ipxe from the TFTP root, when iPXE requests it
-> http://192.168.2.223/boot.ipxe -> http://192.168.2.247/boot.ipxe
-> http://192.168.2.223/menu.ipxe -> http://192.168.2.247/menu.ipxe
``` ```
The generated menu currently exposes entries for: The generated menu currently exposes entries for:
- NixOS Auto-Installer - NixOS installer
- NixOS Minimal
- Debian Minimal
- FreeIPA Server (Rocky Linux 9)
- SystemRescue environment - SystemRescue environment
- iPXE shell - iPXE shell
- Reboot - Reboot
Both NixOS entries chain-load a `netboot.ipxe` staged into their own The NixOS installer entry chain-loads `/srv/pxe/http/nixos/netboot.ipxe`,
directory (`/srv/pxe/http/auto-installer/netboot.ipxe` and which is nixpkgs' own generated netboot iPXE script (correct `init=`/`initrd=`
`/srv/pxe/http/nixos-minimal/netboot.ipxe`), each nixpkgs' own generated kernel parameters included) rather than a hand-rolled boot line — that script
netboot iPXE script (correct `init=`/`initrd=` kernel parameters included) in turn expects its kernel/initrd siblings in the same directory. All three
rather than a hand-rolled boot line — that script in turn expects its files (`bzImage`, `initrd`, `netboot.ipxe`) are built from this flake's own
kernel/initrd siblings in the same directory. Each directory's three files `modules/installer/iso.nix` netboot image (the same one `nix build .#pxe`
(`bzImage`, `initrd`, `netboot.ipxe`) are built from source and staged produces) and staged automatically by
automatically by `modules/pxe-boot/stage-installer-artifacts.nix` via `modules/pxe-boot/stage-installer-artifacts.nix` via `systemd.tmpfiles.rules`
`systemd.tmpfiles.rules` — no manual operator step required: — no manual operator step required.
- `auto-installer` is this flake's own `netbootSystem` (`flake.nix`) — the
same auto-installer image `nix build .#pxe` produces. See
`docs/auto-installer.md`.
- `nixos-minimal` is `netbootMinimalSystem` (`flake.nix`) — nixpkgs'
`netboot-minimal.nix` composed on its own, with none of this flake's
auto-installer wiring (no `common.nix`, no `auto-install.sh`, no baked
host keys or custom users). Same `nix build .#pxe-minimal` mechanism as
the auto-installer image, just a different module composition. Useful
as a plain rescue/inspection shell that doesn't assume anything about
this flake.
Both images set `networking.hostName` to match their menu entry/staged
directory name (`auto-installer` / `nixos-minimal`), so each one's
generated system name (`nixos-system-<name>-*`) is self-describing rather
than the nixpkgs default of `nixos-system-nixos-*` for both.
The Debian Minimal entry chains `http://<pxeServerIp>/debian.ipxe`, which loads
the Debian bookworm netboot kernel and initrd from `/srv/pxe/http/debian/`. The
`fetch-debian-netboot.service` oneshot downloads these files from
`deb.debian.org` on first boot (idempotent — skips if files are already
present):
```text
/srv/pxe/http/debian/linux (Debian bookworm netboot kernel)
/srv/pxe/http/debian/initrd.gz (Debian bookworm netboot initrd)
```
The service requires outbound internet access on the pxe-boot host. To
re-download (e.g. after a Debian point release), delete the files and restart
the service:
```bash
rm /srv/pxe/http/debian/linux /srv/pxe/http/debian/initrd.gz
systemctl restart fetch-debian-netboot.service
```
To update to a different Debian release, change `debianRelease` in
`modules/build-types/pxe-boot.nix` and redeploy.
The **FreeIPA Server (Rocky Linux 9)** entry chains
`http://<pxeServerIp>/rocky-freeipa.ipxe`, which boots the Rocky Linux 9
Anaconda installer with a Kickstart file (`rocky-freeipa.ks`) hosted on the
same server. The `fetch-rocky-pxeboot.service` oneshot downloads the pxeboot
kernel and initrd from the Rocky Linux mirror on first boot (idempotent):
```text
/srv/pxe/http/rocky/vmlinuz (Rocky Linux 9 Anaconda pxeboot kernel)
/srv/pxe/http/rocky/initrd.img (Rocky Linux 9 Anaconda pxeboot initrd)
```
The Kickstart file is generated from the NixOS module and staged at
`/srv/pxe/http/rocky-freeipa.ks`. It performs a fully unattended install:
1. Installs Rocky Linux 9 with `ipa-server` + `ipa-server-dns` packages
2. Configures static IP `192.168.2.138`, hostname `domain-controller.sweet.home`
3. Creates user `wayne` with the `adminSshKey` from `variables.nix`
4. Generates random IPA passwords and writes them to `/root/ipa-credentials.txt`
5. Creates a `freeipa-first-boot.service` oneshot that runs `ipa-server-install`
on first reboot (~20 minutes)
After the install completes:
- SSH in as `wayne@domain-controller` using the admin key
- Monitor FreeIPA install progress: `sudo tail -f /root/freeipa-install.log`
- Retrieve credentials: `sudo cat /root/ipa-credentials.txt` (save to password manager)
- Configure Pi-hole: `server=/sweet.home/192.168.2.138` in dnsmasq
To refresh the pxeboot files (e.g. after a Rocky point release):
```bash
rm /srv/pxe/http/rocky/vmlinuz /srv/pxe/http/rocky/initrd.img
systemctl restart fetch-rocky-pxeboot.service
```
To update to a different Rocky release, change `rockyRelease` in
`modules/build-types/pxe-boot.nix` and redeploy.
The SystemRescue entry expects the source ISO at: The SystemRescue entry expects the source ISO at:
@@ -170,16 +71,13 @@ The SystemRescue entry expects the source ISO at:
/srv/pxe/http/images/systemrescue.iso /srv/pxe/http/images/systemrescue.iso
``` ```
Since `/srv/pxe/http/images` is the NFS-backed symlink, place the ISO on the
NFS share at `server.sweet.home:/tank/pxe-boot/images/systemrescue.iso`.
The `stage-systemrescue.service` oneshot extracts that ISO into: The `stage-systemrescue.service` oneshot extracts that ISO into:
```text ```text
/srv/pxe/http/systemrescue /srv/pxe/http/systemrescue
``` ```
The rescue menu entry then chains `http://192.168.2.223/systemrescue.ipxe`, The rescue menu entry then chains `http://192.168.2.247/systemrescue.ipxe`,
which loads the SystemRescue kernel and initramfs from the extracted tree and which loads the SystemRescue kernel and initramfs from the extracted tree and
uses `archiso_http_srv` to fetch the squashfs payload over HTTP. uses `archiso_http_srv` to fetch the squashfs payload over HTTP.
@@ -196,13 +94,6 @@ After deployment by an operator, basic service checks are:
```bash ```bash
curl http://pxe-boot/boot.ipxe curl http://pxe-boot/boot.ipxe
curl http://pxe-boot/menu.ipxe curl http://pxe-boot/menu.ipxe
curl http://pxe-boot/debian.ipxe
curl -I http://pxe-boot/debian/linux
curl -I http://pxe-boot/debian/initrd.gz
curl http://pxe-boot/rocky-freeipa.ipxe
curl http://pxe-boot/rocky-freeipa.ks
curl -I http://pxe-boot/rocky/vmlinuz
curl -I http://pxe-boot/rocky/initrd.img
curl http://pxe-boot/systemrescue.ipxe curl http://pxe-boot/systemrescue.ipxe
curl -I http://pxe-boot/systemrescue/sysresccd/boot/x86_64/vmlinuz curl -I http://pxe-boot/systemrescue/sysresccd/boot/x86_64/vmlinuz
curl -I http://pxe-boot/systemrescue/sysresccd/boot/x86_64/sysresccd.img curl -I http://pxe-boot/systemrescue/sysresccd/boot/x86_64/sysresccd.img
Generated
+7 -157
View File
@@ -1,62 +1,5 @@
{ {
"nodes": { "nodes": {
"clan-core": {
"inputs": {
"data-mesher": "data-mesher",
"disko": [
"disko"
],
"flake-parts": "flake-parts",
"nix-darwin": "nix-darwin",
"nix-select": "nix-select",
"nixpkgs": [
"nixpkgs"
],
"sops-nix": [
"sops-nix"
],
"systems": "systems",
"treefmt-nix": "treefmt-nix"
},
"locked": {
"lastModified": 1783497933,
"narHash": "sha256-TxmwEews6URFPqOWEHNychtXbFDgLZjbOfEXtvtOm6U=",
"rev": "3dc0221ca09033599fe98055e9bbc81bdf32732a",
"type": "tarball",
"url": "https://git.clan.lol/api/v1/repos/clan/clan-core/archive/3dc0221ca09033599fe98055e9bbc81bdf32732a.tar.gz"
},
"original": {
"type": "tarball",
"url": "https://git.clan.lol/clan/clan-core/archive/26.05.tar.gz"
}
},
"data-mesher": {
"inputs": {
"flake-parts": [
"clan-core",
"flake-parts"
],
"nixpkgs": [
"clan-core",
"nixpkgs"
],
"treefmt-nix": [
"clan-core",
"treefmt-nix"
]
},
"locked": {
"lastModified": 1778718524,
"narHash": "sha256-pXLoI6Ax0EnUK6r34UM1vibVC7CfTu6j72R2692ZzPs=",
"rev": "12c552ad547d87254f33f33bddd1a2cdbeac754d",
"type": "tarball",
"url": "https://git.clan.lol/api/v1/repos/clan/data-mesher/archive/12c552ad547d87254f33f33bddd1a2cdbeac754d.tar.gz"
},
"original": {
"type": "tarball",
"url": "https://git.clan.lol/clan/data-mesher/archive/main.tar.gz"
}
},
"disko": { "disko": {
"inputs": { "inputs": {
"nixpkgs": [ "nixpkgs": [
@@ -108,30 +51,9 @@
"type": "github" "type": "github"
} }
}, },
"flake-parts": {
"inputs": {
"nixpkgs-lib": [
"clan-core",
"nixpkgs"
]
},
"locked": {
"lastModified": 1778716662,
"narHash": "sha256-m1Yf0wZ8j1OHjTc2UwHwyQRSnNeSgLJOd7q5Y45hzi4=",
"owner": "hercules-ci",
"repo": "flake-parts",
"rev": "f7c1a2d347e4c52d5fb8d10cb4d94b5884e546fb",
"type": "github"
},
"original": {
"owner": "hercules-ci",
"repo": "flake-parts",
"type": "github"
}
},
"flake-utils": { "flake-utils": {
"inputs": { "inputs": {
"systems": "systems_2" "systems": "systems"
}, },
"locked": { "locked": {
"lastModified": 1694529238, "lastModified": 1694529238,
@@ -173,11 +95,11 @@
] ]
}, },
"locked": { "locked": {
"lastModified": 1785119570, "lastModified": 1783740085,
"narHash": "sha256-Rgs2xKnGLFWQscxUaXX07oyZeuMDOHEbqDOsgliLFGM=", "narHash": "sha256-qajyHfZY29G2oEQk+uHxmsJcRoBUBXP9maTpFlwP/dI=",
"owner": "nix-community", "owner": "nix-community",
"repo": "home-manager", "repo": "home-manager",
"rev": "d4fd24667c8cbef124bb70a20380cab75ec8474d", "rev": "3cd22efe6471dc7365c822bd9ad73a21e55f38fb",
"type": "github" "type": "github"
}, },
"original": { "original": {
@@ -187,40 +109,6 @@
"type": "github" "type": "github"
} }
}, },
"nix-darwin": {
"inputs": {
"nixpkgs": [
"clan-core",
"nixpkgs"
]
},
"locked": {
"lastModified": 1779036909,
"narHash": "sha256-zXcwYQGCT6pzinK+1dBB2ekTVtfxGZAapb3Evdcu4fY=",
"owner": "nix-darwin",
"repo": "nix-darwin",
"rev": "56c666e108467d87d13508936aade6d567f2a501",
"type": "github"
},
"original": {
"owner": "nix-darwin",
"repo": "nix-darwin",
"type": "github"
}
},
"nix-select": {
"locked": {
"lastModified": 1763303120,
"narHash": "sha256-yxcNOha7Cfv2nhVpz9ZXSNKk0R7wt4AiBklJ8D24rVg=",
"rev": "3d1e3860bef36857a01a2ddecba7cdb0a14c35a9",
"type": "tarball",
"url": "https://git.clan.lol/api/v1/repos/clan/nix-select/archive/3d1e3860bef36857a01a2ddecba7cdb0a14c35a9.tar.gz"
},
"original": {
"type": "tarball",
"url": "https://git.clan.lol/clan/nix-select/archive/main.tar.gz"
}
},
"nixos-conf-editor": { "nixos-conf-editor": {
"inputs": { "inputs": {
"flake-compat": "flake-compat", "flake-compat": "flake-compat",
@@ -259,11 +147,11 @@
}, },
"nixpkgs_2": { "nixpkgs_2": {
"locked": { "locked": {
"lastModified": 1785133411, "lastModified": 1784011430,
"narHash": "sha256-Yjv0WEg39KRYS0rBdTbu6Fc/or/ihAKk13W9sQ6VWd0=", "narHash": "sha256-lDebytrYdd47IBLwvNOD+6AGeoqZ78CIKlp70hzW280=",
"owner": "NixOS", "owner": "NixOS",
"repo": "nixpkgs", "repo": "nixpkgs",
"rev": "2f5a153c270b70cb0f8c11f46d96d6d3bc39f4e3", "rev": "8eeec934ae0dbeca3d7868c059568a65c08b2fc3",
"type": "github" "type": "github"
}, },
"original": { "original": {
@@ -275,7 +163,6 @@
}, },
"root": { "root": {
"inputs": { "inputs": {
"clan-core": "clan-core",
"disko": "disko", "disko": "disko",
"home-manager": "home-manager", "home-manager": "home-manager",
"nixos-conf-editor": "nixos-conf-editor", "nixos-conf-editor": "nixos-conf-editor",
@@ -327,22 +214,6 @@
} }
}, },
"systems": { "systems": {
"locked": {
"lastModified": 1774449309,
"narHash": "sha256-brhZ8DmuGtzkCYHJg4HEd602amKm89Y9ytsFZ5uWD1w=",
"owner": "nix-systems",
"repo": "default",
"rev": "c29398b59d2048c4ab79345812849c9bd15e9150",
"type": "github"
},
"original": {
"owner": "nix-systems",
"ref": "future-26.11",
"repo": "default",
"type": "github"
}
},
"systems_2": {
"locked": { "locked": {
"lastModified": 1681028828, "lastModified": 1681028828,
"narHash": "sha256-Vy1rq5AaRuLzOxct8nz4T6wlgyUR7zLU309k9mBC768=", "narHash": "sha256-Vy1rq5AaRuLzOxct8nz4T6wlgyUR7zLU309k9mBC768=",
@@ -356,27 +227,6 @@
"repo": "default", "repo": "default",
"type": "github" "type": "github"
} }
},
"treefmt-nix": {
"inputs": {
"nixpkgs": [
"clan-core",
"nixpkgs"
]
},
"locked": {
"lastModified": 1780220602,
"narHash": "sha256-eynAfOmbmxJnkp7YewvCEbShNnnYJ9gLLqkzsYtBPeM=",
"owner": "numtide",
"repo": "treefmt-nix",
"rev": "db947814a175b7ca6ded66e21383d938df01c227",
"type": "github"
},
"original": {
"owner": "numtide",
"repo": "treefmt-nix",
"type": "github"
}
} }
}, },
"root": "root", "root": "root",
+10 -98
View File
@@ -16,19 +16,6 @@
url = "github:Mic92/sops-nix"; url = "github:Mic92/sops-nix";
inputs.nixpkgs.follows = "nixpkgs"; inputs.nixpkgs.follows = "nixpkgs";
}; };
clan-core = {
url = "https://git.clan.lol/clan/clan-core/archive/26.05.tar.gz";
# Deduplicate modules: clan-core bundles its own disko and sops-nix
# (both imported by nixosModules.clanCore). Without follows, we'd get
# two different versions of each, and disko's _module.args.diskoLib
# unique option would conflict. With follows, clan-core uses the same
# store paths as us, so NixOS deduplicates the imports.
inputs = {
nixpkgs.follows = "nixpkgs";
disko.follows = "disko";
sops-nix.follows = "sops-nix";
};
};
}; };
outputs = { self, nixpkgs, nixos-conf-editor, home-manager, sops-nix, ... } @ inputs: outputs = { self, nixpkgs, nixos-conf-editor, home-manager, sops-nix, ... } @ inputs:
@@ -45,31 +32,15 @@
# (hostName, hostId, per-machine secrets). Every build type except # (hostName, hostId, per-machine secrets). Every build type except
# nix-cache itself consumes the nix-cache substituter and remote # nix-cache itself consumes the nix-cache substituter and remote
# builder. # builder.
mkTarget = { platform, buildType, hostPath, homeFile ? ./modules/common/home.nix, nameSuffix ? "" }: mkTarget = { platform, buildType, hostPath, homeFile ? ./modules/common/home.nix }:
let let
flakeTarget = "${platform}-${buildType}${nameSuffix}"; flakeTarget = "${platform}-${buildType}";
in in
nixpkgs.lib.nixosSystem { nixpkgs.lib.nixosSystem {
inherit system; inherit system;
modules = [ modules = [
inputs.disko.nixosModules.disko inputs.disko.nixosModules.disko
sops-nix.nixosModules.sops sops-nix.nixosModules.sops
inputs.clan-core.nixosModules.clanCore
{
# Required clan settings. directory is the flake root (where
# vars/ and sops/ directories live); machine.name is the flake
# target name (matches what clan vars generate uses as the key
# under vars/per-machine/). enableRecommendedDefaults = false
# is mandatory: without it, clan unconditionally enables
# networking.useNetworkd, adds packages, and tweaks nix settings
# -- none of which belong here.
clan.core = {
settings.directory = self;
settings.machine.name = flakeTarget;
enableRecommendedDefaults = false;
};
}
./modules/clan/ssh-host-key.nix
./modules/common/configuration.nix ./modules/common/configuration.nix
./modules/platforms/${platform}.nix ./modules/platforms/${platform}.nix
./modules/build-types/${buildType}.nix ./modules/build-types/${buildType}.nix
@@ -94,7 +65,7 @@
# file without a same-option circular dependency (a module # file without a same-option circular dependency (a module
# contributing to environment.etc can't read the merged # contributing to environment.etc can't read the merged
# environment.etc it's itself contributing to). # environment.etc it's itself contributing to).
specialArgs = { inherit inputs vars netbootSystem netbootMinimalSystem flakeTarget; }; specialArgs = { inherit inputs vars netbootSystem flakeTarget; };
}; };
# Generated platform x build-type matrix. pxe-boot has no linode # Generated platform x build-type matrix. pxe-boot has no linode
@@ -109,6 +80,9 @@
proxmox-nix-cache = mkTarget { platform = "proxmox"; buildType = "nix-cache"; hostPath = ./hosts/nix-cache/host.nix; }; proxmox-nix-cache = mkTarget { platform = "proxmox"; buildType = "nix-cache"; hostPath = ./hosts/nix-cache/host.nix; };
lxc-nix-cache = mkTarget { platform = "lxc"; buildType = "nix-cache"; hostPath = ./hosts/nix-cache/host.nix; }; lxc-nix-cache = mkTarget { platform = "lxc"; buildType = "nix-cache"; hostPath = ./hosts/nix-cache/host.nix; };
linode-server = mkTarget { platform = "linode"; buildType = "server"; hostPath = ./hosts/server/host.nix; };
proxmox-server = mkTarget { platform = "proxmox"; buildType = "server"; hostPath = ./hosts/server/host.nix; };
lxc-server = mkTarget { platform = "lxc"; buildType = "server"; hostPath = ./hosts/server/host.nix; };
linode-docker = mkTarget { platform = "linode"; buildType = "docker"; hostPath = ./hosts/docker/host.nix; }; linode-docker = mkTarget { platform = "linode"; buildType = "docker"; hostPath = ./hosts/docker/host.nix; };
proxmox-docker = mkTarget { platform = "proxmox"; buildType = "docker"; hostPath = ./hosts/docker/host.nix; }; proxmox-docker = mkTarget { platform = "proxmox"; buildType = "docker"; hostPath = ./hosts/docker/host.nix; };
@@ -117,22 +91,15 @@
linode-gui = mkTarget { platform = "linode"; buildType = "gui"; hostPath = ./hosts/nixos/host.nix; homeFile = ./hosts/nixos/home.nix; }; linode-gui = mkTarget { platform = "linode"; buildType = "gui"; hostPath = ./hosts/nixos/host.nix; homeFile = ./hosts/nixos/home.nix; };
proxmox-gui = mkTarget { platform = "proxmox"; buildType = "gui"; hostPath = ./hosts/nixos/host.nix; homeFile = ./hosts/nixos/home.nix; }; proxmox-gui = mkTarget { platform = "proxmox"; buildType = "gui"; hostPath = ./hosts/nixos/host.nix; homeFile = ./hosts/nixos/home.nix; };
lxc-gui = mkTarget { platform = "lxc"; buildType = "gui"; hostPath = ./hosts/nixos/host.nix; homeFile = ./hosts/nixos/home.nix; }; lxc-gui = mkTarget { platform = "lxc"; buildType = "gui"; hostPath = ./hosts/nixos/host.nix; homeFile = ./hosts/nixos/home.nix; };
baremetal-gui = mkTarget { platform = "baremetal"; buildType = "gui"; hostPath = ./hosts/nixos/host.nix; homeFile = ./hosts/nixos/home.nix; };
proxmox-pxe-boot = mkTarget { platform = "proxmox"; buildType = "pxe-boot"; hostPath = ./hosts/pxe-boot/host.nix; }; proxmox-pxe-boot = mkTarget { platform = "proxmox"; buildType = "pxe-boot"; hostPath = ./hosts/pxe-boot/host.nix; };
lxc-pxe-boot = mkTarget { platform = "lxc"; buildType = "pxe-boot"; hostPath = ./hosts/pxe-boot/host.nix; }; lxc-pxe-boot = mkTarget { platform = "lxc"; buildType = "pxe-boot"; hostPath = ./hosts/pxe-boot/host.nix; };
linode-tailscale-router = mkTarget { platform = "linode"; buildType = "tailscale-router"; hostPath = ./hosts/tailscale-router/host.nix; }; linode-tailscale-exit-node = mkTarget { platform = "linode"; buildType = "tailscale-exit-node"; hostPath = ./hosts/tailscale-exit-node/host.nix; };
proxmox-tailscale-router = mkTarget { platform = "proxmox"; buildType = "tailscale-router"; hostPath = ./hosts/tailscale-router/host.nix; }; proxmox-tailscale-exit-node = mkTarget { platform = "proxmox"; buildType = "tailscale-exit-node"; hostPath = ./hosts/tailscale-exit-node/host.nix; };
lxc-tailscale-router = mkTarget { platform = "lxc"; buildType = "tailscale-router"; hostPath = ./hosts/tailscale-router/host.nix; }; lxc-tailscale-exit-node = mkTarget { platform = "lxc"; buildType = "tailscale-exit-node"; hostPath = ./hosts/tailscale-exit-node/host.nix; };
lxc-tor-relay = mkTarget { platform = "lxc"; buildType = "tor-relay"; hostPath = ./hosts/tor-relay/host.nix; }; lxc-tor-relay = mkTarget { platform = "lxc"; buildType = "tor-relay"; hostPath = ./hosts/tor-relay/host.nix; };
proxmox-ha-server-1 = mkTarget { platform = "proxmox"; buildType = "ha-server"; hostPath = ./hosts/ha-server-1/host.nix; nameSuffix = "-1"; };
proxmox-ha-server-2 = mkTarget { platform = "proxmox"; buildType = "ha-server"; hostPath = ./hosts/ha-server-2/host.nix; nameSuffix = "-2"; };
proxmox-ha-docker-1 = mkTarget { platform = "proxmox"; buildType = "ha-docker"; hostPath = ./hosts/ha-docker-1/host.nix; nameSuffix = "-1"; };
proxmox-ha-docker-2 = mkTarget { platform = "proxmox"; buildType = "ha-docker"; hostPath = ./hosts/ha-docker-2/host.nix; nameSuffix = "-2"; };
}; };
# Auto-install environments (migrated from the former nix-auto-installer # Auto-install environments (migrated from the former nix-auto-installer
@@ -152,65 +119,19 @@
# Same installer environment, built as netboot (kernel + initrd + # Same installer environment, built as netboot (kernel + initrd +
# iPXE script) instead of an ISO — this is what packages.pxe bundles. # iPXE script) instead of an ISO — this is what packages.pxe bundles.
#
# Deliberately imports common.nix directly, NOT ./modules/installer/iso.nix
# (which pulls in nixpkgs' installation-cd-minimal.nix) -- confirmed live
# that composing the ISO module together with netboot-minimal.nix hangs
# every boot waiting for a device that can never exist on a netboot
# client ("A start job is running for /dev/disk/by-label/nixos-minimal-...").
# Both installation-cd-base.nix and netboot.nix set fileSystems."/" via
# the identical lib.mkImageMediaOverride (mkOverride 60) priority --
# genuinely conflicting root-filesystem strategies (ISO-by-label vs.
# netboot-tmpfs) at the same priority, and the ISO one was winning.
# netboot-minimal.nix's own chain (netboot-base.nix) already imports
# profiles/installation-device.nix independently, so common.nix's
# initialHashedPassword override (which assumes that profile is
# present) still applies correctly without iso.nix in the mix.
#
# networking.hostName is set explicitly (rather than left at nixpkgs'
# own "nixos" default) so this image's generated system name
# (nixos-system-auto-installer-*) matches its iPXE menu entry —
# see modules/build-types/pxe-boot.nix's :auto-installer item — and
# its staged directory, /srv/pxe/http/auto-installer.
netbootSystem = nixpkgs.lib.nixosSystem { netbootSystem = nixpkgs.lib.nixosSystem {
inherit system; inherit system;
modules = [ modules = [
./modules/installer/common.nix ./modules/installer/iso.nix
({ modulesPath, ... }: { ({ modulesPath, ... }: {
imports = [ imports = [
(modulesPath + "/installer/netboot/netboot-minimal.nix") (modulesPath + "/installer/netboot/netboot-minimal.nix")
]; ];
}) })
{ networking.hostName = "auto-installer"; }
]; ];
specialArgs = { inherit vars; }; specialArgs = { inherit vars; };
}; };
# A genuinely vanilla NixOS minimal netboot image: nixpkgs'
# netboot-minimal.nix on its own, with none of this flake's
# auto-installer wiring (no common.nix — no auto-install.sh, no
# baked host keys, no custom users/passwords). Built from source via
# the same nixosSystem + netboot-minimal.nix path as netbootSystem
# above, so both go through an identical build mechanism; the only
# difference is what's composed in. hostName again matches this
# image's iPXE menu entry (:nixos-minimal) and staged directory
# (/srv/pxe/http/nixos-minimal).
netbootMinimalSystem = nixpkgs.lib.nixosSystem {
inherit system;
modules = [
({ modulesPath, ... }: {
imports = [
(modulesPath + "/installer/netboot/netboot-minimal.nix")
];
})
{
networking.hostName = "nixos-minimal";
system.stateVersion = "26.05";
boot.zfs.forceImportRoot = false;
}
];
};
in in
{ {
@@ -232,15 +153,6 @@
{ name = "initrd"; path = netbootSystem.config.system.build.netbootRamdisk; } { name = "initrd"; path = netbootSystem.config.system.build.netbootRamdisk; }
{ name = "kernel"; path = netbootSystem.config.system.build.kernel; } { name = "kernel"; path = netbootSystem.config.system.build.kernel; }
]; ];
# Vanilla NixOS minimal netboot bundle — see netbootMinimalSystem
# above. Staged onto the pxe-boot host alongside packages.pxe by
# modules/pxe-boot/stage-installer-artifacts.nix.
pxe-minimal = pkgs.linkFarm "pxe-minimal" [
{ name = "netboot.ipxe"; path = netbootMinimalSystem.config.system.build.netbootIpxeScript; }
{ name = "initrd"; path = netbootMinimalSystem.config.system.build.netbootRamdisk; }
{ name = "kernel"; path = netbootMinimalSystem.config.system.build.kernel; }
];
}; };
}; };
} }
+3 -17
View File
@@ -1,24 +1,10 @@
{ vars, ... }: _:
{ {
networking = { networking.hostName = "docker";
hostName = "docker"; networking.hostId = "007f0200";
hostId = "007f0200";
useDHCP = false;
interfaces = {
${vars.vmLanInterface}.ipv4.addresses = [{ address = vars.dockerIp; prefixLength = vars.lanPrefixLength; }];
${vars.lxcStorageInterface}.ipv4.addresses = [{ address = vars.dockerStorageIp; prefixLength = vars.haClientPrefixLength; }];
};
defaultGateway = { address = vars.lanGateway; interface = vars.vmLanInterface; };
nameservers = [ vars.domainControllerIp ];
};
boot.zfs.forceImportRoot = false; boot.zfs.forceImportRoot = false;
# Only advertise the LAN interface to IPA DNS. Without this, SSSD registers
# every Docker bridge (172.x.x.x) as an A record for docker.sweet.home —
# the default dyndns.interface = "*" catches them all.
security.ipa.dyndns.interface = vars.lxcLanInterface; # eth0
# Preserved from the pre-refactor `docker` target — stateVersion must never # Preserved from the pre-refactor `docker` target — stateVersion must never
# be bumped on an already-installed machine. # be bumped on an already-installed machine.
system.stateVersion = "25.05"; system.stateVersion = "25.05";
-34
View File
@@ -1,34 +0,0 @@
{ vars, ... }:
{
networking = {
hostName = vars.haDocker1Host;
hostId = "a1d0c4e1";
useDHCP = false;
interfaces = {
# ens18 — LAN management NIC (vmbr0, 192.168.2.0/24)
${vars.vmLanInterface}.ipv4.addresses = [{
address = vars.haDocker1Ip;
prefixLength = vars.lanPrefixLength;
}];
# ens19 — storage-client NIC (vmbr2, 192.168.20.0/24) — NFS from HA cluster
${vars.haDockerStorageInterface}.ipv4.addresses = [{
address = vars.haDocker1StorageIp;
prefixLength = vars.haClientPrefixLength;
}];
# ens20 — swarm cluster NIC (vmbr3, 192.168.30.0/24) — Docker gossip + VXLAN
${vars.haDockerSwarmInterface}.ipv4.addresses = [{
address = vars.haDocker1SwarmIp;
prefixLength = vars.haDockerSwarmPrefixLength;
}];
};
defaultGateway = { address = vars.lanGateway; interface = vars.vmLanInterface; };
nameservers = [ vars.domainControllerIp ];
};
# Only register the LAN IP with IPA DNS. Without this, sssd dyndns
# would also register Docker bridge IPs (172.x.x.x) and the storage/swarm
# NIC IPs as A records for ha-docker-1.sweet.home.
security.ipa.dyndns.interface = vars.vmLanInterface;
system.stateVersion = "26.05";
}
-32
View File
@@ -1,32 +0,0 @@
{ vars, ... }:
{
networking = {
hostName = vars.haDocker2Host;
hostId = "a2d0c4e2";
useDHCP = false;
interfaces = {
# ens18 — LAN management NIC (vmbr0, 192.168.2.0/24)
${vars.vmLanInterface}.ipv4.addresses = [{
address = vars.haDocker2Ip;
prefixLength = vars.lanPrefixLength;
}];
# ens19 — storage-client NIC (vmbr2, 192.168.20.0/24) — NFS from HA cluster
${vars.haDockerStorageInterface}.ipv4.addresses = [{
address = vars.haDocker2StorageIp;
prefixLength = vars.haClientPrefixLength;
}];
# ens20 — swarm cluster NIC (vmbr3, 192.168.30.0/24) — Docker gossip + VXLAN
${vars.haDockerSwarmInterface}.ipv4.addresses = [{
address = vars.haDocker2SwarmIp;
prefixLength = vars.haDockerSwarmPrefixLength;
}];
};
defaultGateway = { address = vars.lanGateway; interface = vars.vmLanInterface; };
nameservers = [ vars.domainControllerIp ];
};
# Only register the LAN IP with IPA DNS — same reasoning as ha-docker-1.
security.ipa.dyndns.interface = vars.vmLanInterface;
system.stateVersion = "26.05";
}
-17
View File
@@ -1,17 +0,0 @@
{ vars, ... }:
{
networking = {
hostName = vars.haServer1Host;
hostId = "3a4b5c6d";
useDHCP = false;
interfaces = {
${vars.vmLanInterface}.ipv4.addresses = [{ address = vars.haServer1Ip; prefixLength = vars.lanPrefixLength; }];
${vars.vmStorageInterface}.ipv4.addresses = [{ address = vars.haServer1StorageIp; prefixLength = vars.haStoragePrefixLength; }];
${vars.vmStorageClientInterface}.ipv4.addresses = [{ address = vars.haServer1ClientIp; prefixLength = vars.haClientPrefixLength; }];
};
defaultGateway = { address = vars.lanGateway; interface = vars.vmLanInterface; };
nameservers = [ vars.domainControllerIp ];
};
system.stateVersion = "26.05";
}
-17
View File
@@ -1,17 +0,0 @@
{ vars, ... }:
{
networking = {
hostName = vars.haServer2Host;
hostId = "7e8f9a0b";
useDHCP = false;
interfaces = {
${vars.vmLanInterface}.ipv4.addresses = [{ address = vars.haServer2Ip; prefixLength = vars.lanPrefixLength; }];
${vars.vmStorageInterface}.ipv4.addresses = [{ address = vars.haServer2StorageIp; prefixLength = vars.haStoragePrefixLength; }];
${vars.vmStorageClientInterface}.ipv4.addresses = [{ address = vars.haServer2ClientIp; prefixLength = vars.haClientPrefixLength; }];
};
defaultGateway = { address = vars.lanGateway; interface = vars.vmLanInterface; };
nameservers = [ vars.domainControllerIp ];
};
system.stateVersion = "26.05";
}
+12 -9
View File
@@ -1,15 +1,18 @@
{ vars, ... }: { vars, ... }:
{ {
networking = { imports = [
hostName = vars.nixCacheHost; (import ../../modules/beszel/host-token.nix {
useDHCP = false; name = "nix-cache";
interfaces.${vars.lxcLanInterface}.ipv4.addresses = [{ sopsFile = ../../secrets/nix-cache.yaml;
address = vars.nixCacheIp; })
prefixLength = vars.lanPrefixLength; ];
}];
defaultGateway = { address = vars.lanGateway; interface = vars.lxcLanInterface; }; networking.hostName = vars.nixCacheHost;
nameservers = [ vars.domainControllerIp ];
services.beszel.agent.environment = {
#DOCKER_HOST = "tcp://docker-socket-proxy:2375";
KEY = "ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIFPR9kwtC4TAeTRu46A7+opZsYpxqkRJ+x/ZyB2GWCeG";
}; };
# Preserved from the pre-refactor `nix-cache` target — stateVersion must # Preserved from the pre-refactor `nix-cache` target — stateVersion must
+1 -5
View File
@@ -19,15 +19,11 @@
nextcloud-client nextcloud-client
# vscode # vscode
chromium chromium
claude-code
fish
sops
]; ];
# Optional: set environment vars # Optional: set environment vars
sessionVariables = { sessionVariables = {
EDITOR = "nano"; EDITOR = "vim";
SOPS_AGE_KEY_FILE = "${config.home.homeDirectory}/.config/sops/age/keys.txt";
}; };
file = { file = {
-9
View File
@@ -1,17 +1,8 @@
_: _:
{ {
imports = [
../../modules/networking/wifi.nix
];
networking.hostName = "nixos"; networking.hostName = "nixos";
# Only needed now that baremetal-gui exists (ZFS root) -- harmless on the
# ext4-rooted linode/proxmox/lxc-gui variants, so set unconditionally
# rather than only on the baremetal platform.
networking.hostId = "de6a9ffc";
# Preserved from the pre-refactor `nixos` target — stateVersion must never # Preserved from the pre-refactor `nixos` target — stateVersion must never
# be bumped on an already-installed machine. # be bumped on an already-installed machine.
system.stateVersion = "25.05"; system.stateVersion = "25.05";
+3 -12
View File
@@ -1,17 +1,8 @@
{ vars, ... }: _:
{ {
networking = { networking.hostName = "pxe-boot";
hostName = "pxe-boot";
useDHCP = false;
interfaces.${vars.lxcLanInterface}.ipv4.addresses = [{
address = vars.pxeServerIp;
prefixLength = vars.lanPrefixLength;
}];
defaultGateway = { address = vars.lanGateway; interface = vars.lxcLanInterface; };
nameservers = [ vars.domainControllerIp ];
};
services.beszel.agent.environment = { };
# Preserved from the pre-refactor `pxe-boot` target — stateVersion must # Preserved from the pre-refactor `pxe-boot` target — stateVersion must
# never be bumped on an already-installed machine. # never be bumped on an already-installed machine.
system.stateVersion = "25.05"; system.stateVersion = "25.05";
+24
View File
@@ -0,0 +1,24 @@
{ vars, ... }:
{
imports = [
(import ../../modules/beszel/host-token.nix {
name = "server";
sopsFile = ../../secrets/server.yaml;
})
];
networking.hostName = vars.nfsServerHost;
networking.hostId = "6689f93e";
services.beszel.agent.environment = {
#DOCKER_HOST = "tcp://docker-socket-proxy:2375";
KEY = "ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIFPR9kwtC4TAeTRu46A7+opZsYpxqkRJ+x/ZyB2GWCeG";
EXTRA_FILESYSTEMS = "${vars.storageRoot}/${vars.nfsShares.dockerVolumes.subpath}";
LOG_LEVEL = "debug";
};
# Preserved from the pre-refactor `server` target — stateVersion must never
# be bumped on an already-installed machine.
system.stateVersion = "25.05";
}
+12
View File
@@ -0,0 +1,12 @@
_:
{
networking.hostName = "exit-node";
# No networking.hostId: only ZFS-touching hosts (server, docker) need one
# for pool-import safety, and this host does neither.
# A genuinely new host (not a pre-refactor carry-over), so it tracks the
# flake's current nixpkgs release rather than being pinned to an older one.
system.stateVersion = "26.05";
}
-19
View File
@@ -1,19 +0,0 @@
{ vars, ... }:
{
networking = {
hostName = "tailscale-router";
useDHCP = false;
interfaces.${vars.lxcLanInterface}.ipv4.addresses = [{
address = vars.tailscaleRouterIp;
prefixLength = vars.lanPrefixLength;
}];
defaultGateway = { address = vars.lanGateway; interface = vars.lxcLanInterface; };
nameservers = [ vars.domainControllerIp ];
};
# No networking.hostId: only ZFS-touching hosts (server, docker) need one
# for pool-import safety, and this host does neither.
system.stateVersion = "26.05";
}
+4 -13
View File
@@ -1,19 +1,10 @@
{ vars, ... }: _:
{ {
networking = { networking.hostName = "tor-relay";
hostName = "tor-relay";
useDHCP = false;
interfaces.${vars.lxcLanInterface}.ipv4.addresses = [{
address = vars.torRelayIp;
prefixLength = vars.lanPrefixLength;
}];
defaultGateway = { address = vars.lanGateway; interface = vars.lxcLanInterface; };
nameservers = [ vars.domainControllerIp ];
};
# No networking.hostId: only ZFS-touching hosts need one for pool-import # No networking.hostId: only ZFS-touching hosts (server, docker) need one
# safety, and this host does neither. # for pool-import safety, and this host does neither.
# A genuinely new host (not a pre-refactor carry-over), so it tracks the # A genuinely new host (not a pre-refactor carry-over), so it tracks the
# flake's current nixpkgs release rather than being pinned to an older one. # flake's current nixpkgs release rather than being pinned to an older one.
+5 -18
View File
@@ -1,23 +1,10 @@
{ config, vars, ... }: { vars, ... }:
{ {
# Universal token shared by all beszel agents. Add to secrets/common.yaml: services.beszel.agent.enable = true;
# sops secrets/common.yaml services.beszel.agent.environment = {
# beszel-token: <value from the beszel hub UI> #DOCKER_HOST = "tcp://docker-socket-proxy:2375";
sops.secrets."beszel-token" = { }; HUB_URL = "http://${vars.dockerHost}.${vars.homeDomain}:${toString vars.ports.beszelHub}";
sops.templates."beszel.env".content = ''
TOKEN=${config.sops.placeholder."beszel-token"}
'';
services.beszel.agent = {
enable = true;
environmentFile = config.sops.templates."beszel.env".path;
environment = {
#DOCKER_HOST = "tcp://docker-socket-proxy:2375";
HUB_URL = "http://${vars.dockerHost}.${vars.homeDomain}:${toString vars.ports.beszelHub}";
KEY = vars.beszelHubKey;
};
}; };
# The upstream module runs beszel-agent under DynamicUser with # The upstream module runs beszel-agent under DynamicUser with
+11
View File
@@ -0,0 +1,11 @@
{ name, sopsFile }:
{ config, ... }:
{
sops.secrets."beszel-token".sopsFile = sopsFile;
sops.templates."${name}-beszel.env".content = ''
TOKEN=${config.sops.placeholder."beszel-token"}
'';
services.beszel.agent.environmentFile = config.sops.templates."${name}-beszel.env".path;
}
+1
View File
@@ -15,6 +15,7 @@
../docker/enable-service.nix ../docker/enable-service.nix
../docker/nextcloud-cron-job.nix ../docker/nextcloud-cron-job.nix
../docker/docker-health-to-gotify.nix ../docker/docker-health-to-gotify.nix
../tailscale/enable-service.nix
../traefik/rotate-logs.nix ../traefik/rotate-logs.nix
../raspi/mount-data.nix ../raspi/mount-data.nix
../services/enable-rpcbind.nix ../services/enable-rpcbind.nix
+1 -78
View File
@@ -1,17 +1,6 @@
{ config, pkgs, lib, inputs, vars, ... }: { config, pkgs, lib, inputs, vars, ... }:
{ {
imports = [
../docker/enable-service.nix
];
nixpkgs.overlays = [
(final: prev: {
docker = prev.docker_29;
docker_cli = prev.docker_29;
})
];
environment.systemPackages = with pkgs; [ environment.systemPackages = with pkgs; [
inputs.nixos-conf-editor.packages.${pkgs.stdenv.hostPlatform.system}.nixos-conf-editor inputs.nixos-conf-editor.packages.${pkgs.stdenv.hostPlatform.system}.nixos-conf-editor
nodejs nodejs
@@ -29,7 +18,7 @@
]; ];
boot.loader.grub.useOSProber = true; boot.loader.grub.useOSProber = true;
programs.direnv.enable = true;
services = { services = {
xserver = { xserver = {
enable = true; enable = true;
@@ -81,70 +70,4 @@
programs.firefox.enable = true; programs.firefox.enable = true;
nixpkgs.config.allowUnfree = true; nixpkgs.config.allowUnfree = true;
# GUI-specific Home Manager additions for the IPA primary user, extending
# the baseline in modules/ipa/client.nix with desktop apps and services
# that only make sense on a graphical workstation.
home-manager.users.${vars.ipaUser} = { pkgs, ... }: {
home = {
packages = with pkgs; [
git
vim
nextcloud-client
chromium
claude-code
fish
sops
];
sessionVariables = {
EDITOR = "nano";
SOPS_AGE_KEY_FILE = "/home/${vars.ipaUser}/.config/sops/age/keys.txt";
};
file = {
".local/share/applications/proxmox-chromium-app.desktop".text = ''
[Desktop Entry]
Type=Application
Name=Proxmox (Chromium)
Exec=chromium --app=https://pve.${vars.homeDomain}:${toString vars.ports.pveWeb} --window-size=1920,1080 --window-position=0,0
Icon=/home/${vars.ipaUser}/.local/share/icons/proxmox.png
Terminal=false
Categories=Hypervisor;
StartupWMClass=PVE
'';
".local/share/applications/pbs-chromium-app.desktop".text = ''
[Desktop Entry]
Type=Application
Name=Proxmox Backup Server (Chromium)
Exec=chromium --app=https://${vars.pbsIp}:${toString vars.ports.pbsWeb} --window-size=1920,1080 --window-position=0,0
Icon=/home/${vars.ipaUser}/.local/share/icons/proxmox.png
Terminal=false
Categories=backup;
'';
".local/share/applications/proxmox-firefox-app.desktop".text = ''
[Desktop Entry]
Type=Application
Name=Proxmox (Firefox)
Exec=firefox --new-instance https://pve.${vars.homeDomain}:${toString vars.ports.pveWeb} --profile ProxmoxWebApp --window-size=1920,1080 --class ProxmoxWebApp
Icon=/home/${vars.ipaUser}/.local/share/icons/proxmox.png
Terminal=false
Categories=Hypervisor;
StartupWMClass=PVE
'';
".local/share/applications/pbs-firefox-app.desktop".text = ''
[Desktop Entry]
Type=Application
Name=Proxmox Backup Server (Firefox)
Exec=firefox --new-window https://${vars.pbsIp}:${toString vars.ports.pbsWeb} --profile PbsWebApp --window-size=1920,1080 --class PbsWebApp
Icon=/home/${vars.ipaUser}/.local/share/icons/proxmox.png
Terminal=false
Categories=backup;
StartupWMClass=PBS
'';
};
};
services.nextcloud-client = {
enable = true;
startInBackground = true;
};
};
} }
-84
View File
@@ -1,84 +0,0 @@
# Docker Swarm node build type.
#
# Produces NixOS hosts that form a Docker Swarm manager cluster. Two nodes
# (ha-docker-1, ha-docker-2) are both managers so either can accept Docker
# API and `docker stack` commands.
#
# Key differences from the existing `docker` build type (used by CT 105):
# - nextcloud-cron-job.nix is EXCLUDED — `docker exec` breaks in swarm
# because the target container may be on the other node. The cron job
# is replaced by a nextcloud-cron sidecar in the Nextcloud stack.
# See docs/internal/docker-swarm-cutover.md.
# - traefik/rotate-logs.nix is EXCLUDED — log rotation moves to Docker's
# json-file log driver (max-size/max-file on the Traefik service
# definition). See docs/internal/docker-swarm-cutover.md.
# - raspi/mount-data.nix is EXCLUDED — specific to CT 105's backup role.
# - Swarm firewall ports (2377/tcp, 7946/tcp+udp, 4789/udp) are opened
# on the swarm NIC (ens20/vmbr3) only.
# - checkReversePath = "loose" is required for the Swarm ingress routing
# mesh: VXLAN return traffic is asymmetric (arrives ens20, exits ens18).
# - beszel-agent is enabled for host-level monitoring.
{ pkgs, vars, ... }:
{
# Pin Docker Engine to version 29, matching CT 105, so image layers cached
# on NFS volumes remain compatible across old and new hosts.
nixpkgs.overlays = [
(final: prev: {
docker = prev.docker_29;
docker_cli = prev.docker_29;
})
];
imports = [
../docker/enable-service.nix
../docker/mount-data.nix
../docker/docker-health-to-gotify.nix
../beszel/enable-agent.nix
../services/enable-rpcbind.nix
];
environment.systemPackages = with pkgs; [
nfs-utils
];
boot.supportedFilesystems = [ "nfs" ];
systemd.tmpfiles.rules = [
# Symlink ~/docker → NFS config mount so the docker-health-to-gotify
# script (and operator convenience) resolves ~/docker/... correctly.
"L+ /home/${vars.primaryUser}/docker - - - - ${vars.nfsShares.dockerConfig.mountpoint}"
"d /mnt/docker 0755 ${vars.primaryUser} users -"
];
users.users.${vars.primaryUser}.extraGroups = [ "docker" ];
networking.firewall = {
# LAN-facing service ports — same as the existing docker build type.
allowedTCPPorts = [
vars.ports.dockerHttp
vars.ports.dockerHttps
vars.ports.dockerExtra
vars.ports.beszelHub
];
# Swarm inter-node ports restricted to the swarm NIC (ens20/vmbr3).
# vmbr3 is an isolated internal bridge — no LAN reachability.
interfaces.${vars.haDockerSwarmInterface} = {
allowedTCPPorts = [
vars.ports.dockerSwarmMgmt # 2377 — Raft + cluster management
vars.ports.dockerSwarmDisc # 7946 — Serf gossip (TCP half)
];
allowedUDPPorts = [
vars.ports.dockerSwarmDisc # 7946 — Serf gossip (UDP half)
vars.ports.dockerSwarmVxlan # 4789 — VXLAN overlay data path
];
};
# Docker Swarm ingress routing mesh creates asymmetric routes: a request
# arrives on ens18 (LAN) for a container that lives on ens20's VXLAN
# overlay; the return path differs from the incoming interface. Strict
# rp_filter drops these packets. "loose" allows them.
checkReversePath = "loose";
};
}
-54
View File
@@ -1,54 +0,0 @@
# HA file server build type: DRBD + XFS + LIO iSCSI + NFS, managed by
# Corosync + Pacemaker. Both ha-server-1 and ha-server-2 use this type.
#
# NFS start/stop:
# services.nfs.server.enable = true configures /etc/exports, wires up
# rpcbind, and loads kernel modules — but nfs-server.service.wantedBy is
# force-cleared so systemd does NOT auto-start it at boot. Pacemaker's
# ha-group resource group (configured by scripts/ha/cluster-init.sh)
# starts and stops nfs-server as part of the failover sequence after the
# XFS mount and iSCSI target are brought up on the new Active node.
#
# Beszel agent:
# Enabled here via enable-agent.nix. The agent KEY (used to pair with
# the Beszel hub) is not set yet — add it to hosts/ha-server-{1,2}/host.nix
# under services.beszel.agent.environment.KEY once the hub accepts the
# new agents, following the pattern in hosts/server/host.nix.
{ lib, pkgs, vars, ... }:
let
# Generates /etc/exports lines for all nfsShares data entries.
# LAN (VLAN 2): NFS via vip-lan (192.168.2.229) for pxe-boot and other LAN clients.
# Storage-client (VLAN 20): NFS via vip-storage (192.168.20.229) for docker and
# future swarm nodes; firewall restricts these ports to haClientCidr only.
mkNfsExports = storageRoot:
lib.concatMapStrings
(share:
" ${storageRoot}/${share.subpath} ${vars.lanCidr}${vars.nfsShares.options}\n" +
" ${storageRoot}/${share.subpath} ${vars.haClientCidr}${vars.nfsShares.options}\n")
(lib.filter builtins.isAttrs (lib.attrValues vars.nfsShares));
in
{
imports = [
../ha/pacemaker-stack.nix
../ha/iscsi-target.nix
../ha/cluster-config.nix
../beszel/enable-agent.nix
];
# xfsprogs: mkfs.xfs/xfs_info needed by cluster-init.sh.
# openiscsi: iscsiadm needed by acceptance-tests.sh T4 (iSCSI discovery check).
environment.systemPackages = [ pkgs.xfsprogs pkgs.openiscsi ];
services.nfs.server = {
enable = true;
exports = mkNfsExports vars.haStorageRoot;
};
# Pacemaker controls nfs-server — prevent systemd from starting it at boot
# on both nodes (only the Active node should be serving NFS).
systemd.services.nfs-server.wantedBy = lib.mkForce [ ];
# Same reason as server.nix: exports use standard auth, not Kerberos.
systemd.services.rpc-svcgssd.enable = false;
}
+36 -327
View File
@@ -6,10 +6,6 @@ let
tftpRoot = "${pxeRoot}/tftp"; tftpRoot = "${pxeRoot}/tftp";
pxeBaseUrl = "http://${vars.pxeServerIp}"; pxeBaseUrl = "http://${vars.pxeServerIp}";
# Base network address extracted from lanCidr (e.g. "192.168.2.0" from
# "192.168.2.0/24") — used by dnsmasq's proxy DHCP range directive.
lanBaseAddr = lib.head (lib.splitString "/" vars.lanCidr);
bootIpxe = pkgs.writeText "boot.ipxe" '' bootIpxe = pkgs.writeText "boot.ipxe" ''
#!ipxe #!ipxe
@@ -25,216 +21,6 @@ let
chain ${pxeBaseUrl}/boot.ipxe chain ${pxeBaseUrl}/boot.ipxe
''; '';
debianRelease = "bookworm";
debianMirror = "https://deb.debian.org/debian";
debianNetbootBase = "${debianMirror}/dists/${debianRelease}/main/installer-amd64/current/images/netboot/debian-installer/amd64";
rockyRelease = "9";
rockyArch = "x86_64";
rockyMirror = "https://dl.rockylinux.org/pub/rocky/${rockyRelease}";
rockyPxebootBase = "${rockyMirror}/BaseOS/${rockyArch}/os/images/pxeboot";
debianIpxe = pkgs.writeText "debian.ipxe" ''
#!ipxe
set base ${pxeBaseUrl}
kernel ''${base}/debian/linux
initrd ''${base}/debian/initrd.gz
boot
'';
fetchDebianNetboot = pkgs.writeShellScript "fetch-debian-netboot" ''
set -eu
dir="${httpRoot}/debian"
mirror="${debianNetbootBase}"
if [ -f "$dir/linux" ] && [ -f "$dir/initrd.gz" ]; then
echo "Debian ${debianRelease} netboot files already present; skipping download."
exit 0
fi
echo "Downloading Debian ${debianRelease} netboot kernel and initrd from $mirror ..."
${pkgs.curl}/bin/curl -fsSL -o "$dir/linux.tmp" "$mirror/linux"
${pkgs.curl}/bin/curl -fsSL -o "$dir/initrd.gz.tmp" "$mirror/initrd.gz"
mv "$dir/linux.tmp" "$dir/linux"
mv "$dir/initrd.gz.tmp" "$dir/initrd.gz"
echo "Debian ${debianRelease} netboot files staged."
'';
# Rocky Linux 9 iPXE script — boots vmlinuz+initrd.img from the staged
# /rocky/ directory and hands Anaconda the hosted Kickstart URL.
# net.ifnames=0 biosdevname=0 ensures the NIC is eth0 in both the
# installer and the installed system (matches the Kickstart NM config).
rockyFreeIpaIpxe = pkgs.writeText "rocky-freeipa.ipxe" ''
#!ipxe
set base ${pxeBaseUrl}
kernel ''${base}/rocky/vmlinuz inst.ks=''${base}/rocky-freeipa.ks inst.repo=${rockyMirror}/BaseOS/${rockyArch}/os/ net.ifnames=0 biosdevname=0 ip=dhcp quiet
initrd ''${base}/rocky/initrd.img
boot
'';
# Kickstart file for ${vars.ipaServer}.
# Installs Rocky Linux 9, sets a static IP, creates ${vars.ipaUser} with
# the admin SSH key, then on first reboot runs ipa-server-install via a
# systemd oneshot service. Passwords are generated at %post time, written
# to /root/ipa-credentials.txt (chmod 600), and read back by the
# first-boot script — never hardcoded here or in the repo.
rockyFreeIpaKs = pkgs.writeText "rocky-freeipa.ks" ''
#version=RHEL9
# Unattended Rocky Linux 9 + FreeIPA install
# Target: ${vars.ipaServer} ${vars.domainControllerIp}
url --url=${rockyMirror}/BaseOS/${rockyArch}/os/
repo --name=appstream --baseurl=${rockyMirror}/AppStream/${rockyArch}/os/
lang en_US.UTF-8
keyboard us
timezone UTC --utc
# DHCP during install; static IP configured in %post via NM config file
network --bootproto=dhcp --device=link --activate
network --hostname=${vars.ipaServer}
selinux --enforcing
firewall --enabled --service=ssh
rootpw --lock
user --name=${vars.ipaUser} --groups=wheel --shell=/bin/bash
sshkey --username=${vars.ipaUser} "${vars.adminSshKey}"
zerombr
clearpart --all --initlabel --drives=sda
# Keep net.ifnames=0 biosdevname=0 in the installed GRUB so the NIC
# stays eth0 after reboot (matches the NM connection file below).
bootloader --location=mbr --boot-drive=sda --append="net.ifnames=0 biosdevname=0"
part /boot --fstype=xfs --size=1024 --ondisk=sda
part swap --fstype=swap --size=2048 --ondisk=sda
part / --fstype=xfs --grow --size=1 --ondisk=sda --asprimary
%packages
@^minimal-environment
ipa-server
ipa-server-dns
%end
reboot
%post --log=/root/ks-post.log
set -euo pipefail
# -- Static IP: write NM connection file directly (NM not running in chroot) --
mkdir -p /etc/NetworkManager/system-connections
cat > /etc/NetworkManager/system-connections/eth0.nmconnection << 'NMCONN'
[connection]
id=eth0
type=ethernet
interface-name=eth0
autoconnect=true
[ethernet]
[ipv4]
method=manual
addresses=${vars.domainControllerIp}/${toString vars.lanPrefixLength}
gateway=${vars.lanGateway}
dns=${vars.domainControllerIp};
dns-search=${vars.homeDomain};
[ipv6]
method=auto
NMCONN
chmod 600 /etc/NetworkManager/system-connections/eth0.nmconnection
# -- /etc/hosts: FQDN must resolve to the real IP (not loopback) for IPA --
sed -i '/domain-controller/d' /etc/hosts
echo '${vars.domainControllerIp} ${vars.ipaServer} domain-controller' >> /etc/hosts
# -- Generate IPA passwords and store securely --
DM_PASS=$(openssl rand -base64 24 | tr -dc 'A-Za-z0-9' | head -c 24)
ADMIN_PASS=$(openssl rand -base64 24 | tr -dc 'A-Za-z0-9' | head -c 24)
printf 'Directory Manager: %s\nIPA Admin: %s\n' "$DM_PASS" "$ADMIN_PASS" \
> /root/ipa-credentials.txt
chmod 600 /root/ipa-credentials.txt
# -- First-boot script: reads passwords back, runs ipa-server-install --
cat > /usr/local/sbin/freeipa-first-boot.sh << 'FIRSTBOOT'
#!/bin/bash
set -euo pipefail
exec >> /root/freeipa-install.log 2>&1
echo "=== FreeIPA first-boot install started at $(date) ==="
DM_PASS=$(grep '^Directory Manager:' /root/ipa-credentials.txt | awk '{print $NF}')
ADMIN_PASS=$(grep '^IPA Admin:' /root/ipa-credentials.txt | awk '{print $NF}')
ipa-server-install \
--realm=${lib.strings.toUpper vars.homeDomain} \
--domain=${vars.homeDomain} \
--hostname=${vars.ipaServer} \
--ds-password="$DM_PASS" \
--admin-password="$ADMIN_PASS" \
--setup-dns \
--forwarder=${vars.domainControllerIp} \
--no-dnssec-validation \
--no-ntp \
--unattended
echo "=== FreeIPA install complete at $(date) ==="
echo "Credentials: /root/ipa-credentials.txt (save to password manager)"
echo "CA backup: /root/cacert.p12 (encrypted with Directory Manager password)"
systemctl disable freeipa-first-boot.service
FIRSTBOOT
chmod 700 /usr/local/sbin/freeipa-first-boot.sh
# -- Systemd oneshot service: runs freeipa-first-boot.sh on first real boot --
cat > /etc/systemd/system/freeipa-first-boot.service << 'UNIT'
[Unit]
Description=FreeIPA first-boot installation
After=network-online.target
Wants=network-online.target
ConditionPathExists=/root/ipa-credentials.txt
[Service]
Type=oneshot
ExecStart=/usr/local/sbin/freeipa-first-boot.sh
TimeoutStartSec=1800
RemainAfterExit=yes
[Install]
WantedBy=multi-user.target
UNIT
mkdir -p /etc/systemd/system/multi-user.target.wants
ln -sf /etc/systemd/system/freeipa-first-boot.service \
/etc/systemd/system/multi-user.target.wants/freeipa-first-boot.service
echo "Kickstart %post complete. FreeIPA installs on first reboot (~20 min)."
%end
'';
fetchRockyPxeboot = pkgs.writeShellScript "fetch-rocky-pxeboot" ''
set -eu
dir="${httpRoot}/rocky"
base="${rockyPxebootBase}"
if [ -f "$dir/vmlinuz" ] && [ -f "$dir/initrd.img" ]; then
echo "Rocky Linux ${rockyRelease} pxeboot files already present; skipping download."
exit 0
fi
echo "Downloading Rocky Linux ${rockyRelease} pxeboot kernel and initrd from $base ..."
${pkgs.curl}/bin/curl -fsSL -o "$dir/vmlinuz.tmp" "$base/vmlinuz"
${pkgs.curl}/bin/curl -fsSL -o "$dir/initrd.img.tmp" "$base/initrd.img"
mv "$dir/vmlinuz.tmp" "$dir/vmlinuz"
mv "$dir/initrd.img.tmp" "$dir/initrd.img"
echo "Rocky Linux ${rockyRelease} pxeboot files staged."
'';
systemRescueIpxe = pkgs.writeText "systemrescue.ipxe" '' systemRescueIpxe = pkgs.writeText "systemrescue.ipxe" ''
#!ipxe #!ipxe
@@ -282,27 +68,15 @@ let
set base ${pxeBaseUrl} set base ${pxeBaseUrl}
menu PXE Boot Menu menu PXE Boot Menu
item auto-installer NixOS Auto-Installer item nixos NixOS Installer
item nixos-minimal NixOS Minimal item rescue Rescue Environment
item debian Debian Minimal item shell iPXE Shell
item rocky-freeipa FreeIPA Server (Rocky Linux 9) item reboot Reboot
item rescue Rescue Environment
item shell iPXE Shell
item reboot Reboot
choose target && goto ''${target} choose target && goto ''${target}
:auto-installer :nixos
chain ''${base}/auto-installer/netboot.ipxe chain ''${base}/nixos/netboot.ipxe
:nixos-minimal
chain ''${base}/nixos-minimal/netboot.ipxe
:debian
chain ''${base}/debian.ipxe
:rocky-freeipa
chain ''${base}/rocky-freeipa.ipxe
:rescue :rescue
chain ''${base}/systemrescue.ipxe chain ''${base}/systemrescue.ipxe
@@ -317,8 +91,6 @@ in
{ {
imports = [ imports = [
../pxe-boot/stage-installer-artifacts.nix ../pxe-boot/stage-installer-artifacts.nix
../pxe-boot/mount-pxe-images.nix
../beszel/enable-agent.nix
]; ];
environment.systemPackages = with pkgs; [ environment.systemPackages = with pkgs; [
@@ -345,107 +117,44 @@ in
atftpd = { atftpd = {
enable = true; enable = true;
root = tftpRoot; root = tftpRoot;
extraOptions = [ "--verbose=5" ]; extraOptions = [
"--verbose=5"
];
}; };
openssh.settings.PermitRootLogin = "yes"; openssh.settings.PermitRootLogin = "yes";
}; };
systemd = { systemd.tmpfiles.rules = [
tmpfiles.rules = [ "d ${pxeRoot} 0755 root root -"
"d ${pxeRoot} 0755 root root -" "d ${httpRoot} 0755 root root -"
"d ${httpRoot} 0755 root root -" "d ${httpRoot}/images 0755 root root -"
"L+ ${httpRoot}/images - - - - ${vars.nfsShares.pxebootImages.mountpoint}" "d ${httpRoot}/nixos 0755 root root -"
"d ${httpRoot}/auto-installer 0755 root root -" "d ${httpRoot}/systemrescue 0755 root root -"
"d ${httpRoot}/nixos-minimal 0755 root root -" "d ${httpRoot}/ubuntu 0755 root root -"
"d ${httpRoot}/systemrescue 0755 root root -" "d ${httpRoot}/rescue 0755 root root -"
"d ${httpRoot}/debian 0755 root root -" "d ${tftpRoot} 0755 root root -"
"d ${httpRoot}/ubuntu 0755 root root -" "C+ ${httpRoot}/boot.ipxe 0644 root root - ${bootIpxe}"
"d ${httpRoot}/rescue 0755 root root -" "C+ ${httpRoot}/menu.ipxe 0644 root root - ${menuIpxe}"
"d ${httpRoot}/rocky 0755 root root -" "C+ ${httpRoot}/systemrescue.ipxe 0644 root root - ${systemRescueIpxe}"
"d ${tftpRoot} 0755 root root -" "C+ ${tftpRoot}/autoexec.ipxe 0644 root root - ${autoexecIpxe}"
"C+ ${httpRoot}/boot.ipxe 0644 root root - ${bootIpxe}" "C+ ${tftpRoot}/ipxe.efi 0644 root root - ${pkgs.ipxe}/ipxe.efi"
"C+ ${httpRoot}/menu.ipxe 0644 root root - ${menuIpxe}" "C+ ${tftpRoot}/undionly.kpxe 0644 root root - ${pkgs.ipxe}/undionly.kpxe"
"C+ ${httpRoot}/debian.ipxe 0644 root root - ${debianIpxe}" ];
"C+ ${httpRoot}/rocky-freeipa.ipxe 0644 root root - ${rockyFreeIpaIpxe}"
"C+ ${httpRoot}/rocky-freeipa.ks 0644 root root - ${rockyFreeIpaKs}" systemd.services.stage-systemrescue = {
"C+ ${httpRoot}/systemrescue.ipxe 0644 root root - ${systemRescueIpxe}" description = "Stage SystemRescue ISO contents for HTTP PXE boot";
"C+ ${tftpRoot}/autoexec.ipxe 0644 root root - ${autoexecIpxe}" after = [
"C+ ${tftpRoot}/ipxe.efi 0644 root root - ${pkgs.ipxe}/ipxe.efi" "local-fs.target"
"C+ ${tftpRoot}/undionly.kpxe 0644 root root - ${pkgs.ipxe}/undionly.kpxe" "systemd-tmpfiles-setup.service"
]; ];
wantedBy = [ "multi-user.target" ];
services = { serviceConfig = {
fetch-debian-netboot = { Type = "oneshot";
description = "Download Debian ${debianRelease} netboot kernel and initrd for HTTP PXE boot"; ExecStart = stageSystemRescue;
after = [
"local-fs.target"
"systemd-tmpfiles-setup.service"
"network-online.target"
];
wants = [ "network-online.target" ];
wantedBy = [ "multi-user.target" ];
serviceConfig = {
Type = "oneshot";
ExecStart = fetchDebianNetboot;
RemainAfterExit = true;
};
};
fetch-rocky-pxeboot = {
description = "Download Rocky Linux ${rockyRelease} pxeboot kernel and initrd for HTTP PXE boot";
after = [
"local-fs.target"
"systemd-tmpfiles-setup.service"
"network-online.target"
];
wants = [ "network-online.target" ];
wantedBy = [ "multi-user.target" ];
serviceConfig = {
Type = "oneshot";
ExecStart = fetchRockyPxeboot;
RemainAfterExit = true;
};
};
stage-systemrescue = {
description = "Stage SystemRescue ISO contents for HTTP PXE boot";
after = [
"local-fs.target"
"systemd-tmpfiles-setup.service"
];
wantedBy = [ "multi-user.target" ];
serviceConfig = {
Type = "oneshot";
ExecStart = stageSystemRescue;
};
};
};
};
services.dnsmasq = {
enable = true;
settings = {
# Disable DNS listener — only proxy DHCP is needed here.
# Without this dnsmasq tries to bind port 53 which systemd-resolved
# already owns, causing startup failure.
port = 0;
dhcp-range = [ "${lanBaseAddr},proxy" ];
dhcp-match = [
"set:ipxe,175"
"set:efi64,option:client-arch,7"
"set:efi64,option:client-arch,9"
];
dhcp-userclass = "set:ipxe,iPXE";
dhcp-boot = [
"tag:ipxe,tag:efi64,http://${vars.pxeServerIp}/boot.ipxe"
"tag:ipxe,http://${vars.pxeServerIp}/boot.ipxe"
"tag:efi64,ipxe.efi,,${vars.pxeServerIp}"
"undionly.kpxe,,${vars.pxeServerIp}"
];
}; };
}; };
networking.firewall.allowedTCPPorts = [ vars.ports.pxeBootHttp ]; networking.firewall.allowedTCPPorts = [ vars.ports.pxeBootHttp ];
networking.firewall.allowedUDPPorts = [ vars.ports.pxeBootTftp vars.ports.dhcp ]; networking.firewall.allowedUDPPorts = [ vars.ports.pxeBootTftp ];
} }
+28
View File
@@ -0,0 +1,28 @@
{ vars, lib, ... }:
{
imports = [
../beszel/enable-agent.nix
../services/zfs/enable-service.nix
];
boot.zfs.extraPools = [ (lib.removePrefix "/" vars.storageRoot) ];
systemd.services.nfs-server = {
after = [ "zfs-mount.service" ];
requires = [ "zfs-mount.service" ];
};
services.nfs.server = {
enable = true;
exports = ''
${vars.storageRoot}/${vars.nfsShares.dockerConfig.subpath} ${vars.lanCidr}(rw,sync,no_subtree_check,no_root_squash)
${vars.storageRoot}/${vars.nfsShares.dockerVolumes.subpath} ${vars.lanCidr}(rw,sync,no_subtree_check,no_root_squash)
${vars.storageRoot}/${vars.nfsShares.dockerDatabases.subpath} ${vars.lanCidr}(rw,sync,no_subtree_check,no_root_squash)
${vars.storageRoot}/${vars.nfsShares.nextcloudData.subpath} ${vars.lanCidr}(rw,sync,no_subtree_check,no_root_squash)
${vars.storageRoot}/${vars.nfsShares.raspiVolumes.subpath} ${vars.lanCidr}(rw,sync,no_subtree_check,no_root_squash)
'';
};
networking.firewall.allowedTCPPorts = [ vars.ports.nfsRpcbind vars.ports.nfsd ];
}
@@ -0,0 +1,21 @@
{ ... }:
{
imports = [
../tailscale/exit-node.nix
];
# "server", not "both": this build type only ever advertises itself as an
# exit node (see ../tailscale/exit-node.nix) -- it doesn't advertise LAN
# subnet routes, so it doesn't need the "client"-side loose reverse-path
# filtering that "both" would also turn on. Deliberately left unbundled
# from LAN-subnet-route advertisement so this build type stays valid on
# every platform, including linode (a remote VPS with no network path to
# the home LAN at all).
services.tailscale.useRoutingFeatures = "server";
# Forwarded exit-node traffic arrives on tailscale0 already
# tailscale-authenticated -- the firewall's normal per-port allow-list
# would otherwise drop it. Standard NixOS/Tailscale exit-node guidance.
networking.firewall.trustedInterfaces = [ "tailscale0" ];
}
-44
View File
@@ -1,44 +0,0 @@
{ vars, ... }:
{
imports = [
../tailscale/subnet-router.nix
../tailscale/ts-dns-forwarder.nix
../beszel/enable-agent.nix
];
# "server", not "both": this build type advertises LAN subnet routes but
# doesn't use another tailscale exit node itself, so it doesn't need the
# "client"-side loose reverse-path filtering that "both" would also enable.
# Deliberately kept explicit here (not just relying on subnet-router.nix's
# own setting) so the intent is clear at the build-type level.
services.tailscale.useRoutingFeatures = "server";
# Advertise the LAN subnet so Tailscale peers can route back to LAN machines.
# Must also be approved in the Tailscale admin console (Machines → Edit route settings).
services.tailscale.extraUpFlags = [ "--advertise-routes=${vars.lanCidr}" ];
networking.firewall = {
# Forwarded subnet-router traffic arrives on tailscale0 already
# tailscale-authenticated -- the firewall's normal per-port allow-list
# would otherwise drop it. Standard NixOS/Tailscale subnet-router guidance.
trustedInterfaces = [ "tailscale0" ];
# SNAT LAN traffic going into Tailscale so the remote peer sees it as
# coming from this router's Tailscale IP rather than a raw LAN IP.
# Without this, Tailscale drops forwarded packets whose source is not a
# recognised Tailscale address.
#
# We target POSTROUTING directly (always-existing built-in chain) rather
# than nixos-nat-post: extraCommands runs after the old nixos-nat-post is
# deleted but before the new one is created, so -A nixos-nat-post silently
# fails. The -C check makes the rule idempotent across firewall reloads.
extraCommands = ''
iptables -t nat -C POSTROUTING -s ${vars.lanCidr} -o tailscale0 -j MASQUERADE 2>/dev/null || \
iptables -t nat -A POSTROUTING -s ${vars.lanCidr} -o tailscale0 -j MASQUERADE
'';
extraStopCommands = ''
iptables -t nat -D POSTROUTING -s ${vars.lanCidr} -o tailscale0 -j MASQUERADE 2>/dev/null || true
'';
};
}
-1
View File
@@ -3,6 +3,5 @@
{ {
imports = [ imports = [
../tor/enable-relay.nix ../tor/enable-relay.nix
../beszel/enable-agent.nix
]; ];
} }
-30
View File
@@ -1,30 +0,0 @@
{ pkgs, ... }: {
# Defines the SSH host key as a clan vars generator so that:
# - `clan vars generate <target>` creates and encrypts the key pair
# - The private key lives at vars/per-machine/<target>/openssh/ssh_host_ed25519_key/secret
# (sops binary-encrypted, admin-key-only; decrypted by the build script)
# - The public key lives at vars/per-machine/<target>/openssh/ssh_host_ed25519_key.pub/value
# (plaintext; used by sync-host-keys.sh to derive the sops age fingerprint)
#
# neededFor = "activation" means clan's deployment tool would upload this
# before running nixos-rebuild/nixos-install (for VM/baremetal via
# nixos-anywhere). For lxc-* hosts, the build script bakes it into the
# tarball directly via NIXOS_HOST_KEYS_DIR -- the neededFor value here
# simply ensures it is NOT mapped to sops.secrets (which would try to
# decrypt it at runtime as a regular service secret, which is wrong: the
# SSH host key reaches the container via the tarball, not sops).
clan.core.vars.generators.openssh = {
files."ssh_host_ed25519_key" = {
secret = true;
neededFor = "activation";
};
files."ssh_host_ed25519_key.pub" = {
secret = false;
neededFor = "activation";
};
runtimeInputs = [ pkgs.openssh ];
script = ''
ssh-keygen -t ed25519 -N "" -C "" -f "$out/ssh_host_ed25519_key"
'';
};
}
+47 -4
View File
@@ -1,7 +1,50 @@
_: { config, pkgs, lib, vars, ... }:
let
# Flake attribute names are now <platform>-<buildtype> (e.g. proxmox-docker)
# and no longer match networking.hostName, since a host's hostname stays
# fixed while the platform backing it can change. Each nixosConfiguration
# stamps its own active target name into /etc/flake-target at build time.
mySwitchCmd = ''
sudo nixos-rebuild switch \
--no-write-lock-file \
--refresh \
--flake git+https://${vars.lanDomain}/beatzaplenty/nixos.git#$(cat /etc/flake-target)
'';
myTestCmd = ''
sudo nixos-rebuild test \
--no-write-lock-file \
--refresh \
--flake git+https://${vars.lanDomain}/beatzaplenty/nixos.git#$(cat /etc/flake-target)
'';
# lxc-* hosts pre-seed their SSH host key at build time (see
# modules/platforms/lxc.nix) so sops-nix's .sops.yaml recipient matches on
# first boot -- without it, secrets permanently fail to decrypt (see that
# file's comment for the confirmed failure). That requires --impure plus
# NIXOS_HOST_KEYS_DIR pointing at the repo's host-keys/ dir, same pattern
# docs/auto-installer.md uses for the installer ISO. A function, not a
# shellAlias, since the target name has to interpolate into the middle of
# the flake attribute path, not just append after it. Must be run from the
# repo root, same as every other host-keys/ command in this repo.
buildImageFn = ''
buildImage() {
if [ -z "$1" ]; then
echo "usage: buildImage <flake-target> (e.g. lxc-docker)" >&2
return 1
fi
NIXOS_HOST_KEYS_DIR="$(pwd)/host-keys" nix build --impure \
".#nixosConfigurations.$1.config.system.build.tarball"
}
'';
in
{ {
# Switch-nix, Test-nix, and buildImage are defined system-wide in programs.bash = {
# modules/common/configuration.nix so all users (including IPA accounts) enable = true;
# get them. Add any Home-Manager-only per-user shell config here. shellAliases = {
"Switch-nix" = mySwitchCmd;
"Test-nix" = myTestCmd;
};
initExtra = buildImageFn;
};
} }
+53 -71
View File
@@ -1,69 +1,44 @@
{ config, lib, pkgs, vars, ... }: { config, lib, pkgs, vars, ... }:
let
switchCmd = ''
sudo nixos-rebuild switch \
--no-write-lock-file \
--refresh \
--flake git+https://${vars.giteaDomain}/${vars.giteaRepoPath}.git?dir=${vars.giteaRepoFlakePath}#$(cat /etc/flake-target)
'';
testCmd = ''
sudo nixos-rebuild test \
--no-write-lock-file \
--refresh \
--flake git+https://${vars.giteaDomain}/${vars.giteaRepoPath}.git?dir=${vars.giteaRepoFlakePath}#$(cat /etc/flake-target)
'';
buildImageFn = ''
buildImage() {
if [ -z "$1" ]; then
echo "usage: buildImage <flake-target> (e.g. lxc-docker)" >&2
return 1
fi
NIXOS_HOST_KEYS_DIR="$(pwd)/host-keys" nix build --impure \
".#nixosConfigurations.$1.config.system.build.tarball"
}
'';
in
{ {
imports = [ imports =
./set-locale.nix [
../ipa/client.nix # Include the results of the hardware scan.
]; # ./hardware-configuration.nix
./set-locale.nix
];
# Use the GRUB 2 boot loader.
# boot.loader.grub.enable = true;
#boot.loader.grub.device = "/dev/sda"; # or "nodev" for efi only
# System-wide shell config so all users (including IPA accounts) get the networking.networkmanager.enable = true; # Easiest to use and most distros use this by default.
# same management aliases as the local nixos user's Home Manager provides.
programs.bash = {
shellAliases = {
"Switch-nix" = switchCmd;
"Test-nix" = testCmd;
};
interactiveShellInit = buildImageFn;
};
networking.networkmanager.enable = true;
# Recommended over the true default (bypasses ZFS's own import safeguards) # Recommended over the true default (bypasses ZFS's own import safeguards)
# per the option's own docs; matches hosts/docker/host.nix and # per the option's own docs; matches hosts/docker/host.nix and
# modules/services/zfs/enable-service.nix. Harmless no-op on hosts without ZFS. # modules/services/zfs/enable-service.nix, which already set this
# explicitly. Harmless no-op on hosts that don't use ZFS at all.
boot.zfs.forceImportRoot = false; boot.zfs.forceImportRoot = false;
# Set your time zone.
time.timeZone = vars.timeZone; time.timeZone = vars.timeZone;
# Enable QEMU agent
services.qemuGuest.enable = true; services.qemuGuest.enable = true;
# Enable docker-compose
environment.systemPackages = with pkgs; [ environment.systemPackages = with pkgs; [
vim vim
btop btop
git git
gcr gcr
jq
]; ];
# Secrets shared by every host, decrypted at activation via each host's # Secrets shared by every host, decrypted at activation via each host's
# SSH host key (sops-nix derives the age key from # existing SSH host key (sops-nix derives the age key from
# /etc/ssh/ssh_host_ed25519_key automatically). hashedPassword secrets need # /etc/ssh/ssh_host_ed25519_key automatically — see modules/common/README
# or docs/ for the sops workflow). hashedPassword/hashedPasswordFile need
# neededForUsers so they're available before the normal secret-activation # neededForUsers so they're available before the normal secret-activation
# step user creation happens very early in boot. # step, since user creation happens very early in boot.
sops = { sops = {
defaultSopsFile = ../../secrets/common.yaml; defaultSopsFile = ../../secrets/common.yaml;
@@ -71,52 +46,59 @@ in
"root-hashedPassword".neededForUsers = true; "root-hashedPassword".neededForUsers = true;
"nixos-hashedPassword".neededForUsers = true; "nixos-hashedPassword".neededForUsers = true;
"nix-github-token" = { }; "nix-github-token" = { };
"nix-gitea-token" = { };
}; };
# nix.conf has no *File-style option for access-tokens, so tokens are # nix.conf doesn't support a *File-style option for access-tokens, so the
# rendered into a runtime-only file (never touches the Nix store) and # token is rendered into a runtime-only file (never touches the Nix store)
# pulled in via nix.conf's native !include directive. # and pulled in via nix.conf's native !include directive.
templates."nix-access-tokens.conf".content = '' templates."nix-github-token.conf".content = ''
access-tokens = github.com=${config.sops.placeholder."nix-github-token"} ${vars.giteaDomain}=${config.sops.placeholder."nix-gitea-token"} access-tokens = github.com=${config.sops.placeholder."nix-github-token"}
''; '';
}; };
nix.extraOptions = '' nix.extraOptions = ''
!include ${config.sops.templates."nix-access-tokens.conf".path} !include ${config.sops.templates."nix-github-token.conf".path}
''; '';
users = { #Set root password
# mutableUsers = false makes update-users-groups.pl enforce hashedPasswordFile users.users.root = {
# on every activation, not just on newly-created accounts. Without this, a hashedPasswordFile = config.sops.secrets."root-hashedPassword".path;
# freshly-built proxmox disk image (activation runs without a usable sops key,
# so both accounts land in shadow with '!') will never have its passwords fixed
# by subsequent boots.
mutableUsers = false;
users.root = {
hashedPasswordFile = config.sops.secrets."root-hashedPassword".path;
};
users.${vars.primaryUser} = {
isNormalUser = true;
extraGroups = [ "wheel" ];
packages = with pkgs; [ tree ];
hashedPasswordFile = config.sops.secrets."nixos-hashedPassword".path;
openssh.authorizedKeys.keys = [ vars.adminSshKey ] ++ vars.extraAdminSshKeys;
};
}; };
# Define a user account. Don't forget to set a password with passwd.
users.users.${vars.primaryUser} = {
isNormalUser = true;
extraGroups = [ "wheel" ]; # Enable sudo for the user.
packages = with pkgs; [
tree
];
hashedPasswordFile = config.sops.secrets."nixos-hashedPassword".path;
openssh.authorizedKeys.keys = [
vars.adminSshKey
"ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAICMJhrfFayLBG+gWtO6oAvgambw5nWWgztiTFEaaaVRH debian@surface"
];
};
# Enable the OpenSSH daemon.
services.openssh.enable = true; services.openssh.enable = true;
#Enable flakes
nix.settings = { nix.settings = {
experimental-features = [ "nix-command" "flakes" ]; experimental-features = [ "nix-command" "flakes" ];
auto-optimise-store = true; auto-optimise-store = true;
}; };
programs.git = { programs.git = {
enable = true; enable = true;
package = pkgs.git; package = pkgs.git;
config.credential.helper = "store"; config = {
credential.helper = "store";
};
}; };
} }
-35
View File
@@ -1,35 +0,0 @@
# Shared activation-script logic to preserve the SSH host key across
# nixos-rebuild on platforms that embed the key via environment.etc (lxc and
# proxmox). When NIXOS_HOST_KEYS_DIR is not set the key is absent from
# environment.etc, and NixOS's etc activation removes any /etc file not in
# the new generation — which would destroy the live key and break sops-nix
# decryption permanently. These scripts save the key to /run before etc
# removes it, then restore it afterward.
#
# Explicit deps enforce the correct ordering: without them the topological
# sort places preserveSshHostKey after etc (confirmed live on lxc-tor-relay:
# position 7 vs etc's position 5), so the key is gone before it can be saved.
_: {
system.activationScripts = {
preserveSshHostKey = ''
if [ -f /etc/ssh/ssh_host_ed25519_key ]; then
cp /etc/ssh/ssh_host_ed25519_key /run/sshd-host-key-preserve.tmp
cp /etc/ssh/ssh_host_ed25519_key.pub /run/sshd-host-key-preserve.pub.tmp
fi
'';
restoreSshHostKey = {
deps = [ "etc" ];
text = ''
if [ ! -f /etc/ssh/ssh_host_ed25519_key ] && [ -f /run/sshd-host-key-preserve.tmp ]; then
install -m 0600 /run/sshd-host-key-preserve.tmp /etc/ssh/ssh_host_ed25519_key
install -m 0644 /run/sshd-host-key-preserve.pub.tmp /etc/ssh/ssh_host_ed25519_key.pub
fi
rm -f /run/sshd-host-key-preserve.tmp /run/sshd-host-key-preserve.pub.tmp
'';
};
etc = { deps = [ "preserveSshHostKey" ]; };
setupSecrets = { deps = [ "restoreSshHostKey" ]; };
};
}
-87
View File
@@ -1,87 +0,0 @@
{ vars, ... }:
{
# ZFS RAID0 (striped, no redundancy) root pool for the bare-metal gui
# host — two disks, each contributing its own top-level vdev. disko's
# zpool `mode` defaults to "" (plain stripe) when left unset, which is
# what gives RAID0 semantics here rather than mirror/raidz.
#
# Device paths are placeholders until the real hardware profile lands —
# fill in vars.guiRootDisk1/guiRootDisk2 (stable /dev/disk/by-id/...
# paths, not /dev/sdX) before running disko against real hardware. Swap
# is deliberately left out for now — sizing that sensibly needs the
# box's actual RAM size, which comes with the hardware profile too.
#
# Not yet imported anywhere: this awaits the new bare-metal platform
# module (alongside modules/boot/efi.nix for systemd-boot, matching
# modules/platforms/proxmox.nix's pattern) once the hardware config is
# in hand.
disko.devices = {
disk = {
disk1 = {
type = "disk";
device = vars.guiRootDisk1;
content = {
type = "gpt";
partitions = {
esp = {
priority = 1;
name = "ESP";
size = "512M";
type = "EF00";
content = {
type = "filesystem";
format = "vfat";
mountpoint = "/boot";
mountOptions = [ "umask=0077" ];
};
};
zfs = {
size = "100%";
content = {
type = "zfs";
pool = "rpool";
};
};
};
};
};
disk2 = {
type = "disk";
device = vars.guiRootDisk2;
content = {
type = "gpt";
partitions = {
zfs = {
size = "100%";
content = {
type = "zfs";
pool = "rpool";
};
};
};
};
};
};
zpool.rpool = {
type = "zpool";
rootFsOptions = {
compression = "zstd";
"com.sun:auto-snapshot" = "false";
};
mountpoint = "/";
options.ashift = "12";
};
};
}
+12 -34
View File
@@ -1,45 +1,23 @@
{ lib, pkgs, vars, ... }: { pkgs, ... }:
let
gid = toString vars.dockerAccessGid;
in
{ {
# virtualisation.docker.enable = true;
virtualisation.docker = { virtualisation.docker = {
enable = true; enable = true;
package = pkgs.docker; package = pkgs.docker;
# listenOptions = [
# "unix:///var/run/docker.sock"
# "tcp://0.0.0.0:2375"
#];
# daemon.settings = {
# metrics-addr = "0.0.0.0:9323";
# experimental = true;
# };
}; };
# Pin the docker group GID to match the IPA "docker-access" group so that
# IPA group membership alone grants access to the Docker socket. Any user
# whose supplementary groups (resolved by SSSD from IPA) include GID
# vars.dockerAccessGid will pass the socket group-permission check without
# any per-host users.groups.docker.members entry.
users.groups.docker.gid = lib.mkForce vars.dockerAccessGid;
users.users.${vars.primaryUser}.extraGroups = [ "docker" ];
environment.systemPackages = with pkgs; [ environment.systemPackages = with pkgs; [
docker-compose docker-compose
docker-buildx docker-buildx
]; ];
# NixOS's group activation uses plain `groupmod` without --non-unique.
# When SSSD is active it exposes the IPA "docker-access" group at
# vars.dockerAccessGid via NSS, so groupmod sees that GID as already in
# use and silently skips the change (warning: "not applying GID change").
# This script runs after the normal "groups" step and applies the change
# with --non-unique (which lets the local docker group share the GID with
# the SSSD-provided IPA group). If the GID actually changed it also
# restarts docker.socket so the socket is recreated with the new GID.
system.activationScripts.docker-group-gid = {
deps = [ "groups" ];
text = ''
current=$(grep "^docker:" /etc/group | cut -d: -f3)
if [ "$current" != "${gid}" ]; then
${pkgs.shadow}/bin/groupmod --non-unique -g ${gid} docker
if ${pkgs.systemd}/bin/systemctl is-active --quiet docker.socket; then
${pkgs.systemd}/bin/systemctl stop docker.service docker.socket
rm -f /var/run/docker.sock
${pkgs.systemd}/bin/systemctl start docker.socket docker.service
fi
fi
'';
};
} }
+18 -13
View File
@@ -10,19 +10,24 @@ let
# non-blocking behavior, so they don't need `nofail` too). # non-blocking behavior, so they don't need `nofail` too).
automountOpts = if config.boot.isContainer then [ "nofail" ] else [ "x-systemd.automount" ]; automountOpts = if config.boot.isContainer then [ "nofail" ] else [ "x-systemd.automount" ];
# FQDN in the storage.home zone — resolves to the Pacemaker vip-storage # A bare hostname here never resolves reliably: systemd-resolved only
# (192.168.20.229) on docker's eth1/vmbr2 interface. Using the DNS name # ever tries LLMNR for single-label names (never DNS, regardless of any
# rather than the raw IP means a future VIP renumber only requires a DNS # configured search domain), and a *global* search domain (the first fix
# update, not a NixOS rebuild. The storage.home zone is served by the same # attempted here) backfires worse -- confirmed live on lxc-docker, adding
# FreeIPA nameserver (domainControllerIp) that docker already uses, so # `networking.search` made systemd-resolved prioritize its domain-matched
# resolution reaches it over eth0 without any extra routing. # but server-less global scope over eth0's correctly-configured one for
nfsServer = vars.haStorageNfsFqdn; # every "*.sweet.home" query, silently sending them to public fallback
storageRoot = vars.haStorageRoot; # DNS instead. `resolvectl query --interface=eth0 server.sweet.home`
# resolved fine throughout, proving the LAN DNS server was never the
# problem -- only the ambient, unqualified device string was. Using the
# FQDN directly sidesteps all of that, matching the pattern
# ../raspi/mount-data.nix already uses for the same reason.
nfsServer = "${vars.nfsServerHost}.${vars.homeDomain}";
in in
{ {
fileSystems = { fileSystems = {
${vars.nfsShares.dockerConfig.mountpoint} = { ${vars.nfsShares.dockerConfig.mountpoint} = {
device = "${nfsServer}:${storageRoot}/${vars.nfsShares.dockerConfig.subpath}"; device = "${nfsServer}:${vars.storageRoot}/${vars.nfsShares.dockerConfig.subpath}";
fsType = "nfs"; fsType = "nfs";
options = [ options = [
@@ -33,7 +38,7 @@ in
}; };
${vars.nfsShares.dockerDatabases.mountpoint} = { ${vars.nfsShares.dockerDatabases.mountpoint} = {
device = "${nfsServer}:${storageRoot}/${vars.nfsShares.dockerDatabases.subpath}"; device = "${nfsServer}:${vars.storageRoot}/${vars.nfsShares.dockerDatabases.subpath}";
fsType = "nfs"; fsType = "nfs";
options = [ options = [
@@ -44,7 +49,7 @@ in
}; };
${vars.nfsShares.dockerVolumes.mountpoint} = { ${vars.nfsShares.dockerVolumes.mountpoint} = {
device = "${nfsServer}:${storageRoot}/${vars.nfsShares.dockerVolumes.subpath}"; device = "${nfsServer}:${vars.storageRoot}/${vars.nfsShares.dockerVolumes.subpath}";
fsType = "nfs"; fsType = "nfs";
options = [ options = [
@@ -55,7 +60,7 @@ in
}; };
${vars.nfsShares.nextcloudData.mountpoint} = { ${vars.nfsShares.nextcloudData.mountpoint} = {
device = "${nfsServer}:${storageRoot}/${vars.nfsShares.nextcloudData.subpath}"; device = "${nfsServer}:${vars.storageRoot}/${vars.nfsShares.nextcloudData.subpath}";
fsType = "nfs"; fsType = "nfs";
options = [ options = [
@@ -66,7 +71,7 @@ in
}; };
${vars.nfsShares.raspiVolumes.mountpoint} = { ${vars.nfsShares.raspiVolumes.mountpoint} = {
device = "${nfsServer}:${storageRoot}/${vars.nfsShares.raspiVolumes.subpath}"; device = "${nfsServer}:${vars.storageRoot}/${vars.nfsShares.raspiVolumes.subpath}";
fsType = "nfs"; fsType = "nfs";
options = [ options = [
-164
View File
@@ -1,164 +0,0 @@
# Cluster-wide HA config shared by both ha-server nodes.
#
# Covers everything that is identical on both nodes and references cluster
# topology (node IPs, hostnames, DRBD resource). Per-node identity
# (hostname, static IP, stateVersion) lives in hosts/ha-server-{1,2}/host.nix.
#
# Corosync authkey:
# /etc/corosync/authkey (mode 0400) is managed by sops-nix below.
# Bootstrap: run scripts/ha/cluster-init.sh on node1 to generate the key,
# then encrypt it with: sops -e --input-type binary /etc/corosync/authkey > secrets/ha-corosync-authkey
# Both host keys must be registered via sync-host-keys.sh first so both nodes can decrypt it.
#
# DRBD fencing:
# resource-only with crm-fence-peer.sh: DRBD calls the Pacemaker-aware
# crm-fence-peer.sh handler before promoting. The handler checks the CIB
# to confirm the peer's DRBD resource is stopped and returns 7 (successfully
# fenced), allowing safe promotion without requiring power-fencing (STONITH).
# The unfence handler crm-unfence-peer.sh clears the outdate flag when the
# peer reconnects. This is the correct setting for Pacemaker+DRBD clusters
# with STONITH disabled; crm-fence-peer.sh replaces the need for a separate
# STONITH device during the testing phase. Switch to resource-and-stonith
# once the fence_pve_ssh STONITH resource is active (see
# scripts/ha/cluster-enable-stonith.sh).
#
# PATH wrapper: when the DRBD kernel module invokes the fence-peer handler
# via the UMH (User Mode Helper) mechanism it provides a minimal PATH that
# omits /run/current-system/sw/bin. crm-fence-peer.sh calls cibadmin,
# crm_mon etc.; if those aren't found a pipeline in the script breaks with
# SIGPIPE. A process killed by signal has WEXITSTATUS() == 0, so the kernel
# sees exit code 0 and logs "fence-peer helper broken, returned 0", looping
# forever. The writeShellScript wrappers below prepend the NixOS sw path
# before exec-ing the real handler, giving it a working Pacemaker toolchain.
{ lib, pkgs, vars, ... }:
let
fencePeerWrapper = pkgs.writeShellScript "drbd-fence-peer" ''
export PATH="/run/current-system/sw/bin:/run/current-system/sw/sbin:$PATH"
exec /run/current-system/sw/lib/drbd/crm-fence-peer.sh "$@"
'';
unfencePeerWrapper = pkgs.writeShellScript "drbd-unfence-peer" ''
export PATH="/run/current-system/sw/bin:/run/current-system/sw/sbin:$PATH"
exec /run/current-system/sw/lib/drbd/crm-unfence-peer.sh "$@"
'';
in
{
# Root SSH access — same key set as the nixos user so all admin keys can reach root.
users.users.root.openssh.authorizedKeys.keys = [ vars.adminSshKey ] ++ vars.extraAdminSshKeys;
# Passwordless sudo for wheel — operator SSHes as nixos and uses sudo for
# cluster management commands (drbdadm, crm*, pcs, etc.)
security.sudo.wheelNeedsPassword = lib.mkForce false;
# DRBD lock-file directory (drbd-utils checks for it; missing → harmless but noisy warnings).
systemd.tmpfiles.rules = [ "d /var/lib/drbd 0750 root root -" ];
# Prevent drbd.service from auto-starting at boot / nixos-rebuild switch.
# Pacemaker's OCF drbd agent calls drbdadm up/down directly when managing
# the resource. If drbd.service also runs drbdadm up all while DRBD is
# already Primary under Pacemaker, apply-al fails with "device busy" (exit 20).
systemd.services.drbd.wantedBy = lib.mkForce [ ];
services.drbd = {
enable = true;
config = ''
global {
usage-count yes;
}
common {
net {
protocol C;
ping-int 1;
verify-alg sha256;
after-sb-0pri discard-zero-changes;
after-sb-1pri discard-secondary;
}
disk {
fencing resource-only;
}
handlers {
fence-peer "${fencePeerWrapper}";
unfence-peer "${unfencePeerWrapper}";
}
}
resource ha-data {
volume 0 {
device /dev/drbd0;
disk ${vars.haServerDrbdDisk};
meta-disk internal;
}
on ${vars.haServer1Host} {
address ${vars.haServer1StorageIp}:${toString vars.ports.haServerDrbd};
}
on ${vars.haServer2Host} {
address ${vars.haServer2StorageIp}:${toString vars.ports.haServerDrbd};
}
}
'';
};
# /etc/corosync/authkey — sops binary secret, identical on both nodes.
# Decryptable by both ha-server host keys (added by sync-host-keys.sh).
sops.secrets.corosync_authkey = {
sopsFile = ../../secrets/ha-corosync-authkey;
format = "binary";
path = "/etc/corosync/authkey";
mode = "0400";
restartUnits = [ "corosync.service" ];
};
# NixOS common config enables NetworkManager by default; HA cluster nodes
# need stable static IPs with predictable interface names — NM is not suitable.
networking.networkmanager.enable = lib.mkForce false;
# services.corosync.enable is set by modules/ha/pacemaker-stack.nix.
services.corosync = {
clusterName = "ha-cluster";
nodelist = [
# ring0: cluster-internal vmbr1 (primary heartbeat + DRBD path)
# ring1: LAN vmbr0 (backup heartbeat only — never carries DRBD)
{ nodeid = 1; name = vars.haServer1Host; ring_addrs = [ vars.haServer1StorageIp vars.haServer1Ip ]; }
{ nodeid = 2; name = vars.haServer2Host; ring_addrs = [ vars.haServer2StorageIp vars.haServer2Ip ]; }
];
};
networking.firewall = {
allowedTCPPorts = [
vars.ports.haServerPacemakerRemoted
vars.ports.haServerPcsd
vars.ports.haServerDrbd
];
allowedUDPPorts = [
vars.ports.haServerCorosync1
vars.ports.haServerCorosync2
vars.ports.haServerCorosyncCrypto
];
# Protocol separation: iSCSI (VLAN 20 / storage clients only),
# NFS (VLAN 2 / LAN only). Cluster-internal subnets accepted wholesale
# since they are isolated bridges with no external uplink.
extraCommands = ''
iptables -A nixos-fw -s ${vars.haServer1Ip}/32 -j nixos-fw-accept
iptables -A nixos-fw -s ${vars.haServer2Ip}/32 -j nixos-fw-accept
iptables -A nixos-fw -s ${vars.haStorageCidr} -j nixos-fw-accept
iptables -A nixos-fw -p tcp -s ${vars.haClientCidr} --dport ${toString vars.ports.haServerIscsi} -j nixos-fw-accept
iptables -A nixos-fw -p tcp -s ${vars.lanCidr} --dport ${toString vars.ports.nfsRpcbind} -j nixos-fw-accept
iptables -A nixos-fw -p udp -s ${vars.lanCidr} --dport ${toString vars.ports.nfsRpcbind} -j nixos-fw-accept
iptables -A nixos-fw -p tcp -s ${vars.lanCidr} --dport ${toString vars.ports.nfsd} -j nixos-fw-accept
iptables -A nixos-fw -p udp -s ${vars.lanCidr} --dport ${toString vars.ports.nfsd} -j nixos-fw-accept
iptables -A nixos-fw -p tcp -s ${vars.lanCidr} --dport ${toString vars.ports.nfsMountd} -j nixos-fw-accept
iptables -A nixos-fw -p udp -s ${vars.lanCidr} --dport ${toString vars.ports.nfsMountd} -j nixos-fw-accept
iptables -A nixos-fw -p tcp -s ${vars.haClientCidr} --dport ${toString vars.ports.nfsRpcbind} -j nixos-fw-accept
iptables -A nixos-fw -p udp -s ${vars.haClientCidr} --dport ${toString vars.ports.nfsRpcbind} -j nixos-fw-accept
iptables -A nixos-fw -p tcp -s ${vars.haClientCidr} --dport ${toString vars.ports.nfsd} -j nixos-fw-accept
iptables -A nixos-fw -p udp -s ${vars.haClientCidr} --dport ${toString vars.ports.nfsd} -j nixos-fw-accept
iptables -A nixos-fw -p tcp -s ${vars.haClientCidr} --dport ${toString vars.ports.nfsMountd} -j nixos-fw-accept
iptables -A nixos-fw -p udp -s ${vars.haClientCidr} --dport ${toString vars.ports.nfsMountd} -j nixos-fw-accept
'';
};
}
-99
View File
@@ -1,99 +0,0 @@
# LIO iSCSI target service (targetctl) for NixOS HA clusters.
#
# Provides the targetctl.service that saves/restores LIO configuration from
# /etc/target/saveconfig.json. Pacemaker manages this service via its
# systemd resource agent (class="systemd" type="targetctl").
#
# Why ExecStop is not simply "targetctl save":
# targetctl save writes the LIO config to JSON but does NOT remove the LIO
# target from the kernel's configfs. As a result, any fileio backing store
# that LIO has open (e.g. iscsi-lun.img on an XFS-over-DRBD filesystem)
# stays referenced in the kernel. The subsequent XFS umount from the
# Filesystem OCF resource then returns EBUSY and either hangs for the full
# op-stop timeout or fails outright, blocking the entire failover.
#
# The ExecStop script here additionally tears down the kernel LIO state
# via rtslib_fb after saving, so the backing-store file descriptor is
# released and umount succeeds immediately.
#
# Empty-config guard:
# The save step is skipped when no iSCSI targets are currently active.
# This prevents the secondary node (where LIO was never started) from
# overwriting a valid saveconfig.json with an empty one when Pacemaker
# stops the iscsi-target resource as part of a failover or cleanup.
{ pkgs, ... }:
let
python3 = pkgs.python3.withPackages (ps: [ ps.rtslib-fb ]);
targetctl = "${python3}/bin/targetctl";
targetctlStop = pkgs.writeScript "targetctl-stop" ''
#!${python3}/bin/python3
import subprocess, sys
import rtslib_fb
root = rtslib_fb.RTSRoot()
targets = list(root.targets)
if targets:
subprocess.run(
["${targetctl}", "save", "/etc/target/saveconfig.json"],
capture_output=True,
)
print(f"saved {len(targets)} iSCSI target(s)")
else:
print("no active LIO targets saveconfig.json unchanged")
for target in targets:
try:
for tpg in list(target.tpgs):
tpg.enable = False
target.delete()
except Exception as e:
print(f"warn (target): {e}", file=sys.stderr)
for so in list(root.storage_objects):
try:
so.delete()
except Exception as e:
print(f"warn (backstore): {e}", file=sys.stderr)
print("LIO kernel target cleared")
'';
in
{
boot.kernelModules = [
"target_core_mod"
"iscsi_target_mod"
"target_core_file"
"target_core_pscsi"
"target_core_user"
"configfs"
];
systemd = {
mounts = [{
where = "/sys/kernel/config";
what = "configfs";
type = "configfs";
wantedBy = [ "multi-user.target" ];
before = [ "targetctl.service" ];
}];
services.targetctl = {
description = "LIO iSCSI target config save/restore";
wantedBy = [ "multi-user.target" ];
after = [ "sys-kernel-config.mount" "network.target" ];
requires = [ "sys-kernel-config.mount" ];
serviceConfig = {
Type = "oneshot";
RemainAfterExit = true;
ExecStart = "${targetctl} restore /etc/target/saveconfig.json";
ExecStop = "${targetctlStop}";
};
unitConfig.ConditionFileNotEmpty = "/etc/target/saveconfig.json";
};
tmpfiles.rules = [
"d /etc/target 0750 root root -"
"f /etc/target/saveconfig.json 0640 root root -"
];
};
environment.systemPackages = [ pkgs.targetcli-fb ];
}
-94
View File
@@ -1,94 +0,0 @@
# Pacemaker + Corosync HA stack for NixOS with known-good workarounds.
#
# Issues fixed here (confirmed through live testing on NixOS 25.11):
#
# 1. StateDirectory ownership reset: systemd's StateDirectory=pacemaker
# creates /var/lib/pacemaker owned root:root. pacemaker-based (the CIB
# daemon) runs as the hacluster user and calls pcmk__daemon_can_write,
# which requires the CIB directory to be owned by hacluster or be
# group-writable by haclient. Workaround: remove StateDirectory and let
# ExecStartPre create every required subdirectory with correct ownership.
#
# 2. HA_SBIN_DIR wrong path: ocf-shellfuncs sets HA_SBIN_DIR to the Nix
# store path of the resource-agents derivation's /sbin, which doesn't
# exist. The DRBD OCF agent uses ${HA_SBIN_DIR}/crm_master, so it exits
# 127 without this override. Fix: export HA_SBIN_DIR=/run/current-system/sw/bin.
#
# 3. Broad PATH for OCF agents: the resource executor (pacemaker-execd) runs
# OCF agent scripts as children. NixOS provides no implicit PATH for
# system services; without an explicit PATH the agents can't find ip, ss,
# mount, umount, drbdadm, etc.
#
# 4. FUSER=true: the Filesystem OCF agent calls check_binary $FUSER (default:
# fuser from psmisc), which is not installed. Setting FUSER=true makes
# check_binary succeed (true is always in PATH) and the subsequent
# "$FUSER -km $mountpoint" becomes a no-op. Pair with force_unmount=false
# on each Filesystem resource unless you want lazy unmount behaviour.
{ lib, pkgs, ... }:
let
ocfBinPath = lib.concatStringsSep ":" [
"${pkgs.iproute2}/bin"
"${pkgs.iproute2}/sbin"
"${pkgs.iputils}/bin"
"${pkgs.util-linux}/bin"
"${pkgs.util-linux}/sbin"
"${pkgs.gawk}/bin"
"${pkgs.gnugrep}/bin"
"${pkgs.gnused}/bin"
"${pkgs.coreutils}/bin"
"${pkgs.bash}/bin"
"${pkgs.procps}/bin"
"${pkgs.xfsprogs}/bin"
"${pkgs.drbd}/bin"
"${pkgs.python3}/bin"
"/run/current-system/sw/bin"
"/run/current-system/sw/sbin"
"/usr/local/sbin"
"/usr/local/bin"
"/usr/sbin"
"/usr/bin"
"/sbin"
"/bin"
];
# Single pre-start script: schemas symlink + directory ownership.
# Runs before pacemakerd so pacemaker-based finds hacluster-owned dirs.
preStartCmd = "${pkgs.bash}/bin/bash -c '"
+ "ln -sfn ${pkgs.pacemaker}/share/pacemaker /var/lib/pacemaker/schemas; "
+ "for d in /var/lib/pacemaker /var/lib/pacemaker/cib /var/lib/pacemaker/cores "
+ "/var/lib/pacemaker/pengine /var/lib/pacemaker/blackbox "
+ "/var/lib/pacemaker/hostcache; do "
+ "mkdir -p \"\\$d\" && chown hacluster:pacemaker \"\\$d\" && chmod 2770 \"\\$d\"; "
+ "done'";
ocfEnv = {
PATH = lib.mkForce ocfBinPath;
OCF_ROOT = "${pkgs.ocf-resource-agents}/usr/lib/ocf";
HA_SBIN_DIR = "/run/current-system/sw/bin";
FUSER = "true";
};
in
{
users.groups.haclient = { };
services.corosync.enable = true;
services.pacemaker.enable = true;
systemd.services = {
pacemaker = {
serviceConfig = {
StateDirectory = lib.mkForce "";
ExecStartPre = lib.mkBefore [ preStartCmd ];
};
environment = ocfEnv;
};
pacemaker-execd.environment = ocfEnv;
};
environment.systemPackages = with pkgs; [
corosync
pacemaker
ocf-resource-agents
];
}
@@ -1,23 +0,0 @@
# Adapted from the output of `nixos-generate-config`, run from a live GUI
# ISO boot on the actual gui-host hardware (AMD CPU). fileSystems and
# swapDevices are deliberately omitted -- the live ISO had no formatted
# disks to detect, and disko (modules/disko/baremetal.nix) generates both
# from the declarative zpool layout anyway.
{ config, lib, pkgs, modulesPath, ... }:
{
imports =
[
(modulesPath + "/installer/scan/not-detected.nix")
];
boot = {
initrd.availableKernelModules = [ "xhci_pci" "ahci" "usbhid" "usb_storage" "sd_mod" ];
initrd.kernelModules = [ ];
kernelModules = [ "kvm-amd" ];
extraModulePackages = [ ];
};
nixpkgs.hostPlatform = lib.mkDefault "x86_64-linux";
hardware.cpu.amd.updateMicrocode = lib.mkDefault config.hardware.enableRedistributableFirmware;
}
+134 -15
View File
@@ -45,21 +45,140 @@
disko disko
]; ];
# Auto-install script, kept as a real, version-controlled shell file at # Write auto-install script to /root
# scripts/installer/auto-install.sh rather than an inline Nix string. etc."auto-install.sh" = {
# It sources scripts/env.sh itself (for LAN_DOMAIN, same as every other text = ''
# script in this repo) rather than relying on Nix-level templating, so #!/run/current-system/sw/bin/bash
# it behaves identically whether it's run straight from a git checkout set -eux
# or from here -- baking scripts/env.sh in alongside it at a matching
# relative path (installer/auto-install.sh -> ../env.sh) is what makes
# that resolve correctly in both places.
etc = {
"nixos-installer/env.sh".source = ../../scripts/env.sh;
"nixos-installer/installer/auto-install.sh" = { set -euo pipefail
source = ../../scripts/installer/auto-install.sh;
mode = "0755"; export FLAKE_BASE_URL="git+https://${vars.lanDomain}/beatzaplenty/nixos.git"
};
echo "Fetching available NixOS hosts from flake..."
# Two categories deliberately excluded from the menu:
# lxc-* these build a config.system.build.tarball meant for
# `pct restore` on Proxmox directly, not an install.
# Running nixos-install against one here would
# bind-mount / onto /mnt and then refuse to touch the
# filesystem it's currently running on see
# docs/auto-installer.md.
# installer this *is* the installer image's own flake target,
# not a deployable host; "installing" it means
# nixos-install-ing a copy of the installer into
# itself.
mapfile -t options < <(
nix eval --json --no-use-registries --no-accept-flake-config --extra-experimental-features "flakes nix-command" \
"''${FLAKE_BASE_URL}#nixosConfigurations" \
--apply builtins.attrNames \
| jq -r '.[]
| select(startswith("lxc-") | not)
| select(. != "installer")'
)
if [[ ''${#options[@]} -eq 0 ]]; then
echo "ERROR: No NixOS hosts found in ''${FLAKE_BASE_URL}#nixosConfigurations" >&2
exit 1
fi
echo "Note: lxc-* targets aren't installed this way build them with"
echo " nix build .#nixosConfigurations.<name>.config.system.build.tarball"
echo "and 'pct restore' the result on Proxmox directly. See docs/auto-installer.md."
echo "Choose the flake profile to install:"
select choice in "''${options[@]}"; do
if [[ -n "$choice" ]]; then
echo "You selected: $choice"
break
else
echo "Invalid selection. Try again."
fi
done
echo "Starting install with flake: ''${FLAKE_BASE_URL}#''${choice}"
# Optional: confirm before proceeding
read -rp "Proceed with installation? (y/N): " confirm
if [[ ! "$confirm" =~ ^[Yy]$ ]]; then
echo "Aborted."
exit 1
fi
# A nix-cache host is *the* substituter/remote-builder for every other
# host once installed (its own config explicitly excludes itself from
# using either see buildType != "nix-cache" in the nixos flake.nix).
# Installing one shouldn't depend on a nix-cache substituter either,
# for the same reason plus in practice "nix-cache" only resolves over
# Tailscale, which a fresh installer environment was never connected to
# anyway, so it's dead weight even for non-nix-cache installs until
# that's sorted out. Override it away here specifically for nix-cache
# targets to keep install-time behaviour consistent with run-time.
nix_extra_opts=()
if [[ "''${choice}" == *-nix-cache ]]; then
echo "Installing a nix-cache host skipping the nix-cache substituter."
nix_extra_opts+=(--option substituters "https://cache.nixos.org/")
fi
# Every host reachable through this menu has a Disko config (lxc-*
# is filtered out above, and is the only category that doesn't
# see docs/auto-installer.md), so this can run unconditionally: no
# need to probe the flake first and branch on whether Disko applies.
disko --mode destroy,format,mount \
--flake "''${FLAKE_BASE_URL}#''${choice}" "''${nix_extra_opts[@]}" --yes-wipe-all-disks
# sops-nix derives this host's decryption key from its own SSH host key
# at *activation* time, which runs before systemd would otherwise
# generate one on first boot. Without pre-seeding it here, secrets
# (including the login password) fail to decrypt on first boot.
# Generate the key with scripts/secrets/prepare-host-key.sh first.
#
# Two places a key can come from, checked in order:
# /etc/host-keys baked into this image at build time (see
# modules/installer/host-keys.nix; only present
# if built with NIXOS_HOST_KEYS_DIR set)
# /root/host-keys scp'd in manually after boot (older fallback,
# still supported for images built without keys)
mkdir -p /root/host-keys
if [[ -f "/etc/host-keys/''${choice}_ssh_host_ed25519_key" ]]; then
echo "Found baked-in SSH host key for ''${choice}, installing to target..."
install -D -m 0600 "/etc/host-keys/''${choice}_ssh_host_ed25519_key" /mnt/etc/ssh/ssh_host_ed25519_key
install -D -m 0644 "/etc/host-keys/''${choice}_ssh_host_ed25519_key.pub" /mnt/etc/ssh/ssh_host_ed25519_key.pub
elif [[ -f "/root/host-keys/''${choice}_ssh_host_ed25519_key" ]]; then
echo "Found pre-seeded SSH host key for ''${choice}, installing to target..."
install -D -m 0600 "/root/host-keys/''${choice}_ssh_host_ed25519_key" /mnt/etc/ssh/ssh_host_ed25519_key
install -D -m 0644 "/root/host-keys/''${choice}_ssh_host_ed25519_key.pub" /mnt/etc/ssh/ssh_host_ed25519_key.pub
else
echo "WARNING: no SSH host key found for ''${choice} (checked /etc/host-keys and /root/host-keys)"
echo "sops-nix secrets (including the login password) will NOT decrypt on first boot."
echo "Run scripts/secrets/prepare-host-key.sh for host ''${choice} on your admin workstation first,"
echo "then either rebuild this image with NIXOS_HOST_KEYS_DIR set, or scp the result to"
echo "/root/host-keys/ on this machine."
read -rp "Continue without a pre-seeded key anyway? (y/N): " skip_key
if [[ ! "$skip_key" =~ ^[Yy]$ ]]; then
echo "Aborted."
exit 1
fi
fi
mkdir -p /mnt/install-tmp
export TMPDIR=/mnt/install-tmp
nixos-install \
--flake "''${FLAKE_BASE_URL}#''${choice}" \
"''${nix_extra_opts[@]}" \
--no-root-password
rm -rf /mnt/install-tmp
# Redundant copy of the host's private key the real one is now at
# /etc/ssh/ssh_host_ed25519_key. Nothing NixOS-managed ever cleans this
# up on its own since it was written imperatively, not declaratively.
rm -rf /root/host-keys
sleep 10
reboot
'';
mode = "0755";
}; };
}; };
@@ -73,7 +192,7 @@
# file-copying/chown. # file-copying/chown.
programs.bash.loginShellInit = '' programs.bash.loginShellInit = ''
if [ -n "$PS1" ] && [ ! -e "$HOME/.auto_install_ran" ]; then if [ -n "$PS1" ] && [ ! -e "$HOME/.auto_install_ran" ]; then
sudo /etc/nixos-installer/installer/auto-install.sh sudo /etc/auto-install.sh
touch "$HOME/.auto_install_ran" touch "$HOME/.auto_install_ran"
fi fi
''; '';
-210
View File
@@ -1,210 +0,0 @@
# Fully declarative FreeIPA domain membership.
#
# Imported by modules/common/configuration.nix — no per-host wiring needed.
# Enables itself automatically on any host that has a sops-encrypted keytab
# at secrets/<hostname>.keytab; is a no-op for all other hosts.
#
# To enroll a new host:
# 0. scripts/secrets/sync-host-keys.sh <flake-target>
# 1. scripts/ipa/create-nixos-ipa-host-account.sh [--ip <addr>] <hostname>
# (adds .sops.yaml rule, runs ipa host-add, encrypts keytab in one step)
# 2. git add secrets/<hostname>.keytab .sops.yaml && git commit
# 3. Deploy — no further steps required.
#
# Manual fallback (if the script isn't usable):
# a. On the FreeIPA server: ipa host-add <fqdn> [--ip-address=<ip>] --force
# b. On the FreeIPA server: ipa-getkeytab -s <ipa-server> -p host/<fqdn> -k /tmp/<host>.keytab
# c. From the repo root (path must match for sops creation rule to apply):
# cp /tmp/<host>.keytab secrets/<host>.keytab
# sops -e --input-type binary -i secrets/<host>.keytab
# d. Commit secrets/<host>.keytab and the updated .sops.yaml, then deploy.
#
# vars dependencies: homeDomain, ipaServer, domainControllerIp, ipaUser
{ config, lib, pkgs, vars, ... }:
let
keytabPath = ../../secrets + "/${config.networking.hostName}.keytab";
enabled = builtins.pathExists keytabPath;
realm = lib.strings.toUpper vars.homeDomain;
fqdn = "${config.networking.hostName}.${vars.homeDomain}";
# "sweet.home" -> "dc=sweet,dc=home"
basedn = lib.strings.concatMapStringsSep "," (c: "dc=${c}") (lib.strings.splitString "." vars.homeDomain);
# security.ipa.certificate expects a derivation (package), not a raw path.
caCertPkg = pkgs.writeText "ipa-ca.crt" (builtins.readFile ../../certs/ipa-ca.crt);
in
lib.mkIf enabled {
networking.domain = lib.mkDefault vars.homeDomain;
networking.nameservers = lib.mkDefault [ vars.domainControllerIp ];
security = {
ipa = {
enable = true;
domain = vars.homeDomain;
inherit realm;
server = vars.ipaServer;
certificate = caCertPkg;
inherit basedn;
ipaHostname = fqdn;
offlinePasswords = true;
cacheCredentials = true;
};
# Create the home directory on first login if it doesn't exist yet.
# IPA users have no pre-created home on the host; without this sshd
# opens a session to a non-existent directory and resets the connection.
# lightdm also needs this so the GUI login path can create the home dir
# if it was not pre-seeded by the tmpfiles rule above (e.g. on first boot
# before SSSD has resolved the user).
pam.services = {
sshd.makeHomeDir = true;
lightdm.makeHomeDir = true;
# pam_unix returns PAM_AUTHINFO_UNAVAIL without prompting when the local
# stub has "!" in shadow (account locked), so PAM_AUTHTOK is never set
# and pam_sss's use_first_pass fails with "No authentication token".
# Changing to try_first_pass makes pam_sss prompt independently when no
# prior module has set the token, restoring IPA password login via
# LightDM and su.
login.rules.auth.sss.settings = lib.mkForce { try_first_pass = true; };
su.rules.auth.sss.settings = lib.mkForce { try_first_pass = true; };
};
# HM with useUserPackages = true (flake.nix) sets users.users.${ipaUser}.packages,
# which forces the stub into /etc/passwd. pam_sss.so with the "localusers" flag
# (added by NixOS when SSSD is enabled) then skips SSSD for any user it finds in
# local /etc/passwd — including this stub — falling through to pam_unix, which has
# no password for the stub → sudo auth always fails.
#
# Fix: NOPASSWD for the IPA user. The IPA user already authenticated to reach a
# shell (SSH public key from IPA or Kerberos), so re-prompting via a broken PAM
# path is security theater on a single-admin homelab.
sudo.extraRules = [{
users = [ vars.ipaUser ];
commands = [{ command = "ALL"; options = [ "NOPASSWD" ]; }];
}];
};
systemd = {
# Fetch SSH public keys from IPA so users can log in with the key stored
# in their IPA profile rather than needing ~/.ssh/authorized_keys on every
# host. sss_ssh_authorizedkeys queries SSSD (which queries IPA LDAP).
#
# /nix/store is 1775 (group-writable by nixbld). OpenSSH 10.0+ rejects
# AuthorizedKeysCommand binaries whose path contains any group-writable
# component, silently skipping the command. Copy to /usr/local/bin (all
# components root-owned, 755) so the path passes sshd's safety check.
tmpfiles.rules = [
"d /usr/local 0755 root root - -"
"d /usr/local/bin 0755 root root - -"
"C+ /usr/local/bin/sss_ssh_authorizedkeys 0555 root root - ${pkgs.sssd}/bin/sss_ssh_authorizedkeys"
# Pre-create the IPA user's home dir so Home Manager activation succeeds
# even before their first login. On a fresh system SSSD may not have
# resolved the user yet — tmpfiles warns and skips in that case (non-fatal),
# and pam_mkhomedir covers the first-login path as a fallback.
"d /home/${vars.ipaUser} 0700 ${vars.ipaUser} ${vars.ipaUser} - -"
];
# security.ipa enables Kerberos (security.krb5) which causes systemd to
# start auth-rpcgss-module.service and rpc-gssd.service for Kerberos NFS
# authentication. LXC containers can't load the auth_rpcgss kernel module
# and don't have /var/lib/nfs/rpc_pipefs, so both services fail.
#
# The NixOS IPA module already adds a drop-in for auth-rpcgss-module.service
# with ConditionPathExists=/etc/krb5.keytab. We use lib.mkForce to win the
# text conflict and add ConditionVirtualization=!container alongside it so
# the service is skipped (not failed) in containers that do have a keytab.
# Same fix for rpc-gssd.service which also fails in containers.
units = lib.mkIf config.boot.isContainer {
"auth-rpcgss-module.service" = {
overrideStrategy = "asDropinIfExists";
text = lib.mkForce ''
[Unit]
ConditionPathExists=
ConditionPathExists=/etc/krb5.keytab
ConditionVirtualization=!container
'';
};
# rpc-gssd also has ConditionPathExists from the NixOS IPA module (and an
# X-Restart-Triggers store path from systemd.nix). Use mkForce to win;
# omit X-Restart-Triggers since this service is skipped in containers anyway.
"rpc-gssd.service" = {
overrideStrategy = "asDropinIfExists";
text = lib.mkForce ''
[Unit]
ConditionPathExists=
ConditionPathExists=/etc/krb5.keytab
ConditionVirtualization=!container
'';
};
};
# home-manager-<user>.service fails on first enrollment because /home/wayne
# doesn't exist until the user's first login (pam_mkhomedir creates it then).
# ConditionPathExists makes systemd skip the service (exit 0, condition not
# met) instead of failing. After first login the dir exists and subsequent
# rebuilds activate HM normally.
services."home-manager-${vars.ipaUser}".unitConfig.ConditionPathExists =
"/home/${vars.ipaUser}";
};
services.openssh.extraConfig = ''
AuthorizedKeysCommand /usr/local/bin/sss_ssh_authorizedkeys %u
AuthorizedKeysCommandUser nobody
'';
# Host keytab: pre-provisioned on the IPA server, sops-encrypted binary.
# Placed at /etc/krb5.keytab before SSSD starts so the host authenticates
# to IPA without running ipa-client-install.
sops.secrets."ipa-host-keytab" = {
sopsFile = keytabPath;
format = "binary";
path = "/etc/krb5.keytab";
owner = "root";
group = "root";
mode = "0600";
restartUnits = [ "sssd.service" ];
};
# NixOS requires isNormalUser/isSystemUser + group on any entry in
# users.users. HM with useUserPackages = true (set in flake.nix) adds a stub
# entry for each HM user so it can install packages to
# /etc/profiles/per-user/<name>/. This definition satisfies those assertions.
# With security.ipa setting "passwd: sss files" in nsswitch, SSSD's IPA entry
# takes priority for NSS lookups — this local stub is only a fallback when
# SSSD is unreachable (at which point auth fails anyway).
users.users.${vars.ipaUser} = {
isNormalUser = true;
group = "users";
extraGroups = [ "wheel" ];
createHome = false;
# "!" is not a password hash — it is the standard "account locked" marker.
# It cannot authenticate anyone locally. It exists solely so NixOS generates
# a shadow entry for this stub user; without one pam_unix returns
# PAM_AUTHINFO_UNAVAIL before prompting, which means PAM_AUTHTOK is never
# set and the subsequent pam_sss use_first_pass call has nothing to work
# with — blocking LightDM and su logins even when IPA/SSSD auth succeeds.
hashedPassword = "!";
};
# Home Manager config for the IPA primary user, applied on every enrolled
# host. Manages what IPA doesn't: dotfiles, user-scoped packages, session
# variables. Switch-nix/Test-nix/buildImage are system-wide (configuration.nix)
# so they don't need to be repeated here.
#
# homeDirectory uses mkForce because HM's NixOS integration module sets it to
# "/var/empty" for users not found in config.users.users at eval time (SSSD
# users aren't visible there).
home-manager.users.${vars.ipaUser} = { pkgs, ... }: {
home = {
username = vars.ipaUser;
homeDirectory = lib.mkForce "/home/${vars.ipaUser}";
stateVersion = "26.05";
packages = with pkgs; [ tmux sshfs ];
sessionVariables.EDITOR = lib.mkDefault "nano";
};
programs.home-manager.enable = true;
programs.bash.enable = true;
};
}
-43
View File
@@ -1,43 +0,0 @@
{ config, lib, vars, ... }:
{
# Prestages a NetworkManager connection profile for vars.wifiSsid so the
# host associates on first boot with no manual nmtui/nmcli step. Guarded
# on a non-empty SSID so leaving the placeholder blank in variables.nix
# is a no-op rather than an empty, broken profile — fill it in once the
# network is known.
#
# The password itself lives in secrets/gui.yaml, not variables.nix --
# NetworkManager's ensureProfiles renders `psk = "$WIFI_PASSWORD"`
# literally into the store (see nixpkgs' own ensureProfiles example,
# which does the same for exactly this reason) and its systemd service
# envsubst-expands it from environmentFiles at activation time, so the
# real value only ever touches /run (root-only, UMask 0177), never the
# Nix store.
sops.secrets."wifi-password" = lib.mkIf (vars.wifiSsid != "") {
sopsFile = ../../secrets/gui.yaml;
};
sops.templates."wifi-password.env" = lib.mkIf (vars.wifiSsid != "") {
content = "WIFI_PASSWORD=${config.sops.placeholder."wifi-password"}";
};
networking.networkmanager.ensureProfiles = lib.mkIf (vars.wifiSsid != "") {
environmentFiles = [ config.sops.templates."wifi-password.env".path ];
profiles.${vars.wifiSsid} = {
connection = {
id = vars.wifiSsid;
type = "wifi";
};
wifi = {
mode = "infrastructure";
ssid = vars.wifiSsid;
};
wifi-security = {
key-mgmt = "wpa-psk";
psk = "$WIFI_PASSWORD";
};
};
};
}
+1 -1
View File
@@ -3,7 +3,7 @@
{ {
nix.settings = { nix.settings = {
substituters = [ substituters = [
"http://${vars.nixCacheHost}.${vars.homeDomain}" "http://${vars.nixCacheHost}"
"https://cache.nixos.org/" "https://cache.nixos.org/"
]; ];
trusted-public-keys = [ trusted-public-keys = [
+4 -4
View File
@@ -8,12 +8,12 @@
# dedicated keypair). If this host doesn't have one yet: # dedicated keypair). If this host doesn't have one yet:
# sudo -u root ssh-keygen -t ed25519 -N '' -f /root/.ssh/id_ed25519 # sudo -u root ssh-keygen -t ed25519 -N '' -f /root/.ssh/id_ed25519
# # then add its .pub to vars.remoteBuilderAuthorizedKeys and rebuild nix-cache # # then add its .pub to vars.remoteBuilderAuthorizedKeys and rebuild nix-cache
# sudo ssh -i /root/.ssh/id_ed25519 nixremote@nix-cache.sweet.home nix-store --version # sudo ssh -i /root/.ssh/id_ed25519 nixremote@nix-cache nix-store --version
# Trust nix-cache's SSH host key declaratively so the nix-daemon (root) # Trust nix-cache's SSH host key declaratively so the nix-daemon (root)
# can connect the first time without a manual ssh-keyscan/known_hosts # can connect the first time without a manual ssh-keyscan/known_hosts
# step on every new client. # step on every new client.
programs.ssh.knownHosts."${vars.nixCacheHost}.${vars.homeDomain}" = { programs.ssh.knownHosts.${vars.nixCacheHost} = {
hostNames = [ "${vars.nixCacheHost}.${vars.homeDomain}" ]; hostNames = [ vars.nixCacheHost ];
publicKey = vars.nixCacheHostKey; publicKey = vars.nixCacheHostKey;
}; };
@@ -22,7 +22,7 @@
buildMachines = [ buildMachines = [
{ {
hostName = "${vars.nixCacheHost}.${vars.homeDomain}"; hostName = vars.nixCacheHost;
sshUser = vars.remoteBuilderUser; sshUser = vars.remoteBuilderUser;
sshKey = "/root/.ssh/id_ed25519"; sshKey = "/root/.ssh/id_ed25519";
inherit (pkgs.stdenv.hostPlatform) system; inherit (pkgs.stdenv.hostPlatform) system;
+1 -1
View File
@@ -20,7 +20,7 @@
nginx = { nginx = {
enable = true; enable = true;
recommendedProxySettings = true; recommendedProxySettings = true;
virtualHosts."${vars.nixCacheHost}.${vars.homeDomain}" = { virtualHosts.${vars.nixCacheHost} = {
locations."/" = { locations."/" = {
proxyPass = "http://${config.services.nix-serve.bindAddress}:${toString config.services.nix-serve.port}"; proxyPass = "http://${config.services.nix-serve.bindAddress}:${toString config.services.nix-serve.port}";
}; };
-38
View File
@@ -1,38 +0,0 @@
{ ... }:
{
imports = [
../hardware-configuration/baremetal.nix
../boot/efi.nix
../disko/baremetal.nix
../services/zfs/enable-service.nix
];
# Needed for real wifi/bluetooth/GPU firmware blobs and CPU microcode
# updates (hardware-configuration/baremetal.nix's amd.updateMicrocode
# keys off this) -- irrelevant on the linode/proxmox/lxc platforms,
# which are all VMs with no real hardware to load firmware for.
hardware.enableRedistributableFirmware = true;
# AMD GPU: the amdgpu kernel driver autoloads from the PCI ID with no
# extra boot.kernelModules entry needed; this is the userspace half --
# the dedicated Xorg driver (not just the generic modesetting fallback)
# plus Mesa OpenGL/Vulkan (amdgpu/RADV), same firmware blobs as above.
# 32-bit support is for compatibility with 32-bit apps/games.
services.xserver.videoDrivers = [ "amdgpu" ];
hardware.graphics = {
enable = true;
enable32Bit = true;
};
# The systemd-based initrd (default here since this host has a ZFS root --
# see modules/disko/baremetal.nix) locks the root account by default, so
# sulogin refuses to hand over a shell if something in the initrd (e.g.
# the ZFS pool import) fails and it drops to emergency mode -- confirmed
# live: it just loops re-entering the target instead of prompting. This
# only affects the pre-switch-root initrd shell, not the installed
# system's own login, and is worth the tradeoff on a box already reachable
# at the physical console.
boot.initrd.systemd.emergencyAccess = true;
}
+20 -60
View File
@@ -52,7 +52,6 @@ in
# LXC container does). # LXC container does).
imports = [ imports = [
(modulesPath + "/virtualisation/proxmox-lxc.nix") (modulesPath + "/virtualisation/proxmox-lxc.nix")
../common/preserve-ssh-host-key.nix
]; ];
proxmoxLXC = { proxmoxLXC = {
@@ -64,22 +63,23 @@ in
# back to decide `pct create`'s --unprivileged flag, so the two stay # back to decide `pct create`'s --unprivileged flag, so the two stay
# in sync). # in sync).
# #
# Any lxc-* host with an NFS fileSystem must be privileged: the kernel's # lxc-docker is the one exception: the kernel's NFS client doesn't set
# NFS client doesn't set FS_USERNS_MOUNT, so mounting NFS from inside # FS_USERNS_MOUNT, so mounting NFS from inside *any* non-init user
# *any* non-init user namespace -- which is exactly what an unprivileged # namespace -- which is exactly what an unprivileged container's
# container's UID-mapped root runs in -- is rejected at the VFS layer # UID-mapped root runs in -- is rejected at the VFS layer with EPERM,
# with EPERM, no matter what Proxmox's own `mount=nfs;nfs4` container # no matter what Proxmox's own `mount=nfs;nfs4` container feature
# feature allows at the AppArmor layer (confirmed live: TCP to the NFS # allows at the AppArmor layer (confirmed live: TCP to the NFS server
# server succeeds, the server's export table matches the container's IP, # succeeds, the server's export table matches the container's IP, and
# and `mount.nfs: Operation not permitted` still fires immediately with # `mount.nfs: Operation not permitted` still fires immediately with no
# no corresponding denial anywhere in the server's logs -- a kernel-level # corresponding denial anywhere in the server's logs -- a kernel-level
# rejection, not a network or export-permission one). Deriving this from # rejection, not a network or export-permission one). Keying off
# fileSystems rather than a per-host override keeps it self-consistent: # hostName rather than something docker-build-type-specific because
# any new lxc-* host that declares an NFS mount automatically gets the # modules/build-types/docker.nix is also composed for linode-docker/
# privilege level it needs without a separate manual flag. # proxmox-docker, which don't import proxmox-lxc.nix at all --setting
privileged = builtins.any # this option there would break their eval with "option does not
(fs: fs.fsType == "nfs" || fs.fsType == "nfs4") # exist" regardless of any mkIf guard, since mkIf only makes a value
(builtins.attrValues config.fileSystems); # conditional, not whether the option needs to exist somewhere.
privileged = config.networking.hostName == "docker";
}; };
boot.loader = { boot.loader = {
@@ -112,12 +112,9 @@ in
# sops-nix's "for users" secrets (password hashes -- installed by the # sops-nix's "for users" secrets (password hashes -- installed by the
# activation script itself, not a systemd service, since they need to # activation script itself, not a systemd service, since they need to
# exist *before* user creation) nor the user-creation step that # exist *before* user creation) nor the user-creation step that
# consumes them ever run on a real lxc-* boot. In this config sops-nix # consumes them ever run on a real lxc-* boot. Regular secrets
# does NOT generate its own boot-time service (confirmed live: no # (nix-serve's key, beszel's token, etc.) work anyway because sops-nix
# sops-nix.service in systemctl list-unit-files on a deployed # provides its own systemd service for those.
# lxc-tor-relay container); /run/secrets is a tmpfs cleared on every
# reboot, so secrets must be reinstalled on each non-first boot by
# nixos-lxc-sops-reinstall (below).
# #
# A systemd service, not boot.postBootCommands: tried that first (it's # A systemd service, not boot.postBootCommands: tried that first (it's
# a genuine, generally-invoked hook -- nixos/modules/system/boot/stage-2-init.sh, # a genuine, generally-invoked hook -- nixos/modules/system/boot/stage-2-init.sh,
@@ -168,41 +165,4 @@ in
touch /var/lib/nixos-lxc-first-boot-activated touch /var/lib/nixos-lxc-first-boot-activated
''; '';
}; };
# Reinstalls sops secrets on every non-first boot. /run/secrets is a
# tmpfs that is cleared on each reboot; without this service, secrets
# are permanently absent after the first boot and every service that
# reads from /run/secrets fails on start.
#
# wantedBy/before network.target: switch-to-configuration test requires
# D-Bus to restart systemd targets after running activation scripts. D-Bus
# is available once basic.target completes (the default After=basic.target
# that DefaultDependencies would otherwise add). Placing the service before
# network.target ensures secrets are ready before any network-dependent
# service (including beszel-agent and nix-serve) starts, while running late
# enough that D-Bus is already up.
#
# ConditionPathExists=... skips this service on the genuine first boot
# (the marker doesn't exist yet); nixos-lxc-first-boot-activate handles
# that case. On every subsequent boot the condition passes and secrets
# are reinstalled before user services start.
#
# SuccessExitStatus=11: switch-to-configuration exits 11 when it cannot
# acquire the activation lock (another switch is already in progress).
# During a nixos-rebuild switch the activation already installs secrets, so
# treating the lock-held case as success is correct.
systemd.services.nixos-lxc-sops-reinstall = {
description = "Reinstall sops secrets on each non-first boot (LXC, /run is tmpfs)";
wantedBy = [ "network.target" ];
before = [ "network.target" ];
unitConfig.ConditionPathExists = "/var/lib/nixos-lxc-first-boot-activated";
serviceConfig = {
Type = "oneshot";
RemainAfterExit = true;
SuccessExitStatus = "11";
};
script = ''
/run/current-system/bin/switch-to-configuration test
'';
};
} }
+1 -42
View File
@@ -1,50 +1,9 @@
{ lib, flakeTarget, ... }: { ... }:
let
# Bakes this exact flake target's pre-generated SSH host key straight
# into /etc/ssh/ -- mirrors lxc.nix's builtins.getEnv pattern (impure
# and empty under normal `nix build`/`nix eval`, so this is a no-op
# unless explicitly opted into with NIXOS_HOST_KEYS_DIR=... --impure).
#
# Unlike --pre-format-files (which places files on the QEMU builder VM's
# rootfs, not the target disk), embedding via environment.etc here means
# nixos-install's own activation step installs the key onto the target
# disk. sshd-keygen then finds it already present and skips generation,
# so the disk image boots with the clan-registered key and sops can
# decrypt on first boot.
#
# Without this, nixos-install's sshd-keygen activation generates a fresh
# key (unregistered in .sops.yaml), sops decryption fails permanently,
# and password hashes are never applied -- confirmed live: passwords
# stayed '!' even with mutableUsers = false because hashedPasswordFile
# pointed to a path that sops never wrote.
hostKeysDirStr = builtins.getEnv "NIXOS_HOST_KEYS_DIR";
hasHostKeysDir = hostKeysDirStr != "" && builtins.pathExists hostKeysDirStr;
hostKeysDir = /. + hostKeysDirStr;
privKeyFile = hostKeysDir + "/${flakeTarget}_ssh_host_ed25519_key";
pubKeyFile = hostKeysDir + "/${flakeTarget}_ssh_host_ed25519_key.pub";
hasKeyForThisTarget =
hasHostKeysDir
&& builtins.pathExists privKeyFile
&& builtins.pathExists pubKeyFile;
in
{ {
imports = [ imports = [
../hardware-configuration/vm/proxmox.nix ../hardware-configuration/vm/proxmox.nix
../boot/efi.nix ../boot/efi.nix
../disko/proxmox.nix ../disko/proxmox.nix
../common/preserve-ssh-host-key.nix
]; ];
environment.etc = lib.mkIf hasKeyForThisTarget {
"ssh/ssh_host_ed25519_key" = {
source = privKeyFile;
mode = "0600";
};
"ssh/ssh_host_ed25519_key.pub" = {
source = pubKeyFile;
mode = "0644";
};
};
} }
-42
View File
@@ -1,42 +0,0 @@
{ config, lib, vars, ... }:
let
# FQDN of the LAN NFS VIP (Pacemaker vip-lan, 192.168.2.229). Defined in
# variables.nix as haLanNfsFqdn; using the FQDN avoids systemd-resolved
# LLMNR quirks and survives a future VIP renumber via a DNS-only update.
nfsServer = vars.haLanNfsFqdn;
in
{
fileSystems.${vars.nfsShares.pxebootImages.mountpoint} = {
device = "${nfsServer}:${vars.haStorageRoot}/${vars.nfsShares.pxebootImages.subpath}";
fsType = "nfs";
options = [
"_netdev"
"noatime"
] ++ (if config.boot.isContainer
# NFSv4 requires rpc_pipefs (sunrpc filesystem), which Proxmox LXC
# containers block unless `features: mount=nfs` is set. Use NFSv3+nolock
# instead: no rpc_pipefs dependency at the protocol level, and rpcbind
# on the server handles port resolution without needing client-side
# sunrpc infrastructure. nofail keeps boot clean if server is unreachable.
then [ "nfsvers=3" "proto=tcp" "nolock" "nofail" ]
else [ "nfsvers=4.2" "x-systemd.automount" ]);
};
# NixOS pulls var-lib-nfs-rpc_pipefs.mount (the sunrpc filesystem) into
# nfs-client.target for any nfs fileSystems entry. In LXC containers the
# sunrpc mount is blocked by Proxmox's AppArmor profile, causing it to fail
# and the activation to report an error even though our mount uses nofail.
# Add ConditionVirtualization=!container via drop-in so systemd skips the
# unit entirely in containers (skip = inactive, not failed), which keeps
# nfs-client.target green and activation clean.
systemd.units = lib.mkIf config.boot.isContainer {
"var-lib-nfs-rpc_pipefs.mount" = {
overrideStrategy = "asDropin";
text = ''
[Unit]
ConditionVirtualization=!container
'';
};
};
}
+14 -23
View File
@@ -1,32 +1,23 @@
{ netbootSystem, netbootMinimalSystem, ... }: { netbootSystem, ... }:
let let
# config.system.build.kernel and .netbootRamdisk are directories, not the # config.system.build.kernel and .netbootRamdisk are directories, not the
# files themselves — nixpkgs' own system.build.kexecTree does the same # files themselves — nixpkgs' own system.build.kexecTree does the same
# ${...}/<file> dereference for the same reason. # ${...}/<file> dereference for the same reason.
mkStageRules = { dirName, system }: inherit (netbootSystem.config.system.boot.loader) kernelFile;
let
inherit (system.config.system.boot.loader) kernelFile;
dir = "/srv/pxe/http/${dirName}";
in
[
# Declared here too (not just in build-types/pxe-boot.nix) so this
# module's C+ rules don't depend on cross-module list-merge ordering —
# tmpfiles' C type needs the target directory to already exist.
"d ${dir} 0755 root root -"
"C+ ${dir}/${kernelFile} 0644 root root - ${system.config.system.build.kernel}/${kernelFile}"
"C+ ${dir}/initrd 0644 root root - ${system.config.system.build.netbootRamdisk}/initrd"
"C+ ${dir}/netboot.ipxe 0644 root root - ${system.config.system.build.netbootIpxeScript}/netboot.ipxe"
];
in in
{ {
# Builds this flake's own installer netboot image (the same one # Builds this flake's own installer netboot image (the same one
# `nix build .#pxe` produces) plus the vanilla NixOS minimal netboot image # `nix build .#pxe` produces) and stages it where menu.ipxe's :nixos
# (`nix build .#pxe-minimal`), and stages both where menu.ipxe's # entry expects it, so the pxe-boot host is self-contained — no manual
# :auto-installer / :nixos-minimal entries expect them, so the pxe-boot # operator step to populate /srv/pxe/http/nixos after deploy.
# host is self-contained — no manual operator step to populate systemd.tmpfiles.rules = [
# /srv/pxe/http after deploy. # Declared here too (not just in build-types/pxe-boot.nix) so this
systemd.tmpfiles.rules = # module's C+ rules don't depend on cross-module list-merge ordering —
mkStageRules { dirName = "auto-installer"; system = netbootSystem; } # tmpfiles' C type needs the target directory to already exist.
++ mkStageRules { dirName = "nixos-minimal"; system = netbootMinimalSystem; }; "d /srv/pxe/http/nixos 0755 root root -"
"C+ /srv/pxe/http/nixos/${kernelFile} 0644 root root - ${netbootSystem.config.system.build.kernel}/${kernelFile}"
"C+ /srv/pxe/http/nixos/initrd 0644 root root - ${netbootSystem.config.system.build.netbootRamdisk}/initrd"
"C+ /srv/pxe/http/nixos/netboot.ipxe 0644 root root - ${netbootSystem.config.system.build.netbootIpxeScript}/netboot.ipxe"
];
} }
+27
View File
@@ -0,0 +1,27 @@
_:
{
imports = [ ./enable-service.nix ];
services.tailscale = {
# Enables the sysctl forwarding settings exit nodes/subnet routers need;
# without this, --advertise-exit-node has no effect.
useRoutingFeatures = "server";
# Lets peers reach this node directly over the tailscale UDP port
# instead of relaying through DERP.
openFirewall = true;
# extraSetFlags (tailscale set, via the always-on tailscaled-set
# service), not extraUpFlags -- extraUpFlags is only ever applied by
# tailscaled-autoconnect, which itself only runs when
# services.tailscale.authKeyFile is set (nothing in this repo sets one,
# so tailscale up is a manual, one-time operator step on every host that
# uses this service). extraSetFlags has no such gate, so
# --advertise-exit-node self-reapplies on every boot once the operator
# has authenticated the node once.
extraSetFlags = [
"--advertise-exit-node"
];
};
}
-35
View File
@@ -1,35 +0,0 @@
{ pkgs, ... }:
{
imports = [ ./enable-service.nix ];
services.tailscale = {
# Enables the sysctl forwarding settings subnet routers need;
# without this, --advertise-routes has no effect.
useRoutingFeatures = "server";
# Lets peers reach this node directly over the tailscale UDP port
# instead of relaying through DERP.
openFirewall = true;
};
# Tailscale recommends these ethtool flags on the uplink interface to get
# full UDP GRO throughput on subnet routers (https://tailscale.com/s/ethtool-config-udp-gro).
# The interface is derived from the default route so it works regardless of
# what the NIC is named on a given host.
systemd.services.tailscale-udp-gro = {
description = "Enable UDP GRO forwarding on uplink for Tailscale subnet router";
after = [ "network-online.target" ];
wants = [ "network-online.target" ];
wantedBy = [ "multi-user.target" ];
path = [ pkgs.ethtool pkgs.iproute2 ];
serviceConfig = {
Type = "oneshot";
RemainAfterExit = true;
ExecStart = pkgs.writeShellScript "tailscale-udp-gro" ''
NETDEV=$(ip -o route get 8.8.8.8 | cut -f 5 -d " ")
ethtool -K "$NETDEV" rx-udp-gro-forwarding on rx-gro-list off
'';
};
};
}
-53
View File
@@ -1,53 +0,0 @@
{ vars, ... }:
{
# Run dnsmasq on the LAN interface as a forwarding-only resolver for
# *.ts.net (Tailscale MagicDNS names). FreeIPA's bind-dyndb-ldap
# cannot reach vars.tailscaleResolverIp directly because the DC is not a
# Tailscale node. This host IS a Tailscale node and can reach it via
# tailscale0, so it acts as an intermediary: FreeIPA has a conditional
# forward zone for ts.net pointing here (vars.tailscaleRouterIp), and this
# dnsmasq instance forwards those queries onward to Tailscale's resolver.
#
# Configure FreeIPA once after deploying this host:
# kinit admin
# ipa dnsforwardzone-add ${vars.tailnetDomain} \
# --forwarder=${vars.tailscaleRouterIp} \
# --forward-policy=only
# Note: IPA refuses to shadow ts.net (a real public TLD); use the
# tailnet-specific subdomain (vars.tailnetDomain) instead.
services.dnsmasq = {
enable = true;
# NixOS's dnsmasq module defaults resolveLocalQueries to true, which adds
# 127.0.0.1 to networking.nameservers and makes dnsmasq bind to
# listen-address=127.0.0.1. This instance is not the host's local
# resolver — it only serves IPA's conditional forwarder for tailnet names.
# The host uses domainControllerIp directly (networking.nameservers in
# host.nix). Without this, all host DNS goes through dnsmasq, which has
# no upstream for general queries (no-resolv=true), breaking resolution.
resolveLocalQueries = false;
settings = {
# Listen only on the LAN interface — not tailscale0 or loopback.
# bind-interfaces prevents dnsmasq from binding to 0.0.0.0 and then
# filtering by interface later; combined with `interface` this ensures
# it genuinely listens only on eth0.
bind-interfaces = true;
interface = [ vars.lxcLanInterface ];
# Forward-only: no local /etc/hosts or /etc/resolv.conf reading, no
# negative caching of NXDOMAIN for names this instance doesn't serve.
# All ts.net queries come from FreeIPA's conditional forwarder and must
# be answered by Tailscale's resolver.
no-hosts = true;
no-resolv = true;
# Forward *.tailnetDomain to Tailscale's internal resolver, scoped to
# the tailnet-specific subdomain rather than all of ts.net (FreeIPA
# refuses to shadow ts.net, a real public TLD).
server = [ "/${vars.tailnetDomain}/${vars.tailscaleResolverIp}" ];
};
};
networking.firewall.allowedUDPPorts = [ vars.ports.dns ];
networking.firewall.allowedTCPPorts = [ vars.ports.dns ];
}
+27 -46
View File
@@ -16,14 +16,6 @@
# #
# --dry-run: adds `nix build --dry-run --no-link` for whatever scope is # --dry-run: adds `nix build --dry-run --no-link` for whatever scope is
# active (changed-files scope by default, full scope under --full-check). # active (changed-files scope by default, full scope under --full-check).
#
# Per-host/per-package eval and dry-run build calls run concurrently (see
# scripts/lib/nix-parallel.sh) since they're independent of each other.
# Concurrency defaults to core count capped by available memory (~1GB/job)
# rather than plain core count, since each concurrent `nix eval` evaluates a
# whole NixOS system closure and can OOM a small/memory-constrained CI
# runner otherwise; override via NIX_PARALLEL_JOBS if a runner has more (or
# less) room than that estimate assumes.
set -euo pipefail set -euo pipefail
script_dir="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" script_dir="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
@@ -31,23 +23,10 @@ script_dir="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
source "${script_dir}/lib/nix-bootstrap.sh" source "${script_dir}/lib/nix-bootstrap.sh"
# shellcheck source=lib/nix-eval.sh # shellcheck source=lib/nix-eval.sh
source "${script_dir}/lib/nix-eval.sh" source "${script_dir}/lib/nix-eval.sh"
# shellcheck source=lib/nix-parallel.sh
source "${script_dir}/lib/nix-parallel.sh"
repo_root="$(cd "${script_dir}/.." && pwd)" repo_root="$(cd "${script_dir}/.." && pwd)"
cd "$repo_root" cd "$repo_root"
# When this repo is a subdirectory of a larger git repo (e.g. a mono-repo
# subtree), `git diff --name-only` outputs paths relative to the outer git
# root, not this directory. Compute a prefix to strip so pattern matching
# below works correctly regardless of nesting depth.
_git_root="$(git rev-parse --show-toplevel 2>/dev/null || echo "$repo_root")"
if [[ "$repo_root" != "$_git_root" ]]; then
_subtree_prefix="${repo_root#"$_git_root"/}/"
else
_subtree_prefix=""
fi
full_check=false full_check=false
dry_run=false dry_run=false
@@ -131,7 +110,7 @@ if ! $full_check; then
base_ref="$(resolve_base_ref)" base_ref="$(resolve_base_ref)"
echo echo
echo "Changed-files scope: diffing against ${base_ref}" echo "Changed-files scope: diffing against ${base_ref}"
mapfile -t changed_files < <(git diff --name-only --diff-filter=ACMR "$base_ref" -- . | sort -u | sed "s|^${_subtree_prefix}||") mapfile -t changed_files < <(git diff --name-only --diff-filter=ACMR "$base_ref" -- . | sort -u)
if [[ ${#changed_files[@]} -eq 0 ]]; then if [[ ${#changed_files[@]} -eq 0 ]]; then
echo "No changed files detected." echo "No changed files detected."
@@ -267,64 +246,66 @@ echo
if [[ ${#hosts[@]} -eq 0 ]]; then if [[ ${#hosts[@]} -eq 0 ]]; then
echo "No hosts affected by changed files; skipping host eval." echo "No hosts affected by changed files; skipping host eval."
else else
echo "Evaluating host toplevel derivations (${scope_desc}, up to ${NIX_PARALLEL_JOBS} at a time)..." echo "Evaluating host toplevel derivations (${scope_desc})..."
# lxc-* hosts deploy via a directly pct-restore-able tarball instead of
# nixos-install (see docs/auto-installer.md); proxmox-* hosts can
# alternatively be built as a standalone disk image (see
# docs/proxmox-images.md). Both are otherwise-unvalidated buildable
# surface, easy to silently break without this.
declare -a host_eval_jobs=()
for host in "${hosts[@]}"; do for host in "${hosts[@]}"; do
host_eval_jobs+=("${host}${NIX_PARALLEL_SEP}.#nixosConfigurations.${host}.config.system.build.toplevel.drvPath") echo "==> $host"
nix eval --raw "${NIX_EVAL_FLAGS[@]}" ".#nixosConfigurations.${host}.config.system.build.toplevel.drvPath"
# lxc-* hosts deploy via a directly pct-restore-able tarball instead of
# nixos-install (see docs/auto-installer.md); proxmox-* hosts can
# alternatively be built as a standalone disk image (see
# docs/proxmox-images.md). Both are otherwise-unvalidated buildable
# surface, easy to silently break without this.
case "$host" in case "$host" in
lxc-*) lxc-*)
host_eval_jobs+=("${host} (tarball)${NIX_PARALLEL_SEP}.#nixosConfigurations.${host}.config.system.build.tarball.drvPath") echo "==> $host (tarball)"
nix eval --raw "${NIX_EVAL_FLAGS[@]}" ".#nixosConfigurations.${host}.config.system.build.tarball.drvPath"
;; ;;
proxmox-*) proxmox-*)
host_eval_jobs+=("${host} (diskoImagesScript)${NIX_PARALLEL_SEP}.#nixosConfigurations.${host}.config.system.build.diskoImagesScript.drvPath") echo "==> $host (diskoImagesScript)"
nix eval --raw "${NIX_EVAL_FLAGS[@]}" ".#nixosConfigurations.${host}.config.system.build.diskoImagesScript.drvPath"
;; ;;
esac esac
done done
run_nix_parallel host_eval_jobs eval --raw "${NIX_EVAL_FLAGS[@]}"
fi fi
echo echo
if ! $eval_packages; then if ! $eval_packages; then
echo "No packages affected by changed files; skipping package eval." echo "No packages affected by changed files; skipping package eval."
else else
echo "Evaluating buildable packages (up to ${NIX_PARALLEL_JOBS} at a time)..." echo "Evaluating buildable packages..."
declare -a package_eval_jobs=()
for pkg in "${all_packages[@]}"; do for pkg in "${all_packages[@]}"; do
package_eval_jobs+=("packages.x86_64-linux.${pkg}${NIX_PARALLEL_SEP}.#packages.x86_64-linux.${pkg}") echo "==> packages.x86_64-linux.${pkg}"
nix eval --raw "${NIX_EVAL_FLAGS[@]}" ".#packages.x86_64-linux.${pkg}"
done done
run_nix_parallel package_eval_jobs eval --raw "${NIX_EVAL_FLAGS[@]}"
fi fi
if $dry_run; then if $dry_run; then
echo echo
echo "Running dry-run builds for the active scope (up to ${NIX_PARALLEL_JOBS} at a time). This will not create result symlinks." echo "Running dry-run builds for the active scope. This will not create result symlinks."
declare -a host_build_jobs=()
for host in "${hosts[@]:-}"; do for host in "${hosts[@]:-}"; do
host_build_jobs+=("Dry-run build: ${host}${NIX_PARALLEL_SEP}.#nixosConfigurations.${host}.config.system.build.toplevel") echo "==> Dry-run build: $host"
nix build --dry-run --no-link "${NIX_EVAL_FLAGS[@]}" ".#nixosConfigurations.${host}.config.system.build.toplevel"
case "$host" in case "$host" in
lxc-*) lxc-*)
host_build_jobs+=("Dry-run build: ${host} (tarball)${NIX_PARALLEL_SEP}.#nixosConfigurations.${host}.config.system.build.tarball") echo "==> Dry-run build: $host (tarball)"
nix build --dry-run --no-link "${NIX_EVAL_FLAGS[@]}" ".#nixosConfigurations.${host}.config.system.build.tarball"
;; ;;
proxmox-*) proxmox-*)
host_build_jobs+=("Dry-run build: ${host} (diskoImagesScript)${NIX_PARALLEL_SEP}.#nixosConfigurations.${host}.config.system.build.diskoImagesScript") echo "==> Dry-run build: $host (diskoImagesScript)"
nix build --dry-run --no-link "${NIX_EVAL_FLAGS[@]}" ".#nixosConfigurations.${host}.config.system.build.diskoImagesScript"
;; ;;
esac esac
done done
run_nix_parallel host_build_jobs build --dry-run --no-link "${NIX_EVAL_FLAGS[@]}"
if $eval_packages; then if $eval_packages; then
echo echo
echo "Running dry-run builds for packages." echo "Running dry-run builds for packages."
declare -a package_build_jobs=()
for pkg in "${all_packages[@]}"; do for pkg in "${all_packages[@]}"; do
package_build_jobs+=("Dry-run build: packages.x86_64-linux.${pkg}${NIX_PARALLEL_SEP}.#packages.x86_64-linux.${pkg}") echo "==> Dry-run build: packages.x86_64-linux.${pkg}"
nix build --dry-run --no-link "${NIX_EVAL_FLAGS[@]}" ".#packages.x86_64-linux.${pkg}"
done done
run_nix_parallel package_build_jobs build --dry-run --no-link "${NIX_EVAL_FLAGS[@]}"
fi fi
fi fi
-1
View File
@@ -68,7 +68,6 @@ cat > "$HOME/.config/nix/nix.conf" <<'EOF'
experimental-features = nix-command flakes experimental-features = nix-command flakes
accept-flake-config = false accept-flake-config = false
warn-dirty = false warn-dirty = false
build-users-group =
EOF EOF
echo "Nix version:" echo "Nix version:"
-595
View File
@@ -1,595 +0,0 @@
#!/usr/bin/env bash
# deploy.sh — Full lifecycle management for the Docker Swarm HA cluster.
#
# Provisions two NixOS Proxmox VMs (ha-docker-1, ha-docker-2) as dual-manager
# Docker Swarm nodes sharing NFS storage from the existing HA file-server
# cluster. Both nodes are managers so either can accept Docker API and
# `docker stack` commands.
#
# Usage:
# scripts/docker-swarm/deploy.sh [options]
# scripts/docker-swarm/deploy.sh --destroy [options]
#
# Phases (all run by default; skip any with --skip-<phase>):
# 1. ensure-bridge Create vmbr3 (swarm cluster bridge) on the Proxmox node.
# 2. sync-keys Generate SSH host keys for both nodes (clan vars).
# 3. ipa-hosts Create IPA host objects + sops-encrypted keytabs.
# 4. create-vms Build NixOS disk images and create VMs via create-proxmox-resource.sh.
# 5. add-hardware Attach vmbr2 (storage) and vmbr3 (swarm) NICs; start VMs.
# 6. boot-wait Wait for SSH on both LAN IPs.
# 7. refresh-sops-keys Detect disko key drift; re-encrypt secrets; commit.
# 8. init-swarm docker swarm init on node1; manager join on node2; label nodes.
# 9. dns Register storage.home and swarm.home A records in FreeIPA.
# 10. verify docker node ls; NFS mount check; swarm health.
#
# Options:
# --node <host> Proxmox host (default: pve1.sweet.home)
# --vmid1 <n> VMID for ha-docker-1 (default: 202)
# --vmid2 <n> VMID for ha-docker-2 (default: 203)
# --storage <pool> Proxmox storage pool (default: local-zfs)
# --swarm-bridge <br> Bridge for Docker Swarm cluster network (default: vmbr3)
# --storage-bridge <br> Bridge for NFS storage network (default: vmbr2)
# --memory <MB> RAM per node (default: 4096)
# --cores <n> vCPUs per node (default: 4)
# --skip-ensure-bridge Skip vmbr3 creation/check
# --skip-sync-keys Skip sync-host-keys.sh (clan vars already exist)
# --skip-ipa-hosts Skip IPA host account creation (keytabs already exist)
# --skip-create-vms Skip VM creation (VMs already exist)
# --skip-add-hardware Skip NIC attachment (already attached)
# --skip-boot-wait Skip boot/SSH wait (VMs already running)
# --skip-refresh-sops-keys Skip sops host-key drift fix
# --skip-init-swarm Skip swarm initialisation (already initialised)
# --skip-dns Skip FreeIPA DNS record creation
# --skip-verify Skip post-deploy health checks
# --force-rebuild Pass --force-rebuild to create-proxmox-resource.sh
# --destroy Stop and delete both VMs (skip all other phases)
# --dry-run Print what would run without executing
# -h|--help Show this message
#
# Prerequisites:
# - SSH access to the Proxmox node as $PROXMOX_SSH_USER (wayne).
# - sops age key in the standard location (used by sync-host-keys.sh).
# - SSH access to domain-controller.sweet.home as $PROXMOX_SSH_USER for DNS phase.
# - For --skip-sync-keys: clan vars already in vars/per-machine/proxmox-ha-docker-{1,2}/.
# - For --skip-ipa-hosts: secrets/ha-docker-{1,2}.keytab already exist and are committed.
set -euo pipefail
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
REPO_ROOT="$(cd "${SCRIPT_DIR}/../.." && pwd)"
# shellcheck source=../env.sh
source "${REPO_ROOT}/scripts/env.sh"
# ── Defaults ──────────────────────────────────────────────────────────────────
NODE="${PVE1_HOST}" # deploy.sh targets pve1 by default (authorised for this cluster)
VMID1=202
VMID2=203
STORAGE="${PROXMOX_STORAGE:-local-zfs}"
SWARM_BRIDGE="vmbr3"
STORAGE_BRIDGE="vmbr2"
MEMORY_MB=4096
CORES=4
SKIP_ENSURE_BRIDGE=false
SKIP_SYNC_KEYS=false
SKIP_IPA_HOSTS=false
SKIP_CREATE_VMS=false
SKIP_ADD_HARDWARE=false
SKIP_BOOT_WAIT=false
SKIP_REFRESH_SOPS_KEYS=false
SKIP_INIT_SWARM=false
SKIP_DNS=false
SKIP_VERIFY=false
FORCE_REBUILD=false
DESTROY=false
DRY_RUN=false
# ── Variables from repo (mirrors variables.nix) ───────────────────────────────
NODE1_HOST="ha-docker-1"
NODE2_HOST="ha-docker-2"
NODE1_LAN_IP="192.168.2.230"
NODE2_LAN_IP="192.168.2.231"
NODE1_SWARM_IP="192.168.30.230"
NODE2_SWARM_IP="192.168.30.231"
NODE1_STORAGE_IP="192.168.20.230"
NODE2_STORAGE_IP="192.168.20.231"
SWARM_CIDR="192.168.30.0/24"
STORAGE_CIDR="192.168.20.0/24"
STORAGE_ZONE="storage.home"
SWARM_ZONE="swarm.home"
SSH_USER="${PROXMOX_SSH_USER:-wayne}"
DC_HOST="${IPA_SERVER:-domain-controller.sweet.home}"
# ── Argument parsing ──────────────────────────────────────────────────────────
usage() {
sed -n '/^# Usage:/,/^[^#]/{ /^#/{ s/^# \?//; p } }' "$0"
exit "${1:-0}"
}
while [[ $# -gt 0 ]]; do
case "$1" in
--node) NODE="$2"; shift 2 ;;
--vmid1) VMID1="$2"; shift 2 ;;
--vmid2) VMID2="$2"; shift 2 ;;
--storage) STORAGE="$2"; shift 2 ;;
--swarm-bridge) SWARM_BRIDGE="$2"; shift 2 ;;
--storage-bridge) STORAGE_BRIDGE="$2"; shift 2 ;;
--memory) MEMORY_MB="$2"; shift 2 ;;
--cores) CORES="$2"; shift 2 ;;
--skip-ensure-bridge) SKIP_ENSURE_BRIDGE=true; shift ;;
--skip-sync-keys) SKIP_SYNC_KEYS=true; shift ;;
--skip-ipa-hosts) SKIP_IPA_HOSTS=true; shift ;;
--skip-create-vms) SKIP_CREATE_VMS=true; shift ;;
--skip-add-hardware) SKIP_ADD_HARDWARE=true; shift ;;
--skip-boot-wait) SKIP_BOOT_WAIT=true; shift ;;
--skip-refresh-sops-keys) SKIP_REFRESH_SOPS_KEYS=true; shift ;;
--skip-init-swarm) SKIP_INIT_SWARM=true; shift ;;
--skip-dns) SKIP_DNS=true; shift ;;
--skip-verify) SKIP_VERIFY=true; shift ;;
--force-rebuild) FORCE_REBUILD=true; shift ;;
--destroy) DESTROY=true; shift ;;
--dry-run) DRY_RUN=true; shift ;;
-h|--help) usage 0 ;;
*) echo "Unknown option: $1" >&2; usage 1 ;;
esac
done
# ── Helpers ───────────────────────────────────────────────────────────────────
log() { echo "==> $*"; }
logn() { echo " $*"; }
err() { echo "ERROR: $*" >&2; exit 1; }
run() {
if $DRY_RUN; then
echo "[dry-run] $*"
else
"$@"
fi
}
pve() {
if $DRY_RUN; then
echo "[dry-run] ssh ${SSH_USER}@${NODE} sudo $*"
else
ssh -i ~/.ssh/id_ed25519 "${SSH_USER}@${NODE}" "sudo $*"
fi
}
pve_check() {
# Read-only probe — always executes even in dry-run.
ssh -i ~/.ssh/id_ed25519 "${SSH_USER}@${NODE}" "sudo $*"
}
SWARM_USER="nixos"
n1() {
ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no -o ConnectTimeout=5 \
"${SWARM_USER}@${NODE1_LAN_IP}" "$@" 2>/dev/null
}
n2() {
ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no -o ConnectTimeout=5 \
"${SWARM_USER}@${NODE2_LAN_IP}" "$@" 2>/dev/null
}
dc() {
# Run ipa commands on domain-controller as $SSH_USER.
if $DRY_RUN; then
echo "[dry-run] ssh ${SSH_USER}@${DC_HOST} $*"
return 0
fi
ssh -i ~/.ssh/id_ed25519 "${SSH_USER}@${DC_HOST}" "$@"
}
wait_for_ssh() {
local ip="$1" label="$2"
if $DRY_RUN; then
logn "[dry-run] Skipping SSH wait for ${label} (${ip})"
return 0
fi
local deadline=$(( $(date +%s) + 300 ))
log "Waiting for SSH on ${label} (${ip}) — up to 5 min..."
while [[ $(date +%s) -lt $deadline ]]; do
if ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no -o ConnectTimeout=3 \
-o BatchMode=yes "${SWARM_USER}@${ip}" true 2>/dev/null; then
logn "${label} is up."
return 0
fi
sleep 5
done
err "Timed out waiting for SSH on ${label} (${ip})"
}
# ── Destroy mode ──────────────────────────────────────────────────────────────
if $DESTROY; then
log "Destroying Docker Swarm VMs (${VMID1}=${NODE1_HOST}, ${VMID2}=${NODE2_HOST}) on ${NODE}"
for vmid in "$VMID1" "$VMID2"; do
STATUS=$(pve "qm status ${vmid} 2>/dev/null" 2>/dev/null || true)
if echo "$STATUS" | grep -q "running"; then
log "Stopping VMID ${vmid}..."
pve "qm stop ${vmid} --skiplock 1"
sleep 5
fi
if $DRY_RUN || pve "qm config ${vmid} >/dev/null 2>&1"; then
log "Deleting VMID ${vmid}..."
run pve "qm destroy ${vmid} --purge 1"
else
logn "VMID ${vmid} not found — already gone."
fi
done
log "Done — swarm VMs destroyed."
exit 0
fi
# ── Phase 1: Ensure swarm bridge ──────────────────────────────────────────────
if ! $SKIP_ENSURE_BRIDGE; then
log "Phase 1: Ensuring swarm bridge ${SWARM_BRIDGE} on ${NODE}"
if pve_check "test -d /sys/class/net/${SWARM_BRIDGE}" &>/dev/null; then
logn "${SWARM_BRIDGE} already exists — skipping."
else
logn "Creating isolated internal bridge ${SWARM_BRIDGE} (no upstream port, ${SWARM_CIDR})"
BRIDGE_CONF="auto ${SWARM_BRIDGE}
iface ${SWARM_BRIDGE} inet manual
bridge-ports none
bridge-stp off
bridge-fd 0"
if $DRY_RUN; then
echo "[dry-run] Would write /etc/network/interfaces.d/${SWARM_BRIDGE}.conf and ifup it"
else
ssh -i ~/.ssh/id_ed25519 "${SSH_USER}@${NODE}" \
"echo '${BRIDGE_CONF}' | sudo tee /etc/network/interfaces.d/${SWARM_BRIDGE}.conf > /dev/null && sudo ifup ${SWARM_BRIDGE}"
logn "${SWARM_BRIDGE} created and brought up."
fi
fi
fi
# ── Phase 2: Sync host keys ───────────────────────────────────────────────────
if ! $SKIP_SYNC_KEYS; then
log "Phase 2: Syncing SSH host keys for both swarm targets"
for target in proxmox-ha-docker-1 proxmox-ha-docker-2; do
CLAN_DIR="${REPO_ROOT}/vars/per-machine/${target}/openssh"
if [[ -d "$CLAN_DIR" ]]; then
logn "Clan vars for ${target} already exist — skipping."
else
logn "Generating host keys for ${target}..."
run bash "${REPO_ROOT}/scripts/secrets/sync-host-keys.sh" "$target"
fi
done
fi
# ── Phase 3: IPA host accounts ────────────────────────────────────────────────
if ! $SKIP_IPA_HOSTS; then
log "Phase 3: Creating IPA host accounts and keytabs"
IPA_SCRIPT="${REPO_ROOT}/scripts/ipa/create-nixos-ipa-host-account.sh"
for host in "${NODE1_HOST}" "${NODE2_HOST}"; do
KEYTAB="${REPO_ROOT}/secrets/${host}.keytab"
if [[ -f "$KEYTAB" ]]; then
logn "Keytab for ${host} already exists — skipping."
else
logn "Creating IPA host account and keytab for ${host}..."
run bash "$IPA_SCRIPT" "$host"
fi
done
if ! $DRY_RUN; then
# Keytabs must be committed and pushed before VMs rebuild from Gitea.
CURRENT_BRANCH="$(git -C "$REPO_ROOT" rev-parse --abbrev-ref HEAD)"
logn "Committing keytabs and pushing to Gitea (branch: ${CURRENT_BRANCH})..."
(cd "${REPO_ROOT}" && \
git add secrets/ha-docker-1.keytab secrets/ha-docker-2.keytab .sops.yaml && \
git commit -m "secrets(ha-docker): add IPA keytabs for ha-docker-1 and ha-docker-2" || true && \
git push origin "${CURRENT_BRANCH}")
logn "Pushed."
fi
fi
# ── Phase 3.5: Prepare Proxmox node for building ─────────────────────────────
if ! $SKIP_CREATE_VMS && ! $DRY_RUN; then
CURRENT_BRANCH="$(git -C "$REPO_ROOT" rev-parse --abbrev-ref HEAD)"
local_ssh() { ssh -i ~/.ssh/id_ed25519 "${SSH_USER}@${NODE}" "$*"; }
if local_ssh "test -d /nix" &>/dev/null && ! local_ssh "test -w /nix" &>/dev/null; then
logn "/nix exists but not writable by ${SSH_USER} — fixing ownership with sudo..."
local_ssh "sudo chown -R ${SSH_USER} /nix"
logn "Done."
fi
unset -f local_ssh
# Use PROXMOX_REMOTE_REPO_DIR (from env.sh) so the build path is consistent
# with what create-proxmox-resource.sh will use. The default is
# /home/<user>/nixos (the standalone nixos repo clone on pve1), but can be
# overridden to e.g. /home/<user>/infrastructure/nixos when the infrastructure
# mono-repo is checked out on pve1 instead.
REMOTE_REPO="${PROXMOX_REMOTE_REPO_DIR:-/home/${SSH_USER}/nixos}"
if pve_check "test -d ${REMOTE_REPO}/.git" &>/dev/null; then
REMOTE_BRANCH=$(ssh -i ~/.ssh/id_ed25519 "${SSH_USER}@${NODE}" \
"cd ${REMOTE_REPO} && git rev-parse --abbrev-ref HEAD 2>/dev/null")
if [[ "$REMOTE_BRANCH" != "$CURRENT_BRANCH" ]]; then
logn "Remote repo (${REMOTE_REPO}) is on '${REMOTE_BRANCH}', switching to '${CURRENT_BRANCH}'..."
if ssh -i ~/.ssh/id_ed25519 "${SSH_USER}@${NODE}" \
"cd ${REMOTE_REPO} && git fetch origin && git checkout '${CURRENT_BRANCH}' && git pull --ff-only" 2>&1; then
logn "Done."
else
logn "WARNING: branch switch failed — proceeding anyway (create-proxmox-resource.sh will retry)"
fi
fi
else
logn "Remote repo ${REMOTE_REPO} not found on ${NODE} — create-proxmox-resource.sh will clone it."
fi
fi
# ── Phase 4: Create VMs ───────────────────────────────────────────────────────
if ! $SKIP_CREATE_VMS; then
log "Phase 4: Building and creating swarm VMs on ${NODE}"
CREATE="${REPO_ROOT}/scripts/proxmox/create-proxmox-resource.sh"
REBUILD_FLAG=""
$FORCE_REBUILD && REBUILD_FLAG="--force-rebuild"
for spec in "${VMID1}:${NODE1_HOST}:proxmox-ha-docker-1" "${VMID2}:${NODE2_HOST}:proxmox-ha-docker-2"; do
IFS=: read -r vmid host_name flake_target <<< "$spec"
log "Creating ${flake_target} (VMID ${vmid}) on ${NODE}..."
# --force-rebuild is always passed: create-proxmox-resource.sh only calls
# sync_remote_host_keys (which bakes the clan-var SSH key into the disk
# image) when it actually builds. Reusing a cached image skips that step
# and leaves the VM unable to decrypt sops secrets on first boot.
run bash "$CREATE" \
--type vm \
--host "$host_name" \
--vmid "$vmid" \
--node "$NODE" \
--storage "$STORAGE" \
--memory "$MEMORY_MB" \
--cores "$CORES" \
--force-rebuild
done
fi
# ── Phase 5: Add NICs and start VMs ──────────────────────────────────────────
if ! $SKIP_ADD_HARDWARE; then
log "Phase 5: Attaching storage (${STORAGE_BRIDGE}) and swarm (${SWARM_BRIDGE}) NICs"
for vmid in "$VMID1" "$VMID2"; do
logn "VMID ${vmid}: stopping to add NICs..."
pve "qm stop ${vmid} --skiplock 1 2>/dev/null; sleep 3" || true
logn "Adding net1 (${STORAGE_BRIDGE} — NFS storage)..."
pve "qm set ${vmid} --net1 virtio,bridge=${STORAGE_BRIDGE},firewall=0"
logn "Adding net2 (${SWARM_BRIDGE} — Docker Swarm)..."
pve "qm set ${vmid} --net2 virtio,bridge=${SWARM_BRIDGE},firewall=0"
logn "Starting VMID ${vmid}..."
pve "qm start ${vmid}"
done
fi
# ── Phase 6: Wait for SSH ─────────────────────────────────────────────────────
if ! $SKIP_BOOT_WAIT; then
log "Phase 6: Waiting for both nodes to come up on LAN IPs"
wait_for_ssh "$NODE1_LAN_IP" "$NODE1_HOST"
wait_for_ssh "$NODE2_LAN_IP" "$NODE2_HOST"
logn "Both nodes are SSHable."
sleep 10 # let systemd finish activation
fi
# ── Phase 7: Refresh sops host-key registrations ─────────────────────────────
#
# Disko builds raw disk images: each new VM boots with a freshly-generated SSH
# host key, not the one pre-seeded in clan vars. Scan the running VMs; if
# their ed25519 keys differ from the clan var, update the clan var, rewrite
# the .sops.yaml anchor, and re-encrypt all affected sops files.
if ! $SKIP_REFRESH_SOPS_KEYS; then
if $DRY_RUN; then
logn "[dry-run] Would scan VM host keys and refresh .sops.yaml / secrets if needed"
else
log "Phase 7: Refreshing sops host-key registrations (disko key drift fix)"
SOPS_UPDATED=false
for spec in \
"${NODE1_LAN_IP}:proxmox-ha-docker-1:${NODE1_HOST}" \
"${NODE2_LAN_IP}:proxmox-ha-docker-2:${NODE2_HOST}"; do
IFS=: read -r node_ip flake_target host_name <<< "$spec"
CLAN_PUB="${REPO_ROOT}/vars/per-machine/${flake_target}/openssh/ssh_host_ed25519_key.pub/value"
logn "Scanning ed25519 host key from ${host_name} (${node_ip})..."
RAW=$(ssh-keyscan -t ed25519 "${node_ip}" 2>/dev/null | grep -v "^#") || true
if [[ -z "$RAW" ]]; then
logn "WARNING: no ed25519 key returned for ${node_ip} — skipping"
continue
fi
SCANNED_TYPE=$(awk '{print $2}' <<< "$RAW")
SCANNED_KEY=$(awk '{print $3}' <<< "$RAW")
SCANNED_PUBKEY="${SCANNED_TYPE} ${SCANNED_KEY} ${host_name}"
CURRENT=$(tr -d '\n' < "$CLAN_PUB" 2>/dev/null || true)
if [[ "$SCANNED_PUBKEY" == "$CURRENT" ]]; then
logn "${host_name}: clan var matches running key — no update needed"
continue
fi
logn "${host_name}: key drift detected — updating clan var"
logn " old: ${CURRENT}"
logn " new: ${SCANNED_PUBKEY}"
echo "$SCANNED_PUBKEY" > "$CLAN_PUB"
SOPS_UPDATED=true
ANCHOR="${flake_target}"
NEW_AGE=$(echo "$SCANNED_PUBKEY" | \
nix run --quiet --no-warn-dirty nixpkgs#ssh-to-age 2>/dev/null)
[[ -z "$NEW_AGE" ]] && err "ssh-to-age produced no output for ${host_name}"
logn " new age key: ${NEW_AGE}"
sed -i "/&${ANCHOR} /s| age[a-z0-9]*$| ${NEW_AGE}|" "${REPO_ROOT}/.sops.yaml"
done
if $SOPS_UPDATED; then
logn "Running sops updatekeys on affected secrets..."
SOPS="nix run --quiet --no-warn-dirty nixpkgs#sops --"
(cd "${REPO_ROOT}" && \
$SOPS updatekeys -y secrets/common.yaml && \
$SOPS updatekeys -y secrets/ha-docker-1.keytab && \
$SOPS updatekeys -y secrets/ha-docker-2.keytab)
logn "Committing refreshed host keys and re-encrypted secrets..."
(cd "${REPO_ROOT}" && \
git add \
vars/per-machine/proxmox-ha-docker-1/openssh/ssh_host_ed25519_key.pub/value \
vars/per-machine/proxmox-ha-docker-2/openssh/ssh_host_ed25519_key.pub/value \
.sops.yaml \
secrets/common.yaml \
secrets/ha-docker-1.keytab \
secrets/ha-docker-2.keytab && \
git commit -m "secrets(ha-docker): refresh sops host-key registrations for new VM instances" || true)
logn "Sops keys refreshed and committed."
fi
fi
fi
# ── Phase 8: Initialise Docker Swarm ─────────────────────────────────────────
if ! $SKIP_INIT_SWARM; then
log "Phase 8: Initialising Docker Swarm"
if $DRY_RUN; then
logn "[dry-run] Would run: docker swarm init --advertise-addr ${NODE1_SWARM_IP} --data-path-addr ${NODE1_SWARM_IP} on ${NODE1_HOST}"
logn "[dry-run] Would join ${NODE2_HOST} as manager"
logn "[dry-run] Would label both nodes"
else
# Check if node1 is already a swarm manager.
if n1 "docker info --format '{{.Swarm.LocalNodeState}}'" 2>/dev/null | grep -qx "active"; then
logn "${NODE1_HOST} is already in a swarm — skipping init."
else
logn "Initialising swarm on ${NODE1_HOST} (advertise: ${NODE1_SWARM_IP})..."
n1 "docker swarm init \
--advertise-addr ${NODE1_SWARM_IP} \
--data-path-addr ${NODE1_SWARM_IP}"
logn "Swarm initialised on ${NODE1_HOST}."
fi
# Check if node2 is already joined.
if n2 "docker info --format '{{.Swarm.LocalNodeState}}'" 2>/dev/null | grep -qx "active"; then
logn "${NODE2_HOST} is already in the swarm — skipping join."
else
logn "Fetching manager join token from ${NODE1_HOST}..."
JOIN_TOKEN=$(n1 "docker swarm join-token manager -q")
[[ -z "$JOIN_TOKEN" ]] && err "Failed to get swarm manager join token from ${NODE1_HOST}"
logn "Joining ${NODE2_HOST} as manager (advertise: ${NODE2_SWARM_IP})..."
n2 "docker swarm join \
--token ${JOIN_TOKEN} \
--advertise-addr ${NODE2_SWARM_IP} \
--data-path-addr ${NODE2_SWARM_IP} \
${NODE1_SWARM_IP}:2377"
logn "${NODE2_HOST} joined as manager."
fi
# Label nodes for service placement constraints.
logn "Labelling swarm nodes..."
n1 "docker node update --label-add node=${NODE1_HOST} ${NODE1_HOST}" || true
n1 "docker node update --label-add node=${NODE2_HOST} ${NODE2_HOST}" || true
logn "Labels applied."
fi
fi
# ── Phase 9: DNS registration ─────────────────────────────────────────────────
if ! $SKIP_DNS; then
log "Phase 9: Registering DNS records in FreeIPA"
if $DRY_RUN; then
logn "[dry-run] Would create/verify ${SWARM_ZONE} zone and add A records"
else
# Check for and create the swarm.home zone if absent.
if ! dc "ipa dnszone-show ${SWARM_ZONE}" >/dev/null 2>&1; then
logn "Creating ${SWARM_ZONE} DNS zone..."
dc "ipa dnszone-add ${SWARM_ZONE} \
--name-server=${DC_HOST}. \
--admin-email=hostmaster@${SWARM_ZONE}"
# Reverse zone for 192.168.30.x
dc "ipa dnszone-add 30.168.192.in-addr.arpa \
--name-server=${DC_HOST}. \
--admin-email=hostmaster@${SWARM_ZONE}" 2>/dev/null || \
logn " (reverse zone 30.168.192.in-addr.arpa already exists or skipped)"
else
logn "${SWARM_ZONE} zone already exists."
fi
# storage.home A records (zone already exists from HA cluster setup).
for spec in "${NODE1_HOST}:${NODE1_STORAGE_IP}" "${NODE2_HOST}:${NODE2_STORAGE_IP}"; do
IFS=: read -r hostname ip <<< "$spec"
logn "Adding ${hostname}.${STORAGE_ZONE}${ip}"
dc "ipa dnsrecord-add ${STORAGE_ZONE} ${hostname} --a-rec=${ip} --a-create-reverse" 2>/dev/null || \
logn " (record already exists or reverse zone missing — continuing)"
done
# swarm.home A records.
for spec in "${NODE1_HOST}:${NODE1_SWARM_IP}" "${NODE2_HOST}:${NODE2_SWARM_IP}"; do
IFS=: read -r hostname ip <<< "$spec"
logn "Adding ${hostname}.${SWARM_ZONE}${ip}"
dc "ipa dnsrecord-add ${SWARM_ZONE} ${hostname} --a-rec=${ip} --a-create-reverse" 2>/dev/null || \
logn " (record already exists — continuing)"
done
fi
fi
# ── Phase 10: Verify ──────────────────────────────────────────────────────────
if ! $SKIP_VERIFY; then
log "Phase 10: Verifying swarm health"
if $DRY_RUN; then
logn "[dry-run] Would verify swarm node list and NFS mounts"
else
logn "Swarm node list:"
n1 "docker node ls" || err "docker node ls failed on ${NODE1_HOST}"
logn "Checking swarm state on both nodes..."
for spec in "${NODE1_LAN_IP}:${NODE1_HOST}" "${NODE2_LAN_IP}:${NODE2_HOST}"; do
IFS=: read -r ip hostname <<< "$spec"
STATE=$(ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no \
"${SWARM_USER}@${ip}" "docker info --format '{{.Swarm.LocalNodeState}}'" 2>/dev/null)
if [[ "$STATE" != "active" ]]; then
err "${hostname} swarm state is '${STATE}', expected 'active'"
fi
logn " ${hostname}: swarm=${STATE}"
done
logn "Checking NFS mounts on both nodes..."
for spec in "${NODE1_LAN_IP}:${NODE1_HOST}" "${NODE2_LAN_IP}:${NODE2_HOST}"; do
IFS=: read -r ip hostname <<< "$spec"
NFS_OK=$(ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no \
"${SWARM_USER}@${ip}" "df -h /mnt/docker/config 2>/dev/null | grep -c nfs || echo 0" 2>/dev/null)
if [[ "$NFS_OK" -ge 1 ]]; then
logn " ${hostname}: /mnt/docker/config NFS mount ✓"
else
logn " WARNING: ${hostname}: /mnt/docker/config does not appear to be NFS-mounted"
logn " (automount may still be pending — try: ssh nixos@${ip} 'ls /mnt/docker/config')"
fi
done
logn "Checking overlay network..."
NETWORKS=$(n1 "docker network ls --filter driver=overlay --format '{{.Name}}'")
if echo "$NETWORKS" | grep -q "ingress"; then
logn " ingress overlay network present ✓"
else
logn " WARNING: ingress overlay network not found — swarm may not be fully initialised"
fi
fi
fi
log "Deploy complete. Both nodes are ready for 'docker stack deploy'."
log "Connect to either manager:"
log " ssh nixos@${NODE1_LAN_IP} (${NODE1_HOST})"
log " ssh nixos@${NODE2_LAN_IP} (${NODE2_HOST})"
+2 -19
View File
@@ -22,7 +22,7 @@
: "${PVE1_HOST:=pve1.sweet.home}" : "${PVE1_HOST:=pve1.sweet.home}"
: "${PVE_TEST_HOST:=pve-test.sweet.home}" : "${PVE_TEST_HOST:=pve-test.sweet.home}"
: "${PROXMOX_HOST:=$PVE1_HOST}" : "${PROXMOX_HOST:=$PVE1_HOST}"
: "${PROXMOX_SSH_USER:=wayne}" : "${PROXMOX_SSH_USER:=root}"
# Where this flake repo lives on the Proxmox node itself. # Where this flake repo lives on the Proxmox node itself.
# scripts/proxmox/create-proxmox-resource.sh builds images directly on the node # scripts/proxmox/create-proxmox-resource.sh builds images directly on the node
@@ -30,7 +30,7 @@
# (from this checkout's own `origin` remote) the first time it doesn't # (from this checkout's own `origin` remote) the first time it doesn't
# find it, installing build tooling via scripts/codex-setup.sh, then # find it, installing build tooling via scripts/codex-setup.sh, then
# `git pull`s it before every subsequent build. # `git pull`s it before every subsequent build.
: "${PROXMOX_REMOTE_REPO_DIR:=/home/${PROXMOX_SSH_USER}/nixos}" : "${PROXMOX_REMOTE_REPO_DIR:=/root/nixos}"
# Storage pool names -- Proxmox's own stock-install defaults, but this # Storage pool names -- Proxmox's own stock-install defaults, but this
# varies a lot by setup (ZFS pool name, custom LVM-thin volume, etc.). # varies a lot by setup (ZFS pool name, custom LVM-thin volume, etc.).
@@ -82,23 +82,6 @@ export PVE1_HOST PVE_TEST_HOST PROXMOX_HOST PROXMOX_SSH_USER PROXMOX_STORAGE \
: "${NIX_CACHE_HOST:=nix-cache}" : "${NIX_CACHE_HOST:=nix-cache}"
export NIX_CACHE_HOST export NIX_CACHE_HOST
# Matches variables.nix's lanDomain (the Gitea host this flake's own repo
# is served from -- see scripts/installer/auto-install.sh's FLAKE_BASE_URL)
# -- update both if it ever changes.
: "${LAN_DOMAIN:=gitea.lan.ddnsgeek.com}"
export LAN_DOMAIN
# Matches variables.nix's homeDomain -- the base LAN domain for service
# subdomains, FreeIPA Kerberos realm, and host FQDNs.
: "${HOME_DOMAIN:=sweet.home}"
export HOME_DOMAIN
# Matches variables.nix's ipaServer -- the FreeIPA server hostname.
# scripts/ipa/create-nixos-ipa-host-account.sh SSHes here to run
# ipa host-add and ipa-getkeytab.
: "${IPA_SERVER:=domain-controller.sweet.home}"
export IPA_SERVER
# nix_extra_opts: call as a plain statement (NOT inside $(...)/<(...) -- # nix_extra_opts: call as a plain statement (NOT inside $(...)/<(...) --
# that forks a subshell, and the whole point is exporting a decision back # that forks a subshell, and the whole point is exporting a decision back
# into *this* shell) to populate the global NIX_OPTS array with whatever # into *this* shell) to populate the global NIX_OPTS array with whatever
-237
View File
@@ -1,237 +0,0 @@
#!/usr/bin/env bash
# gc-hosts.sh — Run nix-collect-garbage -d on all live NixOS hosts.
#
# The host list is rebuilt on every run:
# 1. This workstation (nixos) — always first
# 2. pve1 — always second (non-NixOS Proxmox node with Nix installed)
# 3. Every NixOS guest currently running on pve1 (discovered via pct/qm list)
#
# nix-cache is excluded: gc-ing the shared binary cache evicts store paths
# that other hosts depend on for substitution.
#
# NixOS hosts: tries "sudo -n nix-collect-garbage -d" first (works when
# wheelNeedsPassword = false, e.g. the HA cluster). Falls back to user-level
# "nix-collect-garbage -d" if sudo needs a password — still collects
# unreferenced store paths and old nixos-user profile generations, but leaves
# old system generations in place.
# pve1: runs "bash -l -c nix-collect-garbage -d" as the login user so
# /etc/profile is sourced and the Nix daemon's PATH is set up automatically.
#
# Usage (from repo root):
# bash scripts/gc-hosts.sh [--dry-run]
set -euo pipefail
cd "$(dirname "$0")/.."
source scripts/env.sh 2>/dev/null || true
source scripts/lib/nix-eval.sh 2>/dev/null || true
# ── config ────────────────────────────────────────────────────────────────────
: "${MAX_JOBS:=8}"
: "${NIXOS_USER:=nixos}"
: "${PVE1_SSH_USER:=${PROXMOX_SSH_USER:-wayne}}"
# GC connections use BatchMode — no interactive prompts, just succeed or fail.
SSH_OPTS=(-o StrictHostKeyChecking=no -o BatchMode=yes -o ConnectTimeout=10)
# Discovery connections do NOT use BatchMode so that sudo can prompt if needed
# (pct/qm list require root access on Proxmox).
SSH_QUERY_OPTS=(-o StrictHostKeyChecking=no -o ConnectTimeout=10)
DRY_RUN=0
for arg in "$@"; do
case "$arg" in
--dry-run) DRY_RUN=1 ;;
*) echo "Unknown option: $arg" >&2; exit 1 ;;
esac
done
# ── build the host list ───────────────────────────────────────────────────────
# ORDERED_HOSTS: names in display/execution order.
# HOST_TARGET[name]: SSH target string (user@host).
# HOST_TYPE[name]: "nixos" (try sudo gc, fallback user) | "nix" (login-shell gc).
declare -a ORDERED_HOSTS=()
declare -A HOST_TARGET=()
declare -A HOST_TYPE=()
declare -A _SEEN_HOSTNAMES=() # dedup tracker
_add_host() {
local name="$1" target="$2" type="$3"
if [[ -n "${_SEEN_HOSTNAMES[$name]+_}" ]]; then return; fi
_SEEN_HOSTNAMES[$name]=1
ORDERED_HOSTS+=("$name")
HOST_TARGET[$name]="$target"
HOST_TYPE[$name]="$type"
}
# 1. Workstation (hard-wired first)
_add_host "nixos" "${NIXOS_USER}@nixos" "nixos"
# 2. pve1 (hard-wired second; non-NixOS, no system generations)
_add_host "pve1" "${PVE1_SSH_USER}@${PVE1_HOST}" "nix"
# 3. Dynamically discover running NixOS guests on pve1
#
# create-proxmox-resource.sh names every guest after its NixOS hostname:
# pct create ... --hostname <nixos-hostname> (LXC)
# qm create ... --name <nixos-hostname> (VM)
# So pct/qm list output already contains the NixOS hostname directly.
# We validate against the flake to filter out non-NixOS guests on pve1
# (e.g. FreeIPA, Proxmox Backup Server) that share the same Proxmox node.
echo "Discovering running guests on ${PVE1_HOST}..."
# Eval the flake once to get the set of hostnames that are actually NixOS.
# Values are NixOS hostnames (e.g. "docker"); keys are flake targets ("lxc-docker").
nixos_hostnames=""
nixos_hostnames="$(
nix eval --json "${NIX_EVAL_FLAGS[@]}" .#nixosConfigurations \
--apply 'cfgs: builtins.attrValues (builtins.mapAttrs (_: cfg: cfg.config.networking.hostName) cfgs)' \
2>/dev/null | jq -r '.[]' | sort -u
)" || { echo " warning: flake eval failed — non-NixOS guests will not be filtered" >&2; }
# SSH_QUERY_OPTS (no BatchMode) so sudo can prompt if needed for pct/qm.
if ssh "${SSH_QUERY_OPTS[@]}" "${PVE1_SSH_USER}@${PVE1_HOST}" "true" 2>/dev/null; then
running_guests="$(
ssh "${SSH_QUERY_OPTS[@]}" "${PVE1_SSH_USER}@${PVE1_HOST}" bash -s <<'DISCOVER'
sudo pct list 2>/dev/null | awk 'NR>1 && $2=="running" { print $NF }'
sudo qm list 2>/dev/null | awk 'NR>1 && $3=="running" { print $2 }'
DISCOVER
)" || running_guests=""
while IFS= read -r hostname; do
[[ -z "$hostname" ]] && continue
# Exclude nix-cache.
case "$hostname" in *nix-cache*) continue ;; esac
# Skip if not a flake-managed NixOS host (filters non-NixOS pve1 guests).
if [[ -n "$nixos_hostnames" ]] && ! grep -qxF "$hostname" <<< "$nixos_hostnames"; then
continue
fi
# Skip if already in the list (e.g. a proxmox-gui guest whose hostname is nixos).
if [[ -n "${_SEEN_HOSTNAMES[$hostname]+_}" ]]; then continue; fi
echo " + $hostname"
_add_host "$hostname" "${NIXOS_USER}@${hostname}" "nixos"
done <<< "$(echo "$running_guests" | sort -u)"
else
echo " warning: ${PVE1_HOST} unreachable — skipping dynamic host discovery" >&2
fi
echo ""
echo "Hosts: ${ORDERED_HOSTS[*]}"
echo ""
# ── dry-run ───────────────────────────────────────────────────────────────────
if [[ "$DRY_RUN" -eq 1 ]]; then
echo "[dry-run] commands that would run:"
for host in "${ORDERED_HOSTS[@]}"; do
target="${HOST_TARGET[$host]}"
type="${HOST_TYPE[$host]}"
if [[ "$type" == "nixos" ]]; then
echo " ssh ${SSH_OPTS[*]} $target 'sudo -n nix-collect-garbage -d'"
echo " # fallback: ssh ... $target 'nix-collect-garbage -d'"
else
echo " ssh ${SSH_OPTS[*]} $target 'bash -l -c nix-collect-garbage -d'"
fi
done
exit 0
fi
# ── gc worker ─────────────────────────────────────────────────────────────────
gc_one() {
local host="$1" target="${HOST_TARGET[$1]}" type="${HOST_TYPE[$1]}" logfile="$2"
if ! ssh "${SSH_OPTS[@]}" "$target" "true" 2>>"$logfile"; then
echo "unreachable"; return
fi
if [[ "$type" == "nixos" ]]; then
if ssh "${SSH_OPTS[@]}" "$target" "sudo -n nix-collect-garbage -d" \
>>"$logfile" 2>>"$logfile"; then
echo "ok(sudo)"; return
fi
echo "[sudo needs password — falling back to user-level gc]" >>"$logfile"
if ssh "${SSH_OPTS[@]}" "$target" "nix-collect-garbage -d" \
>>"$logfile" 2>>"$logfile"; then
echo "ok(user)"; return
fi
else
# Non-NixOS node: use a login shell so /etc/profile is sourced and the
# Nix daemon's bin dir is on PATH (set up by /etc/profile.d/nix-daemon.sh
# which the Nix installer adds to /etc/profile).
if ssh "${SSH_OPTS[@]}" "$target" "bash -l -c 'nix-collect-garbage -d'" \
>>"$logfile" 2>>"$logfile"; then
echo "ok"; return
fi
fi
echo "failed:$?"
}
# ── parallel execution ────────────────────────────────────────────────────────
echo "Running gc on ${#ORDERED_HOSTS[@]} hosts (up to ${MAX_JOBS} parallel)..."
echo ""
TMPDIR_GC="$(mktemp -d)"
trap 'rm -rf "$TMPDIR_GC"' EXIT
declare -A LOGS=()
job_count=0
for host in "${ORDERED_HOSTS[@]}"; do
logfile="${TMPDIR_GC}/${host}.log"
resultfile="${TMPDIR_GC}/${host}.result"
LOGS[$host]="$logfile"
: > "$logfile"
( result="$(gc_one "$host" "$logfile")"; echo "$result" > "$resultfile" ) &
(( job_count++ )) || true
if [[ "$job_count" -ge "$MAX_JOBS" ]]; then
wait -n 2>/dev/null || wait
(( job_count-- )) || true
fi
done
wait
# ── summary ───────────────────────────────────────────────────────────────────
echo "Results:"
echo "──────────────────────────────"
ok_hosts=()
warn_hosts=()
fail_hosts=()
for host in "${ORDERED_HOSTS[@]}"; do
result="$(cat "${TMPDIR_GC}/${host}.result" 2>/dev/null || echo "failed:missing")"
case "$result" in
ok|"ok(sudo)"|"ok(user)")
printf " %-22s %s\n" "$host" "$result"
ok_hosts+=("$host") ;;
unreachable)
printf " %-22s UNREACHABLE\n" "$host"
warn_hosts+=("$host") ;;
*)
printf " %-22s FAILED (%s)\n" "$host" "$result"
fail_hosts+=("$host") ;;
esac
done
echo ""
echo " ${#ok_hosts[@]} succeeded, ${#warn_hosts[@]} unreachable, ${#fail_hosts[@]} failed"
for host in "${warn_hosts[@]+"${warn_hosts[@]}"}" "${fail_hosts[@]+"${fail_hosts[@]}"}"; do
logfile="${LOGS[$host]}"
if [[ -s "$logfile" ]]; then
echo ""
echo "── $host ──"
cat "$logfile"
fi
done
echo ""
[[ "${#fail_hosts[@]}" -eq 0 ]]
-231
View File
@@ -1,231 +0,0 @@
#!/usr/bin/env bash
# acceptance-tests.sh — HA cluster acceptance tests (T1T7)
#
# Run from a host with SSH access to both HA nodes (or from node1 itself).
# All 7 tests must pass before considering the cluster production-ready.
# Test values below must match variables.nix haServer* values.
set -euo pipefail
# ── Configuration ─────────────────────────────────────────────────────────
# All values override-able via environment variables; defaults match variables.nix.
NODE1="${NODE1:-ha-server-1}"
NODE2="${NODE2:-ha-server-2}"
NODE1_IP="${NODE1_IP:-192.168.2.228}" # vars.haServer1Ip
NODE2_IP="${NODE2_IP:-192.168.2.227}" # vars.haServer2Ip
VIP="${VIP:-192.168.20.229}" # vars.haServerVip
XFS_MOUNT="${XFS_MOUNT:-/srv/ha-data}" # vars.haStorageRoot
ISCSI_IQN="${ISCSI_IQN:-iqn.2026-01.home.sweet:ha-storage}" # vars.haIscsiIqn
# ──────────────────────────────────────────────────────────────────────────
PASS=0
FAIL=0
RESULTS=()
# Use PASS=$((PASS+1)) instead of ((PASS++)) — the latter evaluates to 0 when
# PASS=0, which triggers set -e and kills the script after the very first PASS.
pass() { echo " PASS: $1"; PASS=$((PASS+1)); RESULTS+=("PASS $1"); }
fail() { echo " FAIL: $1"; FAIL=$((FAIL+1)); RESULTS+=("FAIL $1"); }
HA_USER="nixos"
n1() { ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no -o ConnectTimeout=5 "${HA_USER}@${NODE1_IP}" sudo "$@" 2>/dev/null; }
n2() { ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no -o ConnectTimeout=5 "${HA_USER}@${NODE2_IP}" sudo "$@" 2>/dev/null; }
echo "════════════════════════════════════════════════════"
echo " HA Cluster Acceptance Tests — $(date '+%Y-%m-%d %H:%M:%S')"
echo "════════════════════════════════════════════════════"
# ── Pre-flight: DRBD sync must be complete ────────────────────────────────
# Tests that check disk state, XFS mount, and iSCSI will fail or give false
# results while the initial full sync is in progress. Block until done.
echo ""
echo "Pre-flight: verifying DRBD sync is complete..."
DRBD_PREFLIGHT=$(n1 "drbdadm dstate ha-data 2>/dev/null" 2>/dev/null || echo "unknown")
if ! echo "$DRBD_PREFLIGHT" | grep -q "^UpToDate/UpToDate$"; then
echo ""
echo " ERROR: DRBD initial sync not complete."
echo " Current dstate on $NODE1: $DRBD_PREFLIGHT"
echo ""
echo " Monitor progress:"
echo " ssh nixos@$NODE1_IP 'sudo watch -n3 cat /proc/drbd'"
echo ""
echo " Re-run this script once dstate shows UpToDate/UpToDate."
exit 1
fi
echo " dstate: $DRBD_PREFLIGHT — ready."
# ── Detect Active/Standby nodes ────────────────────────────────────────────
# Use crm_mon to detect which node holds the Promoted (Primary) DRBD resource.
# Pacemaker is authoritative; DRBD role can briefly read as Secondary while
# Pacemaker is mid-transition, giving a false Active/Standby swap.
# Wait up to 90 s for Pacemaker to settle before giving up.
echo ""
echo "Detecting Active/Standby nodes (waiting for Pacemaker to settle)..."
ACTIVE_NODE=""
for i in $(seq 1 30); do
# crm_mon -1 output contains "Promoted: [ <node> ]" for the DRBD master.
CRM_OUT=$(n1 "crm_mon -1" 2>/dev/null || n2 "crm_mon -1" 2>/dev/null || true)
# crm_mon 2.x formats Promoted lines as " * Promoted: [ node ]" — the * bullet
# means ^\s*(Promoted|Masters): never matches; filter Unpromoted first instead.
ACTIVE_NODE=$(echo "$CRM_OUT" | grep -v 'Unpromoted\|Unmanaged' | grep -E '(Promoted|Masters):' | grep -oE '\b(ha-server-[0-9]+)\b' | head -1 || true)
[[ -n "$ACTIVE_NODE" ]] && break
sleep 3
done
if [[ -z "$ACTIVE_NODE" ]]; then
echo " WARNING: could not determine Active node from crm_mon after 90 s — defaulting to $NODE1"
ACTIVE_NODE="$NODE1"
fi
if [[ "$ACTIVE_NODE" == "$NODE1" ]]; then
ACTIVE_IP="$NODE1_IP"
STANDBY_NODE="$NODE2"; STANDBY_IP="$NODE2_IP"
na() { n1 "$@"; }
ns() { n2 "$@"; }
else
ACTIVE_IP="$NODE2_IP"
STANDBY_NODE="$NODE1"; STANDBY_IP="$NODE1_IP"
na() { n2 "$@"; }
ns() { n1 "$@"; }
fi
echo " Active: $ACTIVE_NODE ($ACTIVE_IP)"
echo " Standby: $STANDBY_NODE ($STANDBY_IP)"
# ── T1: Corosync quorum established ──────────────────────────────────────
echo ""
echo "[T1] Corosync quorum"
if na "corosync-quorumtool -s" 2>/dev/null | grep -q "Quorate:.*Yes"; then
pass "cluster has quorum"
else
fail "cluster does not have quorum — check corosync on both nodes"
fi
# ── T2: DRBD Primary on Active node, Secondary on Standby ────────────────
echo ""
echo "[T2] DRBD roles"
DRBD_ROLE=$(na "drbdadm role ha-data" 2>/dev/null || echo "unknown")
if [[ "$DRBD_ROLE" == "Primary/Secondary" || "$DRBD_ROLE" == "Primary" ]]; then
pass "DRBD Primary on $ACTIVE_NODE ($DRBD_ROLE)"
else
fail "unexpected DRBD role on $ACTIVE_NODE: $DRBD_ROLE (expected Primary/Secondary)"
fi
DRBD_DSTATE=$(na "drbdadm dstate ha-data" 2>/dev/null || echo "unknown")
if echo "$DRBD_DSTATE" | grep -q "UpToDate"; then
pass "DRBD disk state UpToDate ($DRBD_DSTATE)"
else
fail "DRBD disk not UpToDate: $DRBD_DSTATE"
fi
# ── T3: XFS mounted at haStorageRoot on the Active node ──────────────────
echo ""
echo "[T3] XFS mount"
if na "mountpoint -q '${XFS_MOUNT}'" 2>/dev/null; then
pass "XFS mounted at ${XFS_MOUNT} on $ACTIVE_NODE"
else
fail "XFS not mounted at ${XFS_MOUNT} on $ACTIVE_NODE"
fi
if ns "mountpoint -q '${XFS_MOUNT}'" 2>/dev/null; then
fail "XFS unexpectedly mounted on $STANDBY_NODE (should only be on Active node)"
else
pass "XFS not mounted on $STANDBY_NODE (correct — Standby)"
fi
# ── T4: iSCSI target visible on Active node ───────────────────────────────
echo ""
echo "[T4] iSCSI target"
IQN_COUNT=$(na "bash -c 'ls /sys/kernel/config/target/iscsi/ 2>/dev/null | grep -c iqn || true'" 2>/dev/null || echo "0")
if [[ "$IQN_COUNT" -ge 1 ]]; then
pass "iSCSI IQN active on $ACTIVE_NODE ($IQN_COUNT target(s))"
else
fail "no iSCSI IQN active on $ACTIVE_NODE"
fi
# iSCSI port reachable from Standby node via VIP.
# Use bash TCP probe (no iscsiadm needed — just checks port 3260 is open).
if ns "bash -c 'echo >/dev/tcp/${VIP}/3260' 2>/dev/null"; then
pass "iSCSI port 3260 reachable from $STANDBY_NODE via VIP ${VIP}"
else
fail "iSCSI port 3260 not reachable from $STANDBY_NODE via ${VIP}"
fi
# ── T5: Failover — standby Active node, verify resources move to Standby ──
echo ""
echo "[T5] Failover (standby $ACTIVE_NODE)"
ACTIVE_CRMD_NAME=$(na "crm_node -n" 2>/dev/null || echo "")
na "crm_standby -N '${ACTIVE_CRMD_NAME}' -v on" 2>/dev/null || true
echo " Waiting up to 120 s for resources to move to $STANDBY_NODE..."
MOVED=false
for i in $(seq 1 120); do
if ns "mountpoint -q '${XFS_MOUNT}'" 2>/dev/null; then
MOVED=true
echo " Resources moved in ${i}s"
break
fi
sleep 1
done
if $MOVED; then
pass "XFS mounted on $STANDBY_NODE after failover"
IQN_ON_STANDBY=$(ns "bash -c 'ls /sys/kernel/config/target/iscsi/ 2>/dev/null | grep -c iqn || true'" 2>/dev/null || echo "0")
[[ "$IQN_ON_STANDBY" -ge 1 ]] \
&& pass "iSCSI target active on $STANDBY_NODE after failover" \
|| fail "iSCSI target NOT active on $STANDBY_NODE after failover"
else
fail "XFS did not mount on $STANDBY_NODE within 120 s — failover incomplete"
fi
# ── T6: Data integrity — file written post-failover readable ─────────────
echo ""
echo "[T6] Data integrity"
# Write a test file on the new Active (former Standby) and verify it.
# Use `echo | sudo tee` for the write: "echo ... > file" via bash -c has the
# redirect interpreted by the remote nixos shell (not sudo), so the file open
# runs as nixos and fails with EACCES on the root-owned XFS mount. Piping
# through sudo tee lets tee (running as root) open the file instead.
TEST_FILE="${XFS_MOUNT}/.acceptance-test-$$"
TEST_CONTENT="ha-acceptance-test-$(date +%s)"
echo "${TEST_CONTENT}" | ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no -o ConnectTimeout=5 "${HA_USER}@${STANDBY_IP}" sudo tee "${TEST_FILE}" > /dev/null 2>/dev/null || true
READBACK=$(ns cat "${TEST_FILE}" 2>/dev/null || echo "")
if [[ "$READBACK" == "$TEST_CONTENT" ]]; then
pass "test file written and read back correctly on $STANDBY_NODE"
else
fail "data integrity check failed (wrote: '$TEST_CONTENT', read: '$READBACK')"
fi
ns rm -f "${TEST_FILE}" 2>/dev/null || true
# ── T7: Node rejoin — un-standby original Active, verify cluster is healthy ─
echo ""
echo "[T7] Node rejoin"
na "crm_standby -N '${ACTIVE_CRMD_NAME}' -v off" 2>/dev/null || true
na "crm_resource --cleanup" 2>/dev/null || true
sleep 5
if na "corosync-quorumtool -s 2>/dev/null | grep -q 'Quorate:.*Yes'"; then
pass "$ACTIVE_NODE rejoined — cluster has quorum"
else
fail "$ACTIVE_NODE did not rejoin with quorum"
fi
DRBD_ROLE_AFTER=$(na "drbdadm role ha-data" 2>/dev/null || echo "unknown")
if echo "$DRBD_ROLE_AFTER" | grep -q "Secondary"; then
pass "$ACTIVE_NODE is DRBD Secondary after rejoin ($DRBD_ROLE_AFTER)"
else
fail "unexpected DRBD role on $ACTIVE_NODE after rejoin: $DRBD_ROLE_AFTER"
fi
# ── Summary ───────────────────────────────────────────────────────────────
echo ""
echo "════════════════════════════════════════════════════"
echo " Results: ${PASS} PASS, ${FAIL} FAIL"
echo "════════════════════════════════════════════════════"
for r in "${RESULTS[@]}"; do echo " $r"; done
echo ""
if [[ "$FAIL" -eq 0 ]]; then
echo "ALL PASS — cluster is production-ready."
exit 0
else
echo "SOME TESTS FAILED — investigate before deploying."
exit 1
fi
-86
View File
@@ -1,86 +0,0 @@
#!/usr/bin/env bash
# cluster-enable-stonith.sh — enable STONITH fence agent after the fence SSH
# key is deployed to both nodes and authorised on the Proxmox host.
#
# Run from ha-server-1 as root AFTER:
# - /etc/pacemaker/fence_pve_ssh exists on both nodes (chmod +x)
# (copy from scripts/ha/fence-pve-ssh.py)
# - /etc/fence-pve-ssh-key (SSH private key) exists on both nodes
# - The corresponding public key is in authorized_keys on PVE_HOST
# - VMID_NODE1 / VMID_NODE2 filled in below
set -euo pipefail
# ── Configuration ─────────────────────────────────────────────────────────
NODE1="ha-server-1"
NODE2="ha-server-2"
VMID_NODE1="" # FILL IN: Proxmox VMID for ha-server-1
VMID_NODE2="" # FILL IN: Proxmox VMID for ha-server-2
PVE_HOST="pve1.sweet.home"
PVE_USER="wayne"
FENCE_KEY="/etc/fence-pve-ssh-key"
FENCE_SCRIPT="/etc/pacemaker/fence_pve_ssh"
# ──────────────────────────────────────────────────────────────────────────
log() { echo "[stonith-setup] $*"; }
die() { echo "[stonith-setup] ERROR: $*" >&2; exit 1; }
[[ $(id -u) -eq 0 ]] || die "must run as root"
[[ -n "$VMID_NODE1" ]] || die "VMID_NODE1 not set — edit this script"
[[ -n "$VMID_NODE2" ]] || die "VMID_NODE2 not set — edit this script"
[[ -f "$FENCE_KEY" ]] || die "fence key not found at $FENCE_KEY"
[[ -f "$FENCE_SCRIPT" ]] || die "fence script not found at $FENCE_SCRIPT"
log "Verifying fence agent can reach ${PVE_HOST}..."
ssh -i "$FENCE_KEY" -o BatchMode=yes -o ConnectTimeout=10 \
-o StrictHostKeyChecking=no "${PVE_USER}@${PVE_HOST}" \
"sudo /usr/sbin/qm list" &>/dev/null \
|| die "Cannot SSH to ${PVE_USER}@${PVE_HOST} — check authorized_keys and sudo"
log "Fence agent SSH connectivity confirmed"
log "Creating Pacemaker STONITH resources..."
cibadmin --create --scope resources --xml-text "
<primitive id=\"stonith-${NODE1}\" class=\"stonith\" type=\"external/fence_pve_ssh\">
<instance_attributes id=\"stonith-${NODE1}-attrs\">
<nvpair id=\"stonith-${NODE1}-plug\" name=\"plug\" value=\"${NODE1}\"/>
<nvpair id=\"stonith-${NODE1}-pve-host\" name=\"pve_host\" value=\"${PVE_HOST}\"/>
<nvpair id=\"stonith-${NODE1}-pve-user\" name=\"pve_user\" value=\"${PVE_USER}\"/>
<nvpair id=\"stonith-${NODE1}-key-file\" name=\"key_file\" value=\"${FENCE_KEY}\"/>
<nvpair id=\"stonith-${NODE1}-vmid1\" name=\"vmid_node1\" value=\"${VMID_NODE1}\"/>
<nvpair id=\"stonith-${NODE1}-vmid2\" name=\"vmid_node2\" value=\"${VMID_NODE2}\"/>
<nvpair id=\"stonith-${NODE1}-host-list\" name=\"pcmk_host_list\" value=\"${NODE1}\"/>
</instance_attributes>
<operations>
<op id=\"stonith-${NODE1}-monitor\" name=\"monitor\" interval=\"30s\" timeout=\"30s\"/>
</operations>
</primitive>
" 2>/dev/null || true
cibadmin --create --scope resources --xml-text "
<primitive id=\"stonith-${NODE2}\" class=\"stonith\" type=\"external/fence_pve_ssh\">
<instance_attributes id=\"stonith-${NODE2}-attrs\">
<nvpair id=\"stonith-${NODE2}-plug\" name=\"plug\" value=\"${NODE2}\"/>
<nvpair id=\"stonith-${NODE2}-pve-host\" name=\"pve_host\" value=\"${PVE_HOST}\"/>
<nvpair id=\"stonith-${NODE2}-pve-user\" name=\"pve_user\" value=\"${PVE_USER}\"/>
<nvpair id=\"stonith-${NODE2}-key-file\" name=\"key_file\" value=\"${FENCE_KEY}\"/>
<nvpair id=\"stonith-${NODE2}-vmid1\" name=\"vmid_node1\" value=\"${VMID_NODE1}\"/>
<nvpair id=\"stonith-${NODE2}-vmid2\" name=\"vmid_node2\" value=\"${VMID_NODE2}\"/>
<nvpair id=\"stonith-${NODE2}-host-list\" name=\"pcmk_host_list\" value=\"${NODE2}\"/>
</instance_attributes>
<operations>
<op id=\"stonith-${NODE2}-monitor\" name=\"monitor\" interval=\"30s\" timeout=\"30s\"/>
</operations>
</primitive>
" 2>/dev/null || true
log "Enabling STONITH and restoring quorum policy..."
crm_attribute -t crm_config -n stonith-enabled -v true
crm_attribute -t crm_config -n no-quorum-policy -v stop
log "DRBD fencing mode must also be updated to resource-only (already the"
log "default in cluster-config.nix; confirm with: cat /etc/drbd.d/ha-data.conf)"
log "Testing fence agent..."
stonith_admin --list-devices && log "Fence devices listed successfully." \
|| warn "stonith_admin --list-devices failed — check config"
log "STONITH enabled. Cluster is now fully HA."
-483
View File
@@ -1,483 +0,0 @@
#!/usr/bin/env bash
# cluster-init.sh — one-time HA cluster initialisation script
#
# Run ONCE from ha-server-1 as root AFTER both VMs are booted and have SSH
# access. It:
# 1. Generates and distributes the corosync authkey
# 2. Waits for corosync quorum and pacemaker
# 3. Initialises DRBD metadata, promotes node1 to primary
# 4. Creates XFS on /dev/drbd0 and mounts it
# 5. Creates the directory tree and iSCSI LUN backing file
# 6. Configures LIO iSCSI target (file-backed LUN)
# 7. Configures Pacemaker resources: DRBD → XFS → iSCSI → NFS → VIP
#
# Prerequisites:
# - Both VMs booted with the ha-server config (nixos-rebuild done)
# - SSH key access from node1 to root@NODE2_IP
# - VMID_NODE1 / VMID_NODE2 filled in below (needed for STONITH setup;
# cluster starts without STONITH, which you enable separately via
# scripts/ha/cluster-enable-stonith.sh)
# - Run as root on ha-server-1
set -euo pipefail
# ── Configuration ─────────────────────────────────────────────────────────
# All values override-able via environment variables; defaults match variables.nix.
NODE1="${NODE1:-ha-server-1}"
NODE2="${NODE2:-ha-server-2}"
NODE1_IP="${NODE1_IP:-192.168.2.228}" # vars.haServer1Ip
NODE2_IP="${NODE2_IP:-192.168.2.227}" # vars.haServer2Ip
VIP="${VIP:-192.168.20.229}" # vars.haServerVip (storage-client, vmbr2, VLAN 20)
VIP_LAN="${VIP_LAN:-192.168.2.229}" # vars.haServerLanVip (LAN, vmbr0)
XFS_MOUNT="${XFS_MOUNT:-/srv/ha-data}" # vars.haStorageRoot
ISCSI_IQN="${ISCSI_IQN:-iqn.2026-01.home.sweet:ha-storage}" # vars.haIscsiIqn
ISCSI_LUN_FILE="${XFS_MOUNT}/iscsi-lun.img"
ISCSI_LUN_SIZE="10G"
DRBD_DEVICE="/dev/drbd0"
# DRBD backing disk — by-id path that resolves correctly on both nodes
# regardless of whether the OS-level name is sda or sdb (Proxmox VM disk
# ordering is not guaranteed). Matches haServerDrbdDisk in variables.nix.
# Override DRBD_DISK if your hardware uses a different controller/slot path.
DRBD_DISK="${DRBD_DISK:-/dev/disk/by-id/scsi-0QEMU_QEMU_HARDDISK_drive-scsi1}"
VMID_NODE1="${VMID_NODE1:-}" # set by deploy.sh; needed for STONITH
VMID_NODE2="${VMID_NODE2:-}"
PVE_HOST="${PVE_HOST:-pve1.sweet.home}"
PVE_USER="${PVE_USER:-wayne}"
# Inter-node SSH: HA_USER is the user to SSH as on NODE2; HA_KEY is the private
# key to use. Default is root-to-root (no key arg). deploy.sh sets HA_USER=nixos
# and HA_KEY=/tmp/cluster-init-key so the script works even when root-to-root SSH
# is not available.
HA_USER="${HA_USER:-root}"
HA_KEY="${HA_KEY:-}"
# NFS dataset subdirectories to create under XFS_MOUNT.
# Must mirror vars.nfsShares subpath values in variables.nix.
NFS_SUBDIRS=(
"docker/config"
"docker/volumes"
"docker/databases"
"docker/nextcloud-data"
"raspi/volumes"
"proxmox/iso"
"proxmox/lxc"
"pxe-boot/images"
)
# ──────────────────────────────────────────────────────────────────────────
log() { echo "[cluster-init] $*"; }
die() { echo "[cluster-init] ERROR: $*" >&2; exit 1; }
warn() { echo "[cluster-init] WARNING: $*" >&2; }
[[ $(id -u) -eq 0 ]] || die "must run as root"
[[ "$(hostname)" == "$NODE1" ]] || die "must run on $NODE1"
# NixOS may not include xfsprogs in root's PATH even when it's in the store.
# If mkfs.xfs is missing, search the Nix store for it.
if ! command -v mkfs.xfs &>/dev/null; then
_xfs_bin=$(find /nix/store -maxdepth 3 -name mkfs.xfs 2>/dev/null | head -1 | xargs dirname 2>/dev/null || true)
[[ -n "$_xfs_bin" ]] && export PATH="$_xfs_bin:$PATH" \
|| die "mkfs.xfs not found — add xfsprogs to ha-server.nix environment.systemPackages and rebuild"
fi
# drbdmeta lives alongside drbdadm but may not be in PATH when run via sudo.
if ! command -v drbdmeta &>/dev/null; then
_drbd_bin=$(dirname "$(command -v drbdadm)" 2>/dev/null || true)
[[ -n "$_drbd_bin" ]] && export PATH="$_drbd_bin:$PATH" \
|| die "drbdmeta not found — is drbd-utils in ha-server environment.systemPackages?"
fi
# Portable 16-hex-char UUID generator (no openssl required).
_rand_uuid() {
cat /proc/sys/kernel/random/uuid 2>/dev/null | tr -d '-' | cut -c1-16 | tr '[:lower:]' '[:upper:]'
}
# Inter-node SSH/SCP helpers — abstract over root-to-root vs nixos+sudo.
_SSH_OPTS="-o StrictHostKeyChecking=no -o ConnectTimeout=10"
[[ -n "$HA_KEY" ]] && _SSH_OPTS="-i $HA_KEY $_SSH_OPTS"
if [[ "$HA_USER" == "root" ]]; then
n2_ssh() { ssh $_SSH_OPTS "root@${NODE2_IP}" "$@"; }
n2_scp() { scp $_SSH_OPTS "$1" "root@${NODE2_IP}:$2"; }
else
# Non-root user with passwordless sudo; wrap each command with sudo.
n2_ssh() { ssh $_SSH_OPTS "${HA_USER}@${NODE2_IP}" sudo "$@"; }
n2_scp() {
# SCP to a tmp path, then sudo-move to the real destination as the remote user.
local src="$1" dst="$2"
local tmp="/tmp/_cluster_init_scp_$$"
scp $_SSH_OPTS "$src" "${HA_USER}@${NODE2_IP}:${tmp}"
ssh $_SSH_OPTS "${HA_USER}@${NODE2_IP}" sudo mv "${tmp}" "${dst}"
}
fi
# ── 0. Corosync authkey ───────────────────────────────────────────────────
AUTHKEY="/etc/corosync/authkey"
mkdir -p /etc/corosync
if [[ ! -f "$AUTHKEY" ]]; then
log "Generating corosync authkey..."
corosync-keygen -k "$AUTHKEY"
chmod 0400 "$AUTHKEY"
fi
log "Distributing authkey to $NODE2..."
n2_ssh "mkdir -p /etc/corosync"
n2_scp "$AUTHKEY" "$AUTHKEY"
n2_ssh "chmod 0400 '${AUTHKEY}'"
log "Restarting corosync and pacemaker on both nodes..."
systemctl restart corosync
n2_ssh "systemctl restart corosync"
sleep 3
log "Starting pacemaker on both nodes (may have failed at boot before authkey was placed)..."
systemctl start pacemaker 2>/dev/null || systemctl restart pacemaker 2>/dev/null || true
n2_ssh "systemctl start pacemaker 2>/dev/null || systemctl restart pacemaker 2>/dev/null || true"
sleep 2
# ── 1. Corosync quorum ────────────────────────────────────────────────────
log "Waiting for corosync quorum..."
for i in $(seq 1 30); do
if corosync-quorumtool -s 2>/dev/null | grep -q 'Quorate:.*Yes'; then
log "Quorum established"
break
fi
[[ $i -eq 30 ]] && die "corosync quorum not established after 60 s"
sleep 2
done
log "Waiting for pacemaker..."
for i in $(seq 1 30); do
if crm_mon -1 &>/dev/null; then
log "Pacemaker running"
break
fi
[[ $i -eq 30 ]] && die "pacemaker not running after 60 s"
sleep 2
done
# ── 2. DRBD initialisation ────────────────────────────────────────────────
# Put both nodes in Pacemaker standby first so it stops managed resources
# cleanly, then enable maintenance-mode so Pacemaker's monitor operations are
# suspended. Without maintenance-mode, Pacemaker keeps monitoring: when it
# sees DRBD Primary on a standby node (that it didn't start), it triggers a
# stop action — killing the initial sync after ~10 s. Maintenance-mode
# disables all start/stop/monitor actions for the duration of the sync; it is
# cleared after UpToDate/UpToDate is confirmed.
log "Setting both nodes to Pacemaker standby for DRBD metadata init..."
crm_standby -N "$NODE1" -v on 2>/dev/null || true
crm_standby -N "$NODE2" -v on 2>/dev/null || true
# Wait for Pacemaker to actually stop DRBD (if it was managing it).
log "Waiting for DRBD to stop under Pacemaker control..."
for i in $(seq 1 30); do
n1_role=$(drbdadm role ha-data 2>/dev/null || echo "Unconfigured")
n2_role=$(n2_ssh "drbdadm role ha-data 2>/dev/null" 2>/dev/null || echo "Unconfigured")
if [[ "$n1_role" == "Unconfigured" ]] && [[ "$n2_role" == "Unconfigured" ]]; then
log "DRBD stopped on both nodes"
break
fi
[[ $i -eq 30 ]] && warn "DRBD still active after 60s standby — forcing down anyway"
sleep 2
done
log "Enabling Pacemaker maintenance-mode (suspends monitor/start/stop during sync)..."
crm_attribute -t crm_config -n maintenance-mode -v true 2>/dev/null || true
log "Detaching DRBD on $NODE1 (belt-and-suspenders after standby)..."
drbdadm down ha-data 2>/dev/null || true
log "Detaching DRBD on $NODE2..."
n2_ssh "drbdadm down ha-data 2>/dev/null || true"
sleep 2
# Ensure /etc/drbd.conf on both nodes points to DRBD_DISK (the stable by-id
# path). VMs built before this fix may have /dev/sda or /dev/sdb hardcoded.
# NixOS makes /etc/drbd.conf a symlink into the read-only Nix store, so
# sed -i on the symlink target would fail — we break the symlink first with
# cp --remove-destination, creating a regular writable copy.
# Rebuild+redeploy (--force-rebuild) to make this permanent.
_PATCH_DRBD=$(mktemp)
cat > "$_PATCH_DRBD" << 'PATCHEOF'
#!/bin/bash
WANT="$1"
conf=/etc/drbd.conf
if [[ -L "$conf" ]]; then
cp --remove-destination "$(readlink -f "$conf")" "$conf"
fi
cur=$(drbdadm sh-ll-dev ha-data 2>/dev/null | head -1 || true)
if [[ -n "$cur" && "$cur" != "$WANT" ]]; then
echo "[cluster-init] WARNING: patching $conf: $cur → $WANT (rebuild to make permanent)"
sed -i "s,${cur},${WANT},g" "$conf"
fi
PATCHEOF
chmod +x "$_PATCH_DRBD"
bash "$_PATCH_DRBD" "$DRBD_DISK"
n2_scp "$_PATCH_DRBD" "/tmp/patch-drbd-disk.sh"
n2_ssh "bash /tmp/patch-drbd-disk.sh ${DRBD_DISK}"
n2_ssh "rm -f /tmp/patch-drbd-disk.sh"
rm -f "$_PATCH_DRBD"
log "Initialising DRBD metadata on $NODE1..."
# Use drbdmeta --force directly for BOTH create-md and write-dev-uuid.
# drbdadm create-md --force passes --force to drbdmeta create-md but NOT to
# the write-dev-uuid sub-call it makes internally, so write-dev-uuid fails when
# the backing disk is still busy and stdin is not a TTY:
# "stdin not a TTY, not waiting for confirmation" → exit 20.
# Calling drbdmeta --force directly bypasses the exclusive-open confirmation on
# both steps without needing a TTY, regardless of whether the device is busy.
# Skip metadata creation only if DRBD is UP and fully synced (UpToDate/UpToDate).
# When the resource is down, drbdadm dstate reads metadata and returns just
# "UpToDate" (no slash) — that must not be treated as "already synced".
# Mismatched UUIDs from an interrupted sync cause instant WFConnection→StandAlone,
# so we always recreate metadata unless the sync is genuinely complete.
if [[ "$(drbdadm dstate ha-data 2>/dev/null)" != "UpToDate/UpToDate" ]]; then
UUID1=$(_rand_uuid)
drbdmeta --force 0 v08 "${DRBD_DISK}" internal create-md
drbdmeta --force 0 v08 "${DRBD_DISK}" internal write-dev-uuid "$UUID1"
fi
log "Initialising DRBD metadata on $NODE2..."
if [[ "$(n2_ssh "drbdadm dstate ha-data 2>/dev/null" 2>/dev/null)" != "UpToDate/UpToDate" ]]; then
UUID2=$(n2_ssh "cat /proc/sys/kernel/random/uuid 2>/dev/null | tr -d '-' | cut -c1-16 | tr '[:lower:]' '[:upper:]'")
n2_ssh "drbdmeta --force 0 v08 ${DRBD_DISK} internal create-md"
n2_ssh "drbdmeta --force 0 v08 ${DRBD_DISK} internal write-dev-uuid ${UUID2}"
fi
log "Bringing up DRBD on both nodes..."
drbdadm up ha-data 2>/dev/null || true
n2_ssh "drbdadm up ha-data" 2>/dev/null || true
log "Forcing $NODE1 to DRBD Primary for initial sync..."
drbdadm primary ha-data --force
# NOTE: Pacemaker standby is intentionally kept ON until after the sync
# completes. Clearing it here races with the OCF DRBD agent: Pacemaker
# sees DRBD in WFConnection/SyncSource and may call drbdadm-down thinking
# something went wrong, killing the sync. Standby is cleared below, after
# UpToDate/UpToDate is confirmed.
log "Waiting for DRBD initial sync to complete (32 GB may take 1020 min)..."
log " (monitor with: watch -n3 cat /proc/drbd)"
_sync_chars=('|' '/' '-' $'\\')
_sync_iter=0
while true; do
_dstate=$(drbdadm dstate ha-data 2>/dev/null || echo "unknown")
if echo "$_dstate" | grep -q "UpToDate/UpToDate"; then
printf "\r%-80s\r" ""
log "DRBD initial sync complete (dstate: $_dstate)"
break
fi
# Parse connection state from /proc/drbd (cs:SyncSource, cs:Connected, cs:StandAlone …)
_cs=$(grep -oE 'cs:[A-Za-z]+' /proc/drbd 2>/dev/null | head -1 | sed 's/cs://' || echo "unknown")
# /proc/drbd uses variable whitespace: "sync'ed: 5.2%" (two spaces).
_pct=$(grep -oE "sync'ed:[[:space:]]+[0-9.]+" /proc/drbd 2>/dev/null | grep -oE "[0-9.]+" | head -1 || echo "")
_eta=$(grep -oE "finish:[[:space:]]+[0-9:]+" /proc/drbd 2>/dev/null | grep -oE "[0-9:]+$" | head -1 || echo "")
_spd=$(grep -oE "speed:[[:space:]]+[0-9,]+" /proc/drbd 2>/dev/null | grep -oE "[0-9,]+$" | head -1 || echo "")
_sync_iter=$(( _sync_iter + 1 ))
_sc="${_sync_chars[$_sync_iter % 4]}"
if [[ "$_cs" == "StandAlone" && $_sync_iter -gt 5 ]]; then
printf "\r%-80s\r" ""
die "DRBD is StandAlone after 15 s — peer connection lost (dstate: $_dstate). " \
"Check corosync/network and re-run cluster-init."
elif [[ -n "$_pct" ]]; then
printf "\r [%s] syncing: %s%% done — ETA %s @ %s K/s " \
"$_sc" "$_pct" "${_eta:-??:??:??}" "${_spd:-?}"
else
printf "\r [%s] cs:%s dstate:%s — waiting for sync to start " "$_sc" "$_cs" "$_dstate"
fi
sleep 3
done
log "Disabling Pacemaker maintenance-mode and clearing standby — handing DRBD back to Pacemaker..."
crm_attribute -t crm_config -n maintenance-mode -v false 2>/dev/null || true
crm_standby -N "$NODE1" -v off 2>/dev/null || true
crm_standby -N "$NODE2" -v off 2>/dev/null || true
# ── 3. XFS filesystem ─────────────────────────────────────────────────────
log "Creating XFS on ${DRBD_DEVICE}..."
if ! xfs_info "${DRBD_DEVICE}" &>/dev/null; then
mkfs.xfs -f "${DRBD_DEVICE}"
fi
log "Mounting ${DRBD_DEVICE} at ${XFS_MOUNT}..."
mkdir -p "${XFS_MOUNT}"
mountpoint -q "${XFS_MOUNT}" || mount "${DRBD_DEVICE}" "${XFS_MOUNT}"
# ── 4. NFS dataset directories ────────────────────────────────────────────
log "Creating NFS dataset directories..."
for subdir in "${NFS_SUBDIRS[@]}"; do
mkdir -p "${XFS_MOUNT}/${subdir}"
done
# ── 5. iSCSI LUN backing file ─────────────────────────────────────────────
log "Creating iSCSI LUN backing file ${ISCSI_LUN_FILE} (${ISCSI_LUN_SIZE})..."
if [[ ! -f "${ISCSI_LUN_FILE}" ]]; then
fallocate -l "${ISCSI_LUN_SIZE}" "${ISCSI_LUN_FILE}"
fi
# ── 6. LIO iSCSI target ───────────────────────────────────────────────────
log "Configuring LIO iSCSI target via targetcli..."
# Note: do NOT bind portal to ${VIP} here — the VIP isn't assigned yet (Pacemaker
# creates it). The default portal (all IPs, port 3260) is correct; Pacemaker's
# VIP resource will make the target reachable at the VIP address.
#
# Clear any existing LIO state first (idempotent: re-run after a partial failure).
# Use specific delete commands — clearconfig does not reliably clear kernel state.
if ls /sys/kernel/config/target/iscsi/ 2>/dev/null | grep -q "${ISCSI_IQN}"; then
log "Clearing existing LIO target ${ISCSI_IQN} before reconfiguration..."
targetcli "/iscsi delete ${ISCSI_IQN}" 2>/dev/null || true
fi
if ls /sys/kernel/config/target/core/ 2>/dev/null | grep -q "fileio"; then
log "Clearing existing LIO backstore ha-lun0 before reconfiguration..."
targetcli "/backstores/fileio delete ha-lun0" 2>/dev/null || true
fi
targetcli <<EOF
/backstores/fileio create name=ha-lun0 file_or_dev=${ISCSI_LUN_FILE} size=0 write_back=false
/iscsi create ${ISCSI_IQN}
/iscsi/${ISCSI_IQN}/tpg1/luns create /backstores/fileio/ha-lun0
/iscsi/${ISCSI_IQN}/tpg1 set attribute authentication=0
/iscsi/${ISCSI_IQN}/tpg1 set attribute demo_mode_write_protect=0
saveconfig /etc/target/saveconfig.json
EOF
log "Tearing down LIO kernel objects — Pacemaker will restore via targetctl on the Active node..."
# LIO holds the backing file open; clear kernel state now so the XFS unmount succeeds.
# Use specific delete commands (clearconfig does not reliably clear kernel configfs state).
targetcli "/iscsi delete ${ISCSI_IQN}" 2>/dev/null || warn "LIO iscsi delete failed — umount may fail"
targetcli "/backstores/fileio delete ha-lun0" 2>/dev/null || warn "LIO backstore delete failed"
log "Distributing iSCSI saveconfig to $NODE2..."
n2_scp /etc/target/saveconfig.json /etc/target/saveconfig.json
log "Unmounting ${XFS_MOUNT} — Pacemaker manages it..."
umount "${XFS_MOUNT}" || { sync; umount -l "${XFS_MOUNT}"; }
log "Demoting DRBD to Secondary — Pacemaker manages primary role..."
drbdadm role ha-data 2>/dev/null | grep -q "^Primary" && drbdadm secondary ha-data || true
# ── 7. Pacemaker resources ────────────────────────────────────────────────
log "Configuring Pacemaker cluster properties..."
crm_attribute -t crm_config -n stonith-enabled -v false
crm_attribute -t crm_config -n no-quorum-policy -v ignore
log "Creating Pacemaker resources via cibadmin..."
# Use cibadmin --replace with pacemaker-4.0-compatible XML.
# Key schema rules for pacemaker-4.0:
# - globally-unique must be in <meta_attributes>, not a direct <clone> attribute
# - promoted-max / promoted-node-max (not master-max / master-node-max)
# - constraint with-rsc-role="Promoted" (not "Master")
cibadmin --replace --scope resources --xml-text '<resources>
<clone id="ms-drbd0">
<meta_attributes id="ms-drbd0-meta">
<nvpair id="ms-drbd0-globally-unique" name="globally-unique" value="false"/>
<nvpair id="ms-drbd0-promotable" name="promotable" value="true"/>
<nvpair id="ms-drbd0-promoted-max" name="promoted-max" value="1"/>
<nvpair id="ms-drbd0-promoted-node-max" name="promoted-node-max" value="1"/>
<nvpair id="ms-drbd0-clone-max" name="clone-max" value="2"/>
<nvpair id="ms-drbd0-clone-node-max" name="clone-node-max" value="1"/>
<nvpair id="ms-drbd0-notify" name="notify" value="true"/>
<nvpair id="ms-drbd0-interleave" name="interleave" value="true"/>
</meta_attributes>
<primitive id="drbd0" class="ocf" type="drbd" provider="linbit">
<instance_attributes id="drbd0-attrs">
<nvpair id="drbd0-resource" name="drbd_resource" value="ha-data"/>
</instance_attributes>
<operations>
<op id="drbd0-start" name="start" interval="0" timeout="240s"/>
<op id="drbd0-stop" name="stop" interval="0" timeout="120s"/>
<op id="drbd0-promote" name="promote" interval="0" timeout="240s"/>
<op id="drbd0-demote" name="demote" interval="0" timeout="90s"/>
<op id="drbd0-monitor-promoted" name="monitor" interval="20s" timeout="20s" role="Promoted"/>
<op id="drbd0-monitor-unpromoted" name="monitor" interval="30s" timeout="20s" role="Unpromoted"/>
</operations>
</primitive>
</clone>
<group id="ha-group">
<primitive id="xfs-data" class="ocf" type="Filesystem" provider="heartbeat">
<instance_attributes id="xfs-data-attrs">
<nvpair id="xfs-data-device" name="device" value="/dev/drbd0"/>
<nvpair id="xfs-data-directory" name="directory" value="/srv/ha-data"/>
<nvpair id="xfs-data-fstype" name="fstype" value="xfs"/>
<nvpair id="xfs-data-options" name="options" value="defaults"/>
<nvpair id="xfs-data-force_unmount" name="force_unmount" value="true"/>
</instance_attributes>
<operations>
<op id="xfs-data-start" name="start" interval="0" timeout="60s"/>
<op id="xfs-data-stop" name="stop" interval="0" timeout="60s"/>
<op id="xfs-data-monitor" name="monitor" interval="20s" timeout="40s"/>
</operations>
</primitive>
<primitive id="iscsi-target" class="systemd" type="targetctl">
<operations>
<op id="iscsi-start" name="start" interval="0" timeout="60s"/>
<op id="iscsi-stop" name="stop" interval="0" timeout="60s"/>
<op id="iscsi-monitor" name="monitor" interval="20s" timeout="40s"/>
</operations>
</primitive>
<primitive id="nfs-server" class="systemd" type="nfs-server">
<operations>
<op id="nfs-start" name="start" interval="0" timeout="60s"/>
<op id="nfs-stop" name="stop" interval="0" timeout="60s"/>
<op id="nfs-monitor" name="monitor" interval="30s" timeout="40s"/>
</operations>
</primitive>
<primitive id="vip-storage" class="ocf" type="IPaddr2" provider="heartbeat">
<instance_attributes id="vip-storage-attrs">
<nvpair id="vip-storage-ip" name="ip" value="192.168.20.229"/>
<nvpair id="vip-storage-cidr" name="cidr_netmask" value="24"/>
<nvpair id="vip-storage-nic" name="nic" value="ens20"/>
</instance_attributes>
<operations>
<op id="vip-storage-start" name="start" interval="0" timeout="20s"/>
<op id="vip-storage-stop" name="stop" interval="0" timeout="20s"/>
<op id="vip-storage-monitor" name="monitor" interval="10s" timeout="20s"/>
</operations>
</primitive>
<primitive id="vip-lan" class="ocf" type="IPaddr2" provider="heartbeat">
<instance_attributes id="vip-lan-attrs">
<nvpair id="vip-lan-ip" name="ip" value="192.168.2.229"/>
<nvpair id="vip-lan-cidr" name="cidr_netmask" value="24"/>
<nvpair id="vip-lan-nic" name="nic" value="ens18"/>
</instance_attributes>
<operations>
<op id="vip-lan-start" name="start" interval="0" timeout="20s"/>
<op id="vip-lan-stop" name="stop" interval="0" timeout="20s"/>
<op id="vip-lan-monitor" name="monitor" interval="10s" timeout="20s"/>
</operations>
</primitive>
</group>
</resources>'
log "Adding Pacemaker ordering and colocation constraints..."
cibadmin --replace --scope constraints --xml-text '<constraints>
<rsc_order id="order-drbd-group" first="ms-drbd0" first-action="promote" then="ha-group" then-action="start" kind="Mandatory"/>
<rsc_colocation id="coloc-group-with-drbd" score="INFINITY" rsc="ha-group" with-rsc="ms-drbd0" with-rsc-role="Promoted"/>
</constraints>'
log "Clearing stale Pacemaker failure history..."
crm_resource --cleanup 2>/dev/null || true
log "Waiting for resources to start..."
for i in $(seq 1 60); do
if crm_resource -r vip-storage --locate 2>/dev/null | grep -q "running on"; then
log "VIPs are up: $(crm_resource -r vip-storage --locate)"
break
fi
[[ $i -eq 60 ]] && { warn "VIPs not up after 120 s — check: crm_mon -1"; break; }
sleep 2
done
log ""
log "═══════════════════════════════════════════════════════════════"
log " HA cluster initialised."
log ""
log " crm_mon -1 — cluster status"
log " iscsiadm -m discovery -t st -p ${VIP} — verify iSCSI (storage net)"
log " iscsiadm -m discovery -t st -p ${VIP_LAN} — verify iSCSI (LAN)"
log " showmount -e ${VIP} — verify NFS exports (storage net)"
log " showmount -e ${VIP_LAN} — verify NFS exports (LAN)"
log ""
log " To enable STONITH (after deploying fence SSH key):"
log " 1. Fill in VMID_NODE1 / VMID_NODE2 in cluster-enable-stonith.sh"
log " 2. Copy scripts/ha/fence-pve-ssh.py to /etc/pacemaker/fence_pve_ssh"
log " on both nodes (chmod +x)"
log " 3. Generate and distribute the fence SSH key"
log " (see docs or cluster-enable-stonith.sh header)"
log " 4. bash scripts/ha/cluster-enable-stonith.sh"
log "═══════════════════════════════════════════════════════════════"
-499
View File
@@ -1,499 +0,0 @@
#!/usr/bin/env bash
# deploy.sh — Full lifecycle management for the HA file-server cluster.
#
# Handles everything from zero (no VMs, no secrets) through a running,
# tested cluster, and optionally tears it back down.
#
# Usage:
# scripts/ha/deploy.sh [options]
# scripts/ha/deploy.sh --destroy [options]
#
# Phases (all run by default; skip any with --skip-*):
# 1. ensure-bridge Create storage bridge (vmbr1) on the Proxmox node if absent.
# 2. sync-keys Generate SSH host keys and register age keys for both nodes.
# 3. create-vms Build disk images and create both VMs via create-proxmox-resource.sh.
# 4. add-hardware Attach storage NIC (vmbr1) and DRBD data disk to each VM.
# 5. boot-wait Start VMs, wait for SSH on both nodes.
# 6. cluster-init Form the cluster: DRBD, corosync, Pacemaker, NFS, VIP.
# Also encrypts the generated corosync authkey into the repo.
# 7. run-tests Run acceptance tests (T1T7).
#
# Options:
# --node <host> Proxmox host to deploy on (default: pve1.sweet.home)
# --vmid1 <n> VMID for ha-server-1 (default: 200)
# --vmid2 <n> VMID for ha-server-2 (default: 201)
# --storage <pool> Proxmox storage pool (default: local-zfs)
# --storage-bridge <br> Bridge for HA storage network (default: vmbr1)
# --drbd-disk-gb <n> DRBD data disk size in GB (default: 32)
# --memory <MB> RAM per node (default: 4096)
# --cores <n> vCPUs per node (default: 4)
# --skip-ensure-bridge Skip storage bridge creation/check
# --skip-sync-keys Skip sync-host-keys.sh (clan vars already exist)
# --skip-create-vms Skip VM creation (VMs already exist)
# --skip-add-hardware Skip net1/scsi1 attachment (already attached)
# --skip-boot-wait Skip boot/SSH wait (VMs already running)
# --skip-refresh-sops-keys Skip scanning running VMs for fresh SSH host keys
# --skip-cluster-init Skip cluster formation (cluster already configured)
# --skip-tests Skip acceptance tests
# --force-rebuild Pass --force-rebuild to create-proxmox-resource.sh
# --destroy Stop and delete both VMs (skip all other phases)
# --dry-run Print what would run without executing
# -h|--help Show this message
#
# Prerequisites:
# - SSH access to the Proxmox node as $PROXMOX_SSH_USER (wayne).
# - sops age key in the standard location (used by sync-host-keys.sh).
# - For --skip-sync-keys: clan vars already in vars/per-machine/proxmox-ha-server-{1,2}/.
# - For full tests: secrets/common.yaml decryptable on both nodes (run
# `sops updatekeys secrets/common.yaml` after sync-keys).
set -euo pipefail
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
REPO_ROOT="$(cd "$SCRIPT_DIR/../.." && pwd)"
# shellcheck source=../env.sh
source "${REPO_ROOT}/scripts/env.sh"
# ── Defaults ──────────────────────────────────────────────────────────────────
NODE="${PROXMOX_HOST:-$PVE1_HOST}"
VMID1=200
VMID2=201
STORAGE="${PROXMOX_STORAGE:-local-zfs}"
STORAGE_BRIDGE="vmbr1"
DRBD_DISK_GB=32
MEMORY_MB=4096
CORES=4
SKIP_ENSURE_BRIDGE=false
SKIP_SYNC_KEYS=false
SKIP_CREATE_VMS=false
SKIP_ADD_HARDWARE=false
SKIP_BOOT_WAIT=false
SKIP_REFRESH_SOPS_KEYS=false
SKIP_CLUSTER_INIT=false
SKIP_TESTS=false
FORCE_REBUILD=false
DESTROY=false
DRY_RUN=false
# ── Variables from repo ───────────────────────────────────────────────────────
NODE1_HOST="ha-server-1"
NODE2_HOST="ha-server-2"
NODE1_IP="192.168.2.228"
NODE2_IP="192.168.2.227"
STORAGE_IP1="192.168.10.228"
STORAGE_IP2="192.168.10.227"
STORAGE_CIDR="192.168.10.224/29"
SSH_USER="${PROXMOX_SSH_USER:-wayne}"
# ── Argument parsing ──────────────────────────────────────────────────────────
usage() {
sed -n '/^# Usage:/,/^[^#]/{ /^#/{ s/^# \?//; p } }' "$0"
exit "${1:-0}"
}
while [[ $# -gt 0 ]]; do
case "$1" in
--node) NODE="$2"; shift 2 ;;
--vmid1) VMID1="$2"; shift 2 ;;
--vmid2) VMID2="$2"; shift 2 ;;
--storage) STORAGE="$2"; shift 2 ;;
--storage-bridge) STORAGE_BRIDGE="$2"; shift 2 ;;
--drbd-disk-gb) DRBD_DISK_GB="$2"; shift 2 ;;
--memory) MEMORY_MB="$2"; shift 2 ;;
--cores) CORES="$2"; shift 2 ;;
--skip-ensure-bridge) SKIP_ENSURE_BRIDGE=true; shift ;;
--skip-sync-keys) SKIP_SYNC_KEYS=true; shift ;;
--skip-create-vms) SKIP_CREATE_VMS=true; shift ;;
--skip-add-hardware) SKIP_ADD_HARDWARE=true; shift ;;
--skip-boot-wait) SKIP_BOOT_WAIT=true; shift ;;
--skip-refresh-sops-keys) SKIP_REFRESH_SOPS_KEYS=true; shift ;;
--skip-cluster-init) SKIP_CLUSTER_INIT=true; shift ;;
--skip-tests) SKIP_TESTS=true; shift ;;
--force-rebuild) FORCE_REBUILD=true; shift ;;
--destroy) DESTROY=true; shift ;;
--dry-run) DRY_RUN=true; shift ;;
-h|--help) usage 0 ;;
*) echo "Unknown option: $1" >&2; usage 1 ;;
esac
done
# ── Helpers ───────────────────────────────────────────────────────────────────
log() { echo "==> $*"; }
logn() { echo " $*"; }
err() { echo "ERROR: $*" >&2; exit 1; }
run() {
if $DRY_RUN; then
echo "[dry-run] $*"
else
"$@"
fi
}
pve() {
# Run a command on the Proxmox node via SSH.
if $DRY_RUN; then
echo "[dry-run] ssh ${SSH_USER}@${NODE} sudo $*"
else
ssh -i ~/.ssh/id_ed25519 "${SSH_USER}@${NODE}" "sudo $*"
fi
}
pve_check() {
# Run a read-only probe on the Proxmox node — always executes even in dry-run.
ssh -i ~/.ssh/id_ed25519 "${SSH_USER}@${NODE}" "sudo $*"
}
HA_USER="nixos"
n1() {
# Run a command on ha-server-1 via SSH as nixos with sudo.
ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no -o ConnectTimeout=5 "${HA_USER}@${NODE1_IP}" sudo "$@" 2>/dev/null
}
n2() {
# Run a command on ha-server-2 via SSH as nixos with sudo.
ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no -o ConnectTimeout=5 "${HA_USER}@${NODE2_IP}" sudo "$@" 2>/dev/null
}
wait_for_ssh() {
local ip="$1" label="$2"
if $DRY_RUN; then
logn "[dry-run] Skipping SSH wait for ${label} (${ip})"
return 0
fi
local deadline=$(( $(date +%s) + 300 ))
log "Waiting for SSH on ${label} (${ip}) — up to 5 min..."
while [[ $(date +%s) -lt $deadline ]]; do
if ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no -o ConnectTimeout=3 \
-o BatchMode=yes "${HA_USER}@${ip}" true 2>/dev/null; then
logn "${label} is up."
return 0
fi
sleep 5
done
err "Timed out waiting for SSH on ${label} (${ip})"
}
# ── Destroy mode ─────────────────────────────────────────────────────────────
if $DESTROY; then
log "Destroying HA cluster VMs (${VMID1}=${NODE1_HOST}, ${VMID2}=${NODE2_HOST}) on ${NODE}"
for vmid in "$VMID1" "$VMID2"; do
STATUS=$(pve "qm status ${vmid} 2>/dev/null" 2>/dev/null || true)
if echo "$STATUS" | grep -q "running"; then
log "Stopping VMID ${vmid}..."
pve "qm stop ${vmid} --skiplock 1"
sleep 5
fi
if $DRY_RUN || pve "qm config ${vmid} >/dev/null 2>&1"; then
log "Deleting VMID ${vmid}..."
run pve "qm destroy ${vmid} --purge 1"
else
logn "VMID ${vmid} not found — already gone."
fi
done
log "Done — cluster VMs destroyed."
exit 0
fi
# ── Phase 1: Storage bridge ───────────────────────────────────────────────────
if ! $SKIP_ENSURE_BRIDGE; then
log "Phase 1: Ensuring storage bridge ${STORAGE_BRIDGE} on ${NODE}"
if pve_check "test -d /sys/class/net/${STORAGE_BRIDGE}" &>/dev/null; then
logn "${STORAGE_BRIDGE} already exists — skipping."
else
logn "Creating isolated internal bridge ${STORAGE_BRIDGE} (no upstream port, ${STORAGE_CIDR})"
BRIDGE_CONF="auto ${STORAGE_BRIDGE}
iface ${STORAGE_BRIDGE} inet manual
bridge-ports none
bridge-stp off
bridge-fd 0"
if $DRY_RUN; then
echo "[dry-run] Would write /etc/network/interfaces.d/${STORAGE_BRIDGE}.conf and ifup it"
else
ssh -i ~/.ssh/id_ed25519 "${SSH_USER}@${NODE}" \
"echo '${BRIDGE_CONF}' | sudo tee /etc/network/interfaces.d/${STORAGE_BRIDGE}.conf > /dev/null && sudo ifup ${STORAGE_BRIDGE}"
logn "${STORAGE_BRIDGE} created and brought up."
fi
fi
fi
# ── Phase 2: Sync host keys ───────────────────────────────────────────────────
if ! $SKIP_SYNC_KEYS; then
log "Phase 2: Syncing SSH host keys for both HA targets"
for target in proxmox-ha-server-1 proxmox-ha-server-2; do
CLAN_DIR="${REPO_ROOT}/vars/per-machine/${target}/openssh"
if [[ -d "$CLAN_DIR" ]]; then
logn "Clan vars for ${target} already exist — skipping."
else
logn "Generating host keys for ${target}..."
run bash "${REPO_ROOT}/scripts/secrets/sync-host-keys.sh" "$target"
fi
done
fi
# ── Phase 2.5: Prepare Proxmox node for building ─────────────────────────────
if ! $SKIP_CREATE_VMS && ! $DRY_RUN; then
CURRENT_BRANCH="$(git -C "$REPO_ROOT" rev-parse --abbrev-ref HEAD)"
# Fix /nix ownership if it exists but belongs to a different UID.
# pve1's IPA-enrolled wayne (UID 50002) can't write to a store created by
# another UID — passwordless sudo corrects it once.
# Use direct SSH (no sudo) for the writability check so we test wayne's own
# access, not root's.
local_ssh() { ssh -i ~/.ssh/id_ed25519 "${SSH_USER}@${NODE}" "$*"; }
if local_ssh "test -d /nix" &>/dev/null && ! local_ssh "test -w /nix" &>/dev/null; then
logn "/nix exists but not writable by ${SSH_USER} — fixing ownership with sudo (one-time)..."
local_ssh "sudo chown -R ${SSH_USER} /nix"
logn "Done."
fi
unset -f local_ssh
# Ensure the remote clone is on the correct branch so create-proxmox-resource.sh
# builds from the same commits we're deploying.
REMOTE_REPO="/home/${SSH_USER}/nixos"
if pve_check "test -d ${REMOTE_REPO}/.git" &>/dev/null; then
REMOTE_BRANCH=$(ssh -i ~/.ssh/id_ed25519 "${SSH_USER}@${NODE}" \
"cd ${REMOTE_REPO} && git rev-parse --abbrev-ref HEAD 2>/dev/null")
if [[ "$REMOTE_BRANCH" != "$CURRENT_BRANCH" ]]; then
logn "Remote clone is on '${REMOTE_BRANCH}', switching to '${CURRENT_BRANCH}'..."
ssh -i ~/.ssh/id_ed25519 "${SSH_USER}@${NODE}" \
"cd ${REMOTE_REPO} && git fetch origin && git checkout '${CURRENT_BRANCH}' && git pull --ff-only"
logn "Done."
fi
fi
fi
# ── Phase 3: Create VMs ───────────────────────────────────────────────────────
if ! $SKIP_CREATE_VMS; then
log "Phase 3: Building and creating VMs on ${NODE}"
CREATE="${REPO_ROOT}/scripts/proxmox/create-proxmox-resource.sh"
for spec in "${VMID1}:ha-server-1:proxmox-ha-server-1" "${VMID2}:ha-server-2:proxmox-ha-server-2"; do
IFS=: read -r vmid host_name flake_target <<< "$spec"
log "Creating ${flake_target} (VMID ${vmid}) on ${NODE}..."
# Always --force-rebuild: create-proxmox-resource.sh only calls
# sync_remote_host_keys (which populates host-keys/ for proxmox.nix to
# bake the clan-var SSH key into the disko image) when it actually builds.
# Reusing a cached image skips that step, so destroy+recreate would reuse
# an image with a stale/random key baked in → sops fails on first boot.
run bash "$CREATE" \
--type vm \
--host "$host_name" \
--vmid "$vmid" \
--node "$NODE" \
--storage "$STORAGE" \
--memory "$MEMORY_MB" \
--cores "$CORES" \
--force-rebuild
done
fi
# ── Phase 4: Add storage NIC and DRBD disk ────────────────────────────────────
if ! $SKIP_ADD_HARDWARE; then
log "Phase 4: Attaching storage NIC (${STORAGE_BRIDGE}) and DRBD disk (${DRBD_DISK_GB}G) to each VM"
for vmid in "$VMID1" "$VMID2"; do
log " VMID ${vmid}: stopping to add hardware..."
pve "qm stop ${vmid} --skiplock 1 2>/dev/null; sleep 3" || true
logn "Adding net1 (${STORAGE_BRIDGE})..."
pve "qm set ${vmid} --net1 virtio,bridge=${STORAGE_BRIDGE},firewall=0"
logn "Adding scsi1 (${STORAGE}:${DRBD_DISK_GB}G for DRBD)..."
pve "qm set ${vmid} --scsi1 ${STORAGE}:${DRBD_DISK_GB},format=raw"
logn "Starting VMID ${vmid}..."
pve "qm start ${vmid}"
done
fi
# ── Phase 5: Wait for SSH ─────────────────────────────────────────────────────
if ! $SKIP_BOOT_WAIT; then
log "Phase 5: Waiting for both nodes to come up"
wait_for_ssh "$NODE1_IP" "$NODE1_HOST"
wait_for_ssh "$NODE2_IP" "$NODE2_HOST"
logn "Both nodes are SSHable."
# Give systemd a few seconds to settle after activation
sleep 10
fi
# ── Phase 5.5: Refresh sops host-key registrations ───────────────────────────
#
# Disko builds raw disk images; each new VM boots with a freshly-generated SSH
# host key rather than the one pre-seeded in clan vars. This phase scans the
# actual running VMs, and if their ed25519 host keys differ from what clan vars
# record: updates the clan var pub-key files, rewrites the .sops.yaml age-key
# anchors, and re-encrypts all affected sops files so the nodes can decrypt
# secrets on the next nixos-rebuild. Safe no-op when keys haven't changed.
if ! $SKIP_REFRESH_SOPS_KEYS; then
if $DRY_RUN; then
logn "[dry-run] Would scan VM host keys and refresh .sops.yaml / secrets if needed"
else
log "Phase 5.5: Refreshing sops host-key registrations (disko key drift fix)"
SOPS_UPDATED=false
for spec in \
"${NODE1_IP}:proxmox-ha-server-1:${NODE1_HOST}" \
"${NODE2_IP}:proxmox-ha-server-2:${NODE2_HOST}"; do
IFS=: read -r node_ip flake_target host_name <<< "$spec"
CLAN_PUB="${REPO_ROOT}/vars/per-machine/${flake_target}/openssh/ssh_host_ed25519_key.pub/value"
logn "Scanning ed25519 host key from ${host_name} (${node_ip})..."
RAW=$(ssh-keyscan -t ed25519 "${node_ip}" 2>/dev/null | grep -v "^#") || true
if [[ -z "$RAW" ]]; then
logn "WARNING: no ed25519 key returned by ssh-keyscan for ${node_ip} — skipping"
continue
fi
# ssh-keyscan returns: <ip> ssh-ed25519 <b64key>
SCANNED_TYPE=$(awk '{print $2}' <<< "$RAW")
SCANNED_KEY=$(awk '{print $3}' <<< "$RAW")
SCANNED_PUBKEY="${SCANNED_TYPE} ${SCANNED_KEY} ${host_name}"
CURRENT=$(tr -d '\n' < "$CLAN_PUB" 2>/dev/null || true)
if [[ "$SCANNED_PUBKEY" == "$CURRENT" ]]; then
logn "${host_name}: clan var matches running key — no update needed"
continue
fi
logn "${host_name}: key drift detected — updating clan var"
logn " old: ${CURRENT}"
logn " new: ${SCANNED_PUBKEY}"
echo "$SCANNED_PUBKEY" > "$CLAN_PUB"
SOPS_UPDATED=true
# Rewrite the .sops.yaml anchor for this host with the new age key.
ANCHOR="${flake_target}" # e.g. proxmox-ha-server-1
NEW_AGE=$(echo "$SCANNED_PUBKEY" | \
nix run --quiet --no-warn-dirty nixpkgs#ssh-to-age 2>/dev/null)
if [[ -z "$NEW_AGE" ]]; then
err "ssh-to-age produced no output for ${host_name} — check nixpkgs#ssh-to-age"
fi
logn " new age key: ${NEW_AGE}"
sed -i "/&${ANCHOR} /s| age[a-z0-9]*$| ${NEW_AGE}|" "${REPO_ROOT}/.sops.yaml"
done
if $SOPS_UPDATED; then
logn "Running sops updatekeys on affected secrets..."
SOPS="nix run --quiet --no-warn-dirty nixpkgs#sops --"
(cd "${REPO_ROOT}" && \
$SOPS updatekeys -y secrets/common.yaml && \
$SOPS updatekeys -y secrets/ha-server-1.yaml && \
$SOPS updatekeys -y secrets/ha-server-2.yaml && \
$SOPS updatekeys -y secrets/ha-server-1.keytab && \
$SOPS updatekeys -y secrets/ha-server-2.keytab)
# Note: ha-corosync-authkey is re-generated and re-encrypted by cluster-init below.
logn "Committing refreshed host keys and re-encrypted secrets..."
(cd "${REPO_ROOT}" && \
git add \
vars/per-machine/proxmox-ha-server-1/openssh/ssh_host_ed25519_key.pub/value \
vars/per-machine/proxmox-ha-server-2/openssh/ssh_host_ed25519_key.pub/value \
.sops.yaml \
secrets/common.yaml \
secrets/ha-server-1.yaml \
secrets/ha-server-2.yaml \
secrets/ha-server-1.keytab \
secrets/ha-server-2.keytab && \
git commit -m "secrets(ha): refresh sops host-key registrations for new VM instances" || true)
logn "Sops keys refreshed and committed."
fi
fi
fi
# ── Phase 6: Cluster init ─────────────────────────────────────────────────────
if ! $SKIP_CLUSTER_INIT; then
log "Phase 6: Initialising HA cluster"
CLUSTER_INIT="${REPO_ROOT}/scripts/ha/cluster-init.sh"
[[ -x "$CLUSTER_INIT" ]] || chmod +x "$CLUSTER_INIT"
if $DRY_RUN; then
logn "[dry-run] Would generate temp key, authorise on ${NODE2_HOST}, scp cluster-init.sh to ${NODE1_HOST}, and run it as root via sudo"
else
# Generate a temp keypair so cluster-init.sh can SSH node1→node2 as ${HA_USER}.
# Root on node1 has no keys; a temp key bridging node1→node2 nixos solves this.
TEMP_KEY="${REPO_ROOT}/.tmp-cluster-init-key"
TEMP_KEY_PUB="${TEMP_KEY}.pub"
rm -f "$TEMP_KEY" "$TEMP_KEY_PUB"
ssh-keygen -t ed25519 -f "$TEMP_KEY" -N "" -C "cluster-init-temp-$(date +%s)" -q
TEMP_PUBKEY=$(cat "$TEMP_KEY_PUB")
logn "Authorising temp key on ${NODE2_HOST} for ${HA_USER}..."
ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no "${HA_USER}@${NODE2_IP}" \
"mkdir -p ~/.ssh && chmod 700 ~/.ssh && echo '${TEMP_PUBKEY}' >> ~/.ssh/authorized_keys"
logn "Placing temp key on ${NODE1_HOST} for root..."
scp -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no \
"$TEMP_KEY" "${HA_USER}@${NODE1_IP}:/tmp/cluster-init-key"
ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no "${HA_USER}@${NODE1_IP}" \
"sudo mkdir -p /root/.ssh && sudo cp /tmp/cluster-init-key /root/.ssh/cluster-init-key && \
sudo chmod 600 /root/.ssh/cluster-init-key && rm -f /tmp/cluster-init-key"
logn "Uploading cluster-init.sh to ${NODE1_HOST}..."
scp -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no \
"$CLUSTER_INIT" "${HA_USER}@${NODE1_IP}:/tmp/cluster-init.sh"
logn "Running cluster-init.sh on ${NODE1_HOST}..."
ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no "${HA_USER}@${NODE1_IP}" \
"sudo env NODE1=${NODE1_HOST} NODE2=${NODE2_HOST} \
NODE1_IP=${NODE1_IP} NODE2_IP=${NODE2_IP} \
VIP=192.168.20.229 XFS_MOUNT=/srv/ha-data \
ISCSI_IQN=iqn.2026-01.home.sweet:ha-storage \
VMID_NODE1=${VMID1} VMID_NODE2=${VMID2} \
HA_USER=${HA_USER} HA_KEY=/root/.ssh/cluster-init-key \
bash /tmp/cluster-init.sh"
logn "Cleaning up temp key from both nodes..."
ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no "${HA_USER}@${NODE2_IP}" \
"sed -i '/cluster-init-temp/d' ~/.ssh/authorized_keys" 2>/dev/null || true
ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no "${HA_USER}@${NODE1_IP}" \
"sudo rm -f /root/.ssh/cluster-init-key" 2>/dev/null || true
rm -f "$TEMP_KEY" "$TEMP_KEY_PUB"
# Encrypt the corosync authkey generated by cluster-init and commit it.
log " Encrypting corosync authkey into secrets/ha-corosync-authkey..."
AUTHKEY_TMP="${REPO_ROOT}/secrets/ha-corosync-authkey.tmp"
ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no "${HA_USER}@${NODE1_IP}" \
"sudo cat /etc/corosync/authkey" > "$AUTHKEY_TMP"
if [[ ! -s "$AUTHKEY_TMP" ]]; then
err "corosync authkey on node1 is empty — cluster-init may have failed."
fi
mv "$AUTHKEY_TMP" "${REPO_ROOT}/secrets/ha-corosync-authkey"
(cd "${REPO_ROOT}" && nix run nixpkgs#sops -- -e --input-type binary -i secrets/ha-corosync-authkey)
logn "Authkey encrypted. Committing..."
(cd "${REPO_ROOT}" && git add secrets/ha-corosync-authkey && \
git commit -m "secrets(ha): encrypt corosync authkey generated by cluster-init")
logn "Committed."
fi
fi
# ── Phase 7: Acceptance tests ─────────────────────────────────────────────────
if ! $SKIP_TESTS; then
log "Phase 7: Running acceptance tests (T1T7)"
if $DRY_RUN; then
logn "[dry-run] Would run acceptance-tests.sh against ${NODE1_HOST}/${NODE2_HOST}"
else
NODE1="$NODE1_HOST" NODE2="$NODE2_HOST" \
NODE1_IP="$NODE1_IP" NODE2_IP="$NODE2_IP" \
VIP="192.168.20.229" \
bash "${REPO_ROOT}/scripts/ha/acceptance-tests.sh"
fi
fi
log "Deploy complete."
-256
View File
@@ -1,256 +0,0 @@
#!/usr/bin/env bash
# failover.sh — graceful HA cluster failover
#
# Detects which node is active and moves all resources to the other node by
# putting the active node into Pacemaker standby. Waits for the XFS mount to
# appear on the target before returning.
#
# Usage:
# scripts/ha/failover.sh [--to node1|node2] [--force] [--timeout <s>] [--dry-run]
#
# --to node1|node2 target node (default: the node that is NOT currently active)
# --force skip the interactive confirmation prompt
# --timeout <s> seconds to wait for resources to move (default: 120)
# --dry-run show what would be done without changing anything
set -euo pipefail
# ── Configuration ─────────────────────────────────────────────────────────
NODE1="${NODE1:-ha-server-1}"
NODE2="${NODE2:-ha-server-2}"
NODE1_IP="${NODE1_IP:-192.168.2.228}"
NODE2_IP="${NODE2_IP:-192.168.2.227}"
VIP="${VIP:-192.168.20.229}"
XFS_MOUNT="${XFS_MOUNT:-/srv/ha-data}"
HA_USER="${HA_USER:-nixos}"
# ──────────────────────────────────────────────────────────────────────────
n1() { ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no -o ConnectTimeout=5 "${HA_USER}@${NODE1_IP}" sudo "$@" 2>/dev/null; }
n2() { ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no -o ConnectTimeout=5 "${HA_USER}@${NODE2_IP}" sudo "$@" 2>/dev/null; }
# ── Argument parsing ───────────────────────────────────────────────────────
TARGET_NODE=""
FORCE=false
DRY_RUN=false
TIMEOUT=120
while [[ $# -gt 0 ]]; do
case "$1" in
--to)
shift
case "${1:-}" in
node1|ha-server-1) TARGET_NODE="$NODE1" ;;
node2|ha-server-2) TARGET_NODE="$NODE2" ;;
*) echo "ERROR: --to must be node1, node2, ha-server-1, or ha-server-2"; exit 1 ;;
esac
;;
--force) FORCE=true ;;
--dry-run) DRY_RUN=true ;;
--timeout) shift; TIMEOUT="${1:?--timeout requires a value}" ;;
*) echo "Unknown argument: $1"; echo "Usage: $0 [--to node1|node2] [--force] [--timeout <s>] [--dry-run]"; exit 1 ;;
esac
shift
done
DRY_PREFIX=""
$DRY_RUN && DRY_PREFIX="[dry-run] "
echo "════════════════════════════════════════════════════"
echo " HA Cluster Failover — $(date '+%Y-%m-%d %H:%M:%S')"
$DRY_RUN && echo " MODE: dry-run — no changes will be made"
echo "════════════════════════════════════════════════════"
# ── Detect active node ─────────────────────────────────────────────────────
echo ""
echo "Detecting active node..."
CRM_OUT=""
if ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no -o ConnectTimeout=5 "${HA_USER}@${NODE1_IP}" true 2>/dev/null; then
CRM_OUT=$(n1 "crm_mon -1" 2>/dev/null || true)
fi
if [[ -z "$CRM_OUT" ]]; then
if ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no -o ConnectTimeout=5 "${HA_USER}@${NODE2_IP}" true 2>/dev/null; then
CRM_OUT=$(n2 "crm_mon -1" 2>/dev/null || true)
fi
fi
# crm_mon 2.x formats Promoted lines as " * Promoted: [ node ]" — the * bullet
# means ^\s*(Promoted|Masters): never matches; filter Unpromoted first instead.
ACTIVE_NODE=$(echo "$CRM_OUT" | grep -v 'Unpromoted\|Unmanaged' | \
grep -E '(Promoted|Masters):' | \
grep -oE '\b(ha-server-[0-9]+)\b' | head -1 || true)
if [[ -z "$ACTIVE_NODE" ]]; then
echo ""
echo "ERROR: could not determine active node from crm_mon."
echo " Is Pacemaker still settling? Try running scripts/ha/health.sh first."
echo " If Pacemaker is down on both nodes, manual recovery is required."
exit 1
fi
if [[ "$ACTIVE_NODE" == "$NODE1" ]]; then
ACTIVE_IP="$NODE1_IP"
STANDBY_NODE="$NODE2"
STANDBY_IP="$NODE2_IP"
na() { n1 "$@"; }
ns() { n2 "$@"; }
else
ACTIVE_IP="$NODE2_IP"
STANDBY_NODE="$NODE1"
STANDBY_IP="$NODE1_IP"
na() { n2 "$@"; }
ns() { n1 "$@"; }
fi
echo " Active: $ACTIVE_NODE ($ACTIVE_IP)"
echo " Standby: $STANDBY_NODE ($STANDBY_IP)"
# ── Validate target ────────────────────────────────────────────────────────
if [[ -n "$TARGET_NODE" ]]; then
if [[ "$TARGET_NODE" == "$ACTIVE_NODE" ]]; then
echo ""
echo "ERROR: $TARGET_NODE is already the active node — nothing to do."
exit 1
fi
echo " Target: $TARGET_NODE (as requested)"
else
echo " Target: $STANDBY_NODE (auto — the other node)"
fi
# ── Pre-checks ─────────────────────────────────────────────────────────────
echo ""
echo "Pre-checks..."
DRBD_DSTATE=$(na "drbdadm dstate ha-data 2>/dev/null" 2>/dev/null || echo "unknown")
if ! echo "$DRBD_DSTATE" | grep -q "UpToDate/UpToDate"; then
echo ""
echo " WARNING: DRBD dstate is '$DRBD_DSTATE' (not UpToDate/UpToDate)."
echo " Failing over with a partially-synced disk risks split-brain."
if ! $FORCE; then
echo " Use --force to proceed anyway (not recommended)."
exit 1
fi
echo " --force specified — proceeding despite non-ideal DRBD state."
else
echo " DRBD dstate: $DRBD_DSTATE — OK"
fi
QUORUM_OK=$(na "corosync-quorumtool -s 2>/dev/null | grep -c 'Quorate:.*Yes'" 2>/dev/null || echo "0")
if [[ "$QUORUM_OK" -lt 1 ]]; then
echo " ERROR: cluster does not have quorum — failover would be unsafe."
exit 1
fi
echo " Quorum: OK"
# ── Confirm ────────────────────────────────────────────────────────────────
if ! $FORCE && ! $DRY_RUN; then
echo ""
echo " This will move all resources from $ACTIVE_NODE$STANDBY_NODE."
echo " VIP and services will be unreachable for ~1030 seconds."
printf " Proceed? [y/N] "
read -r ANSWER
[[ "${ANSWER,,}" == "y" || "${ANSWER,,}" == "yes" ]] || { echo "Aborted."; exit 0; }
fi
# ── Capture active node's crm_node name ───────────────────────────────────
# crm_node -n returns the node name as registered in Pacemaker (may differ
# from hostname if Pacemaker was configured with explicit node names).
ACTIVE_CRMD_NAME=$(na "crm_node -n 2>/dev/null" 2>/dev/null || echo "$ACTIVE_NODE")
# ── Perform failover ───────────────────────────────────────────────────────
echo ""
echo "${DRY_PREFIX}Putting $ACTIVE_NODE into standby (resources will migrate to $STANDBY_NODE)..."
if ! $DRY_RUN; then
na "crm_standby -N '${ACTIVE_CRMD_NAME}' -v on" 2>/dev/null || true
fi
# ── Wait for resources to move ─────────────────────────────────────────────
echo "${DRY_PREFIX}Waiting up to ${TIMEOUT}s for XFS to mount on $STANDBY_NODE..."
MOVED=false
SPIN_CHARS=('|' '/' '-' '\')
SPIN_I=0
if $DRY_RUN; then
echo " [dry-run] would wait for mountpoint $XFS_MOUNT on $STANDBY_NODE"
MOVED=true
else
for i in $(seq 1 "$TIMEOUT"); do
if ns "mountpoint -q '${XFS_MOUNT}' 2>/dev/null" 2>/dev/null; then
printf "\r%-80s\r" ""
echo " Resources moved in ${i}s"
MOVED=true
break
fi
SPIN_I=$(( SPIN_I + 1 ))
SC="${SPIN_CHARS[$((SPIN_I % 4))]}"
printf "\r [%s] waiting... (%ds) " "$SC" "$i"
sleep 1
done
fi
if ! $MOVED; then
echo ""
echo "ERROR: XFS did not mount on $STANDBY_NODE within ${TIMEOUT}s."
echo ""
echo " Current resource state:"
na "crm_mon -1 2>/dev/null" 2>/dev/null | grep -E 'Started|Stopped|Promoted|Unpromoted|FAILED' | sed 's/^/ /' || true
echo ""
echo " Clearing standby to restore $ACTIVE_NODE (undo the failover attempt)..."
na "crm_standby -N '${ACTIVE_CRMD_NAME}' -v off" 2>/dev/null || true
na "crm_resource --cleanup" 2>/dev/null || true
exit 1
fi
# ── Clear failure history ──────────────────────────────────────────────────
echo "${DRY_PREFIX}Clearing Pacemaker failure history..."
if ! $DRY_RUN; then
ns "crm_resource --cleanup 2>/dev/null" 2>/dev/null || true
fi
# ── Re-enable original active node as standby ─────────────────────────────
echo "${DRY_PREFIX}Re-enabling $ACTIVE_NODE (now standby — will not claim resources)..."
if ! $DRY_RUN; then
na "crm_standby -N '${ACTIVE_CRMD_NAME}' -v off" 2>/dev/null || true
fi
# ── Wait briefly for DRBD resync to begin ─────────────────────────────────
if ! $DRY_RUN; then
sleep 5
fi
# ── Final state ────────────────────────────────────────────────────────────
echo ""
echo "Failover complete. Final state:"
echo ""
CRM_OUT_AFTER=""
if ! $DRY_RUN; then
CRM_OUT_AFTER=$(ns "crm_mon -1" 2>/dev/null || na "crm_mon -1" 2>/dev/null || true)
else
CRM_OUT_AFTER="$CRM_OUT"
fi
NEW_ACTIVE=$(echo "$CRM_OUT_AFTER" | grep -v 'Unpromoted\|Unmanaged' | \
grep -E '(Promoted|Masters):' | \
grep -oE '\b(ha-server-[0-9]+)\b' | head -1 || true)
if [[ -n "$NEW_ACTIVE" ]]; then
if [[ "$NEW_ACTIVE" == "$ACTIVE_NODE" ]]; then
echo " WARNING: $ACTIVE_NODE is still showing as active in crm_mon."
echo " Pacemaker may still be settling — check again in a few seconds."
else
echo " Active: $NEW_ACTIVE"
echo " Standby: $ACTIVE_NODE"
fi
fi
echo ""
RESOURCES_AFTER=$(echo "$CRM_OUT_AFTER" | awk '/Full List of Resources/,0' | tail -n +2 || true)
[[ -z "$RESOURCES_AFTER" ]] && RESOURCES_AFTER=$(echo "$CRM_OUT_AFTER" | \
grep -E 'Started|Stopped|Promoted|Unpromoted|FAILED|Master|Slave' || true)
[[ -n "$RESOURCES_AFTER" ]] && echo "$RESOURCES_AFTER" | sed 's/^/ /'
echo ""
echo " (DRBD resync of $ACTIVE_NODE may take a moment; monitor with:"
echo " ssh nixos@${ACTIVE_IP} 'sudo watch -n3 cat /proc/drbd')"
echo ""
echo "════════════════════════════════════════════════════"
-179
View File
@@ -1,179 +0,0 @@
#!/usr/bin/env python3
"""
fence_pve_ssh - Proxmox VE SSH fence agent for Pacemaker.
Uses SSH to reach the Proxmox host and run 'qm stop/start <vmid>'.
Deploy to /etc/pacemaker/fence_pve_ssh on both HA nodes (chmod +x).
Configuration (as pacemaker stonith resource attributes):
pve_host Proxmox host to SSH to (default: pve1.sweet.home)
pve_user SSH user (default: wayne)
key_file SSH private key path (default: /etc/fence-pve-ssh-key)
vmid_node1 VMID for ha-server-1
vmid_node2 VMID for ha-server-2
plug Node name to act on (set by pacemaker: ha-server-1 or ha-server-2)
action Action: off|on|reboot|status|list|metadata
"""
import argparse
import subprocess
import sys
import os
METADATA = """<?xml version="1.0" ?>
<resource-agent name="fence_pve_ssh" shortdesc="Proxmox VE SSH fence agent (test lab)">
<longdesc>Fences a VM on a Proxmox VE host by SSHing to the PVE host and
running qm stop/start. For test use only.</longdesc>
<vendor-url>https://proxmox.com</vendor-url>
<parameters>
<parameter name="action" required="1" unique="0">
<getopt mixed="-a, --action=[action]"/>
<content type="string" default="reboot"/>
<shortdesc lang="en">Fencing action: off|on|reboot|status|list</shortdesc>
</parameter>
<parameter name="plug" required="0" unique="0">
<getopt mixed="-n, --plug=[nodename]"/>
<content type="string"/>
<shortdesc lang="en">Cluster node name to fence</shortdesc>
</parameter>
<parameter name="pve_host" required="0" unique="0">
<getopt mixed="--pve-host=[host]"/>
<content type="string" default="pve1.sweet.home"/>
<shortdesc lang="en">Proxmox VE host to SSH to</shortdesc>
</parameter>
<parameter name="pve_user" required="0" unique="0">
<getopt mixed="--pve-user=[user]"/>
<content type="string" default="wayne"/>
<shortdesc lang="en">SSH user on the Proxmox host</shortdesc>
</parameter>
<parameter name="key_file" required="0" unique="0">
<getopt mixed="--key-file=[path]"/>
<content type="string" default="/etc/fence-pve-ssh-key"/>
<shortdesc lang="en">SSH private key file path</shortdesc>
</parameter>
<parameter name="vmid_node1" required="1" unique="0">
<getopt mixed="--vmid-node1=[vmid]"/>
<content type="string"/>
<shortdesc lang="en">VMID for ha-test-node1</shortdesc>
</parameter>
<parameter name="vmid_node2" required="1" unique="0">
<getopt mixed="--vmid-node2=[vmid]"/>
<content type="string"/>
<shortdesc lang="en">VMID for ha-test-node2</shortdesc>
</parameter>
</parameters>
<actions>
<action name="off" timeout="60s"/>
<action name="on" timeout="60s"/>
<action name="reboot" timeout="60s"/>
<action name="status" timeout="30s"/>
<action name="list" timeout="10s"/>
<action name="metadata" timeout="5s"/>
</actions>
</resource-agent>
"""
def parse_args():
p = argparse.ArgumentParser(add_help=False)
p.add_argument("-a", "--action", default="reboot")
p.add_argument("-n", "--plug")
p.add_argument("--pve-host", default="pve1.sweet.home")
p.add_argument("--pve-user", default="wayne")
p.add_argument("--key-file", default="/etc/fence-pve-ssh-key")
p.add_argument("--vmid-node1")
p.add_argument("--vmid-node2")
# Allow remaining unknown args (pacemaker may pass extra ones)
return p.parse_known_args()[0]
def ssh(pve_host, pve_user, key_file, cmd):
result = subprocess.run(
[
"ssh",
"-i", key_file,
"-o", "StrictHostKeyChecking=no",
"-o", "BatchMode=yes",
"-o", "ConnectTimeout=10",
f"{pve_user}@{pve_host}",
cmd,
],
capture_output=True,
text=True,
timeout=30,
)
return result
def get_vmid(args):
node = args.plug
if not node:
print("ERROR: --plug not specified", file=sys.stderr)
sys.exit(1)
mapping = {
"ha-server-1": args.vmid_node1,
"ha-server-2": args.vmid_node2,
}
vmid = mapping.get(node)
if not vmid:
print(f"ERROR: unknown node '{node}'", file=sys.stderr)
sys.exit(1)
return vmid
def main():
args = parse_args()
action = args.action.lower()
if action == "metadata":
print(METADATA)
sys.exit(0)
if action == "list":
if args.vmid_node1:
print("ha-server-1")
if args.vmid_node2:
print("ha-server-2")
sys.exit(0)
vmid = get_vmid(args)
if not os.path.exists(args.key_file):
print(f"ERROR: SSH key not found at {args.key_file}", file=sys.stderr)
sys.exit(1)
if action in ("off", "reboot"):
print(f"Stopping VM {vmid} ({args.plug}) on {args.pve_host}...")
r = ssh(args.pve_host, args.pve_user, args.key_file,
f"sudo /usr/sbin/qm stop {vmid}")
if r.returncode != 0:
print(f"ERROR stopping VM: {r.stderr}", file=sys.stderr)
sys.exit(1)
print(f"VM {vmid} stopped")
if action in ("on", "reboot"):
print(f"Starting VM {vmid} ({args.plug}) on {args.pve_host}...")
r = ssh(args.pve_host, args.pve_user, args.key_file,
f"sudo /usr/sbin/qm start {vmid}")
if r.returncode != 0:
print(f"ERROR starting VM: {r.stderr}", file=sys.stderr)
sys.exit(1)
print(f"VM {vmid} started")
if action == "status":
r = ssh(args.pve_host, args.pve_user, args.key_file,
f"sudo /usr/sbin/qm status {vmid}")
if r.returncode != 0:
print(f"ERROR querying VM status: {r.stderr}", file=sys.stderr)
sys.exit(1)
# qm status returns "status: running" or "status: stopped"
status_line = r.stdout.strip()
print(status_line)
if "stopped" in status_line:
sys.exit(2) # pacemaker interprets exit 2 as "off"
sys.exit(0) # running = exit 0
if __name__ == "__main__":
main()
-206
View File
@@ -1,206 +0,0 @@
#!/usr/bin/env bash
# health.sh — HA cluster health snapshot (read-only, non-destructive)
#
# Prints a compact status panel across both nodes: SSH reachability, quorum,
# DRBD state, Pacemaker resources, and service ports via the VIP.
# Run from any host with SSH access to the HA nodes.
set -euo pipefail
# ── Configuration ─────────────────────────────────────────────────────────
NODE1="${NODE1:-ha-server-1}"
NODE2="${NODE2:-ha-server-2}"
NODE1_IP="${NODE1_IP:-192.168.2.228}" # vars.haServer1Ip
NODE2_IP="${NODE2_IP:-192.168.2.227}" # vars.haServer2Ip
VIP="${VIP:-192.168.20.229}" # vars.haServerVip (storage-client, VLAN 20 — internal only)
VIP_LAN="${VIP_LAN:-192.168.2.229}" # vars.haServerLanVip (LAN, VLAN 2 — reachable from workstation)
XFS_MOUNT="${XFS_MOUNT:-/srv/ha-data}" # vars.haStorageRoot
HA_USER="${HA_USER:-nixos}"
# ──────────────────────────────────────────────────────────────────────────
REACHABLE_1=false
REACHABLE_2=false
n1() { ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no -o ConnectTimeout=5 "${HA_USER}@${NODE1_IP}" sudo "$@" 2>/dev/null; }
n2() { ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no -o ConnectTimeout=5 "${HA_USER}@${NODE2_IP}" sudo "$@" 2>/dev/null; }
probe_node() {
local ip=$1
ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no -o ConnectTimeout=5 "${HA_USER}@${ip}" true 2>/dev/null && echo "ONLINE" || echo "OFFLINE"
}
section() { echo ""; echo "── $* ──"; }
echo "════════════════════════════════════════════════════"
echo " HA Cluster Health — $(date '+%Y-%m-%d %H:%M:%S')"
echo "════════════════════════════════════════════════════"
# ── Node reachability ──────────────────────────────────────────────────────
section "Nodes"
N1_STATUS=$(probe_node "$NODE1_IP")
N2_STATUS=$(probe_node "$NODE2_IP")
[[ "$N1_STATUS" == "ONLINE" ]] && REACHABLE_1=true
[[ "$N2_STATUS" == "ONLINE" ]] && REACHABLE_2=true
if ! $REACHABLE_1 && ! $REACHABLE_2; then
echo " ERROR: both nodes unreachable — cannot continue."
exit 1
fi
# ── Detect active node ─────────────────────────────────────────────────────
# crm_mon 2.x formats the Promoted line as " * Promoted: [ node ]" — the
# bullet * means ^\s*(Promoted|Masters): never matches. Filter out Unpromoted
# first, then match anywhere on the line.
ACTIVE_NODE=""
CRM_OUT=""
if $REACHABLE_1; then
CRM_OUT=$(n1 "crm_mon -1" 2>/dev/null || true)
elif $REACHABLE_2; then
CRM_OUT=$(n2 "crm_mon -1" 2>/dev/null || true)
fi
ACTIVE_NODE=$(echo "${CRM_OUT:-}" | grep -v 'Unpromoted\|Unmanaged' | \
grep -E '(Promoted|Masters):' | \
grep -oE '\b(ha-server-[0-9]+)\b' | head -1 || true)
STANDBY_NODE=""
if [[ "$ACTIVE_NODE" == "$NODE1" ]]; then
STANDBY_NODE="$NODE2"
elif [[ "$ACTIVE_NODE" == "$NODE2" ]]; then
STANDBY_NODE="$NODE1"
fi
n1_tag=""; n2_tag=""
[[ "$ACTIVE_NODE" == "$NODE1" ]] && n1_tag=" [ACTIVE]" || n1_tag=" [STANDBY]"
[[ "$ACTIVE_NODE" == "$NODE2" ]] && n2_tag=" [ACTIVE]" || n2_tag=" [STANDBY]"
[[ -z "$ACTIVE_NODE" ]] && { n1_tag=""; n2_tag=""; }
printf " %-14s [%s]%s\n" "$NODE1" "$N1_STATUS" "$n1_tag"
printf " %-14s [%s]%s\n" "$NODE2" "$N2_STATUS" "$n2_tag"
if [[ -z "$ACTIVE_NODE" ]]; then
echo ""
echo " WARNING: could not determine active node from crm_mon."
echo " Pacemaker may still be settling, or both nodes may be in standby."
fi
# ── Quorum ─────────────────────────────────────────────────────────────────
section "Quorum"
if $REACHABLE_1; then
QUORUM=$(n1 "corosync-quorumtool -s 2>/dev/null" 2>/dev/null || echo "")
elif $REACHABLE_2; then
QUORUM=$(n2 "corosync-quorumtool -s 2>/dev/null" 2>/dev/null || echo "")
fi
if [[ -z "${QUORUM:-}" ]]; then
echo " corosync-quorumtool: unavailable"
else
QUORATE=$(echo "$QUORUM" | grep "Quorate:" | awk '{print $2}' || echo "?")
VOTES=$(echo "$QUORUM" | grep "Total votes:" | awk '{print $3}' || echo "?")
NEEDED=$(echo "$QUORUM" | grep "Quorum votes:" | awk '{print $3}' || echo "?")
echo " Quorate: $QUORATE Votes: $VOTES / Expected: $NEEDED"
fi
# ── DRBD ───────────────────────────────────────────────────────────────────
section "DRBD (ha-data)"
drbd_info_from() {
local node=$1 run=$2
local role dstate cs pct
role=$($run "drbdadm role ha-data 2>/dev/null" 2>/dev/null || echo "unknown")
dstate=$($run "drbdadm dstate ha-data 2>/dev/null" 2>/dev/null || echo "unknown")
cs=$($run "grep -oE 'cs:[A-Za-z]+' /proc/drbd 2>/dev/null | head -1 | sed 's/cs://'" 2>/dev/null || echo "unknown")
pct=$($run "grep -oE \"sync'ed:[[:space:]]+[0-9.]+\" /proc/drbd 2>/dev/null | grep -oE '[0-9.]+$' | head -1" 2>/dev/null || echo "")
printf " %-14s role: %-22s dstate: %-25s cs: %s" \
"$node" "${role:-unknown}" "${dstate:-unknown}" "${cs:-unknown}"
[[ -n "$pct" ]] && printf " syncing: %s%%" "$pct"
echo ""
}
$REACHABLE_1 && drbd_info_from "$NODE1" n1 || echo " $NODE1 [OFFLINE]"
$REACHABLE_2 && drbd_info_from "$NODE2" n2 || echo " $NODE2 [OFFLINE]"
# ── Pacemaker (full crm_mon output) ───────────────────────────────────────
section "Pacemaker"
if [[ -n "${CRM_OUT:-}" ]]; then
echo "$CRM_OUT" | sed 's/^/ /'
else
echo " crm_mon returned no output — trying again without suppression:"
if $REACHABLE_1; then
n1 "crm_mon -1" || true
elif $REACHABLE_2; then
n2 "crm_mon -1" || true
fi
fi
# ── XFS mount ─────────────────────────────────────────────────────────────
section "XFS Mount ($XFS_MOUNT)"
check_mount() {
local node=$1 run=$2
local status
if $run "mountpoint -q '$XFS_MOUNT' 2>/dev/null" 2>/dev/null; then
local usage
usage=$($run "df -h '$XFS_MOUNT' 2>/dev/null | tail -1 | awk '{print \$3\"/\"\$2\" used (\"\$5\")\";}'" 2>/dev/null || echo "")
status="mounted"
[[ -n "$usage" ]] && status="mounted $usage"
else
status="not mounted"
fi
printf " %-14s %s\n" "$node" "$status"
}
$REACHABLE_1 && check_mount "$NODE1" n1 || echo " $NODE1 [OFFLINE]"
$REACHABLE_2 && check_mount "$NODE2" n2 || echo " $NODE2 [OFFLINE]"
# ── LAN VIP (NFS) — reachable from workstation ────────────────────────────────
section "LAN VIP ($VIP_LAN) — NFS"
if ping -c1 -W2 "$VIP_LAN" >/dev/null 2>&1; then
echo " Ping OK"
else
echo " Ping UNREACHABLE"
fi
if bash -c "echo >/dev/tcp/${VIP_LAN}/2049" 2>/dev/null; then
printf " %-10s port %-5s OK\n" "NFS" "2049"
else
printf " %-10s port %-5s UNREACHABLE\n" "NFS" "2049"
fi
# ── Storage VIP (NFS + iSCSI) — VLAN 20 internal bridge, tested via active node ─
section "Storage VIP ($VIP) — NFS + iSCSI (via ${ACTIVE_NODE:-unknown})"
run_active_raw() {
local active_ip=""
[[ "$ACTIVE_NODE" == "$NODE1" ]] && active_ip="$NODE1_IP"
[[ "$ACTIVE_NODE" == "$NODE2" ]] && active_ip="$NODE2_IP"
[[ -z "$active_ip" ]] && return 1
ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no -o ConnectTimeout=5 \
"${HA_USER}@${active_ip}" "$@" 2>/dev/null
}
if [[ -z "$ACTIVE_NODE" ]]; then
echo " Cannot determine active node — skipping"
else
if run_active_raw "ping -c1 -W2 '$VIP' >/dev/null 2>&1"; then
echo " Ping OK"
else
echo " Ping UNREACHABLE"
fi
if run_active_raw "bash -c 'echo >/dev/tcp/${VIP}/2049' 2>/dev/null"; then
printf " %-10s port %-5s OK\n" "NFS" "2049"
else
printf " %-10s port %-5s UNREACHABLE\n" "NFS" "2049"
fi
if run_active_raw "bash -c 'echo >/dev/tcp/${VIP}/3260' 2>/dev/null"; then
printf " %-10s port %-5s OK\n" "iSCSI" "3260"
else
printf " %-10s port %-5s UNREACHABLE\n" "iSCSI" "3260"
fi
fi
echo ""
echo "════════════════════════════════════════════════════"
if [[ -n "$ACTIVE_NODE" ]]; then
echo " Active: $ACTIVE_NODE Standby: $STANDBY_NODE"
else
echo " Active: unknown (Pacemaker not settled)"
fi
echo "════════════════════════════════════════════════════"
echo ""
-274
View File
@@ -1,274 +0,0 @@
#!/usr/bin/env bash
# resize-data-disk.sh — online resize of the HA cluster data disk
#
# Three-phase process (all online-safe, no downtime required):
# 1. Proxmox: grow scsi1 on both HA VMs (qm resize)
# 2. Guest: rescan block device on both nodes so the kernel sees the new size
# 3. DRBD + XFS: drbdadm resize, then xfs_growfs — active node only
#
# Usage:
# scripts/ha/resize-data-disk.sh --size +20G [--force] [--dry-run]
#
# --size +NNg amount to grow scsi1 by, e.g. +20G, +50G (required)
# XFS and DRBD cannot shrink; only positive deltas accepted
# --force skip the interactive confirmation prompt
# --dry-run show what would be done without changing anything
set -euo pipefail
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
# shellcheck source=../env.sh
source "${SCRIPT_DIR}/../env.sh"
# ── Configuration ─────────────────────────────────────────────────────────
NODE1="${NODE1:-ha-server-1}"
NODE2="${NODE2:-ha-server-2}"
NODE1_IP="${NODE1_IP:-192.168.2.228}"
NODE2_IP="${NODE2_IP:-192.168.2.227}"
XFS_MOUNT="${XFS_MOUNT:-/srv/ha-data}"
DRBD_RESOURCE="${DRBD_RESOURCE:-ha-data}"
DATA_DISK_SLOT="${DATA_DISK_SLOT:-scsi1}" # Proxmox disk name (scsi1 = data disk)
HA_USER="${HA_USER:-nixos}"
PVE_HOST="${PVE_HOST:-${PVE1_HOST}}"
PVE_SSH_USER="${PVE_SSH_USER:-${PROXMOX_SSH_USER}}"
PVE_SUDO=""
[[ "$PVE_SSH_USER" != "root" ]] && PVE_SUDO="sudo"
# By-id symlink for the data disk; basename resolves to the raw block device.
# matches variables.nix's haServerDrbdDisk.
DATA_DISK_BYID="${DATA_DISK_BYID:-scsi-0QEMU_QEMU_HARDDISK_drive-${DATA_DISK_SLOT}}"
# ──────────────────────────────────────────────────────────────────────────
pve() { ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no -o ConnectTimeout=10 \
"${PVE_SSH_USER}@${PVE_HOST}" "$@"; }
n1() { ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no -o ConnectTimeout=5 \
"${HA_USER}@${NODE1_IP}" sudo "$@" 2>/dev/null; }
n2() { ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no -o ConnectTimeout=5 \
"${HA_USER}@${NODE2_IP}" sudo "$@" 2>/dev/null; }
# ── Argument parsing ───────────────────────────────────────────────────────
SIZE=""
FORCE=false
DRY_RUN=false
while [[ $# -gt 0 ]]; do
case "$1" in
--size) shift; SIZE="${1:?--size requires a value (e.g. +20G)}" ;;
--force) FORCE=true ;;
--dry-run) DRY_RUN=true ;;
*) echo "Unknown argument: $1"
echo "Usage: $0 --size +NNg [--force] [--dry-run]"
exit 1 ;;
esac
shift
done
if [[ -z "$SIZE" ]]; then
echo "ERROR: --size is required (e.g. --size +20G)"
echo "Usage: $0 --size +NNg [--force] [--dry-run]"
exit 1
fi
# Only positive deltas — qm resize, DRBD, and XFS all refuse to shrink.
if [[ ! "$SIZE" =~ ^\+[0-9]+(G|M|T)$ ]]; then
echo "ERROR: --size must be a positive delta like +20G, +50G, +500M, +2T"
echo " (qm resize, drbdadm resize, and xfs_growfs can only grow, not shrink)"
exit 1
fi
DRY_PREFIX=""
$DRY_RUN && DRY_PREFIX="[dry-run] "
echo "════════════════════════════════════════════════════"
echo " HA Data Disk Resize — $(date '+%Y-%m-%d %H:%M:%S')"
echo " Size delta: $SIZE • Proxmox: $PVE_HOST"
$DRY_RUN && echo " MODE: dry-run — no changes will be made"
echo "════════════════════════════════════════════════════"
# ── Detect active node ─────────────────────────────────────────────────────
echo ""
echo "Detecting active node..."
CRM_OUT=""
if ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no -o ConnectTimeout=5 \
"${HA_USER}@${NODE1_IP}" true 2>/dev/null; then
CRM_OUT=$(n1 "crm_mon -1" 2>/dev/null || true)
fi
if [[ -z "$CRM_OUT" ]]; then
if ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no -o ConnectTimeout=5 \
"${HA_USER}@${NODE2_IP}" true 2>/dev/null; then
CRM_OUT=$(n2 "crm_mon -1" 2>/dev/null || true)
fi
fi
# crm_mon 2.x formats Promoted lines as " * Promoted: [ node ]" (bullet *),
# so ^\s*(Promoted|Masters): never matches; filter Unpromoted first instead.
ACTIVE_NODE=$(echo "$CRM_OUT" | grep -v 'Unpromoted\|Unmanaged' | \
grep -E '(Promoted|Masters):' | \
grep -oE '\b(ha-server-[0-9]+)\b' | head -1 || true)
if [[ -z "$ACTIVE_NODE" ]]; then
echo "ERROR: could not determine active node from crm_mon."
echo " Is Pacemaker still settling? Try running scripts/ha/health.sh first."
exit 1
fi
if [[ "$ACTIVE_NODE" == "$NODE1" ]]; then
ACTIVE_IP="$NODE1_IP"
na() { n1 "$@"; }
else
ACTIVE_IP="$NODE2_IP"
na() { n2 "$@"; }
fi
echo " Active node: $ACTIVE_NODE ($ACTIVE_IP)"
# ── Pre-check DRBD state ───────────────────────────────────────────────────
echo ""
echo "Pre-checks..."
DRBD_DSTATE=$(na "drbdadm dstate ${DRBD_RESOURCE} 2>/dev/null" 2>/dev/null || echo "unknown")
if ! echo "$DRBD_DSTATE" | grep -q "UpToDate/UpToDate"; then
echo " WARNING: DRBD dstate is '$DRBD_DSTATE' (expected UpToDate/UpToDate)."
echo " Resizing with a partially-synced disk may cause issues."
if ! $FORCE; then
echo " Use --force to proceed anyway."
exit 1
fi
echo " --force specified — proceeding despite non-ideal DRBD state."
else
echo " DRBD dstate: $DRBD_DSTATE — OK"
fi
# ── Find VMIDs on Proxmox ─────────────────────────────────────────────────
echo ""
echo "Looking up VM IDs on ${PVE_HOST}..."
QM_LIST=$(pve "$PVE_SUDO qm list 2>/dev/null" || true)
VMID1=$(echo "$QM_LIST" | awk -v name="$NODE1" '$0 ~ name {print $1}' | head -1)
VMID2=$(echo "$QM_LIST" | awk -v name="$NODE2" '$0 ~ name {print $1}' | head -1)
if [[ -z "$VMID1" ]]; then
echo " ERROR: could not find VMID for $NODE1 on $PVE_HOST"
echo " qm list output:"
echo "$QM_LIST" | sed 's/^/ /'
exit 1
fi
if [[ -z "$VMID2" ]]; then
echo " ERROR: could not find VMID for $NODE2 on $PVE_HOST"
echo " qm list output:"
echo "$QM_LIST" | sed 's/^/ /'
exit 1
fi
echo " $NODE1: VMID $VMID1"
echo " $NODE2: VMID $VMID2"
# ── Confirm ────────────────────────────────────────────────────────────────
if ! $FORCE && ! $DRY_RUN; then
echo ""
echo " Plan:"
echo " Phase 1 — qm resize $VMID1 ${DATA_DISK_SLOT} ${SIZE} (on $PVE_HOST)"
echo " qm resize $VMID2 ${DATA_DISK_SLOT} ${SIZE} (on $PVE_HOST)"
echo " Phase 2 — block device rescan on $NODE1 and $NODE2"
echo " Phase 3 — drbdadm resize + xfs_growfs on $ACTIVE_NODE"
echo " No downtime required (all operations are online-safe)."
printf " Proceed? [y/N] "
read -r ANSWER
[[ "${ANSWER,,}" == "y" || "${ANSWER,,}" == "yes" ]] || { echo "Aborted."; exit 0; }
fi
# ═══════════════════════════════════════════════════════════════
# Phase 1 — Resize both VM data disks in Proxmox
# ═══════════════════════════════════════════════════════════════
echo ""
echo "── Phase 1 — Proxmox disk resize (${DATA_DISK_SLOT} ${SIZE} on both VMs) ──"
echo " ${DRY_PREFIX}qm resize $VMID1 ${DATA_DISK_SLOT} ${SIZE} ($NODE1 on ${PVE_HOST})"
if ! $DRY_RUN; then
pve "$PVE_SUDO qm resize $VMID1 ${DATA_DISK_SLOT} ${SIZE}"
fi
echo " ${DRY_PREFIX}qm resize $VMID2 ${DATA_DISK_SLOT} ${SIZE} ($NODE2 on ${PVE_HOST})"
if ! $DRY_RUN; then
pve "$PVE_SUDO qm resize $VMID2 ${DATA_DISK_SLOT} ${SIZE}"
fi
echo " Phase 1 done."
# ═══════════════════════════════════════════════════════════════
# Phase 2 — Rescan block device on both guest nodes
# ═══════════════════════════════════════════════════════════════
echo ""
echo "── Phase 2 — Block device rescan (both nodes) ──"
rescan_node() {
local node_name=$1 run_fn=$2
# Resolve block device name from the stable by-id symlink on the guest.
# Read-only lookup — safe to run even in dry-run so we show the real device.
local blk_dev=""
blk_dev=$($run_fn "bash -c 'basename \$(readlink -f /dev/disk/by-id/${DATA_DISK_BYID})'" 2>/dev/null || true)
if [[ -z "$blk_dev" ]]; then
echo " ERROR: /dev/disk/by-id/${DATA_DISK_BYID} not found on $node_name" >&2
echo " Check DATA_DISK_BYID or DATA_DISK_SLOT configuration." >&2
exit 1
fi
echo " ${DRY_PREFIX}Rescanning /dev/${blk_dev} on ${node_name}..."
if ! $DRY_RUN; then
$run_fn "bash -c 'echo 1 > /sys/block/${blk_dev}/device/rescan'" 2>/dev/null
local new_size
new_size=$($run_fn "lsblk -nd -o SIZE /dev/${blk_dev} 2>/dev/null" 2>/dev/null || echo "unknown")
echo " /dev/${blk_dev} on $node_name now reports: $new_size"
fi
}
rescan_node "$NODE1" n1
rescan_node "$NODE2" n2
echo " Phase 2 done."
# ═══════════════════════════════════════════════════════════════
# Phase 3 — Grow DRBD metadata, then XFS (active node only)
# ═══════════════════════════════════════════════════════════════
echo ""
echo "── Phase 3 — DRBD resize + XFS grow (on active node: $ACTIVE_NODE) ──"
echo " ${DRY_PREFIX}drbdadm resize ${DRBD_RESOURCE}"
if ! $DRY_RUN; then
na "drbdadm resize ${DRBD_RESOURCE}"
fi
echo " ${DRY_PREFIX}xfs_growfs ${XFS_MOUNT}"
if ! $DRY_RUN; then
na "xfs_growfs ${XFS_MOUNT}"
fi
echo " Phase 3 done."
# ── Verify ────────────────────────────────────────────────────────────────
echo ""
echo "── Verify ──"
if ! $DRY_RUN; then
DF_OUT=$(na "df -h '${XFS_MOUNT}' 2>/dev/null" 2>/dev/null || echo "")
if [[ -n "$DF_OUT" ]]; then
echo " ${XFS_MOUNT}:"
echo "$DF_OUT" | sed 's/^/ /'
fi
DRBD_DSTATE_AFTER=$(na "drbdadm dstate ${DRBD_RESOURCE} 2>/dev/null" 2>/dev/null || echo "unknown")
echo " DRBD dstate: $DRBD_DSTATE_AFTER"
if ! echo "$DRBD_DSTATE_AFTER" | grep -q "UpToDate/UpToDate"; then
echo " NOTE: DRBD is resyncing — normal immediately after resize."
echo " Monitor: ssh nixos@${ACTIVE_IP} 'sudo watch -n3 cat /proc/drbd'"
fi
else
echo " [dry-run] would verify df -h ${XFS_MOUNT} and drbdadm dstate on $ACTIVE_NODE"
fi
echo ""
echo "════════════════════════════════════════════════════"
echo " Resize complete."
echo " Active node: $ACTIVE_NODE"
echo "════════════════════════════════════════════════════"
echo ""
-206
View File
@@ -1,206 +0,0 @@
#!/usr/bin/env nix-shell
#!nix-shell -i bash -p jq disko nixos-install-tools zfs
# shellcheck shell=bash
# The only genuinely external tools this script calls directly: `jq`
# (parsing the `nix eval` host list), `disko`/`nixos-install` (the
# install itself), and `zpool` (exporting a ZFS root pool before reboot,
# see the comment above that call below). Everything disko shells out to
# internally (parted/sgdisk/mkfs.*/zfs/...) is self-contained -- disko's
# own generated scripts hardcode absolute Nix store paths for those, they
# don't rely on this script's PATH at all (confirmed by inspecting a
# generated system.build.formatScript). The built installer image
# (modules/installer/common.nix, plus the upstream
# installation-cd-minimal.nix it imports via iso.nix) already has all
# four in environment.systemPackages, so this nix-shell wrapper is a
# fast no-op there; it's what makes the script also work standalone
# (e.g. run directly from a checkout on a stock ISO), where they aren't
# guaranteed.
set -eux
set -euo pipefail
# shellcheck source=../env.sh
source "$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)/env.sh"
export FLAKE_BASE_URL="git+https://${LAN_DOMAIN}/beatzaplenty/nixos.git"
echo "Fetching available NixOS hosts from flake..."
# Two categories deliberately excluded from the menu:
# lxc-* — these build a config.system.build.tarball meant for
# `pct restore` on Proxmox directly, not an install.
# Running nixos-install against one here would
# bind-mount / onto /mnt and then refuse to touch the
# filesystem it's currently running on — see
# docs/auto-installer.md.
# installer — this *is* the installer image's own flake target,
# not a deployable host; "installing" it means
# nixos-install-ing a copy of the installer into
# itself.
mapfile -t options < <(
nix eval --json --no-use-registries --no-accept-flake-config --extra-experimental-features "flakes nix-command" \
"${FLAKE_BASE_URL}#nixosConfigurations" \
--apply builtins.attrNames \
| jq -r '.[]
| select(startswith("lxc-") | not)
| select(. != "installer")'
)
if [[ ${#options[@]} -eq 0 ]]; then
echo "ERROR: No NixOS hosts found in ${FLAKE_BASE_URL}#nixosConfigurations" >&2
exit 1
fi
echo "Note: lxc-* targets aren't installed this way — build them with"
echo " nix build .#nixosConfigurations.<name>.config.system.build.tarball"
echo "and 'pct restore' the result on Proxmox directly. See docs/auto-installer.md."
echo "Choose the flake profile to install:"
select choice in "${options[@]}"; do
if [[ -n "$choice" ]]; then
echo "You selected: $choice"
break
else
echo "Invalid selection. Try again."
fi
done
echo "Starting install with flake: ${FLAKE_BASE_URL}#${choice}"
# Optional: confirm before proceeding
read -rp "Proceed with installation? (y/N): " confirm
if [[ ! "$confirm" =~ ^[Yy]$ ]]; then
echo "Aborted."
exit 1
fi
# A nix-cache host is *the* substituter/remote-builder for every other
# host once installed (its own config explicitly excludes itself from
# using either — see buildType != "nix-cache" in the nixos flake.nix).
# Installing one shouldn't depend on a nix-cache substituter either,
# for the same reason — plus in practice "nix-cache" only resolves over
# Tailscale, which a fresh installer environment was never connected to
# anyway, so it's dead weight even for non-nix-cache installs until
# that's sorted out. Override it away here specifically for nix-cache
# targets to keep install-time behaviour consistent with run-time.
nix_extra_opts=()
if [[ "${choice}" == *-nix-cache ]]; then
echo "Installing a nix-cache host — skipping the nix-cache substituter."
nix_extra_opts+=(--option substituters "https://cache.nixos.org/")
fi
# Every host reachable through this menu has a Disko config (lxc-*
# is filtered out above, and is the only category that doesn't —
# see docs/auto-installer.md), so this can run unconditionally: no
# need to probe the flake first and branch on whether Disko applies.
disko --mode destroy,format,mount \
--flake "${FLAKE_BASE_URL}#${choice}" "${nix_extra_opts[@]}" --yes-wipe-all-disks
# sops-nix derives this host's decryption key from its own SSH host key
# at *activation* time, which runs before systemd would otherwise
# generate one on first boot. Without pre-seeding it here, secrets
# (including the login password) fail to decrypt on first boot.
# Generate the key with scripts/secrets/prepare-host-key.sh first.
#
# Two places a key can come from, checked in order:
# /etc/host-keys — baked into this image at build time (see
# modules/installer/host-keys.nix; only present
# if built with NIXOS_HOST_KEYS_DIR set)
# /root/host-keys — scp'd in manually after boot (older fallback,
# still supported for images built without keys)
mkdir -p /root/host-keys
if [[ -f "/etc/host-keys/${choice}_ssh_host_ed25519_key" ]]; then
echo "Found baked-in SSH host key for ${choice}, installing to target..."
install -D -m 0600 "/etc/host-keys/${choice}_ssh_host_ed25519_key" /mnt/etc/ssh/ssh_host_ed25519_key
install -D -m 0644 "/etc/host-keys/${choice}_ssh_host_ed25519_key.pub" /mnt/etc/ssh/ssh_host_ed25519_key.pub
elif [[ -f "/root/host-keys/${choice}_ssh_host_ed25519_key" ]]; then
echo "Found pre-seeded SSH host key for ${choice}, installing to target..."
install -D -m 0600 "/root/host-keys/${choice}_ssh_host_ed25519_key" /mnt/etc/ssh/ssh_host_ed25519_key
install -D -m 0644 "/root/host-keys/${choice}_ssh_host_ed25519_key.pub" /mnt/etc/ssh/ssh_host_ed25519_key.pub
else
# Third place a key can come from: an arbitrary path the operator
# points at interactively (e.g. a USB stick, a mount from another
# machine) -- only offered when there's an actual human at the other
# end of stdin to ask, never in a non-interactive run.
key_copied=0
if [[ -t 0 ]]; then
echo "No SSH host key found for ${choice} (checked /etc/host-keys and /root/host-keys)."
read -rp "Path to a directory containing ${choice}_ssh_host_ed25519_key(.pub) (blank to skip): " key_src_dir
if [[ -n "$key_src_dir" && -f "${key_src_dir}/${choice}_ssh_host_ed25519_key" && -f "${key_src_dir}/${choice}_ssh_host_ed25519_key.pub" ]]; then
cp "${key_src_dir}/${choice}_ssh_host_ed25519_key" "${key_src_dir}/${choice}_ssh_host_ed25519_key.pub" /root/host-keys/
key_copied=1
elif [[ -n "$key_src_dir" ]]; then
echo "WARNING: ${choice}_ssh_host_ed25519_key(.pub) not found in ${key_src_dir}."
fi
fi
if [[ "$key_copied" -eq 1 ]]; then
echo "Copied SSH host key for ${choice} from ${key_src_dir}, installing to target..."
install -D -m 0600 "/root/host-keys/${choice}_ssh_host_ed25519_key" /mnt/etc/ssh/ssh_host_ed25519_key
install -D -m 0644 "/root/host-keys/${choice}_ssh_host_ed25519_key.pub" /mnt/etc/ssh/ssh_host_ed25519_key.pub
else
echo "WARNING: no SSH host key found for ${choice} (checked /etc/host-keys and /root/host-keys)"
echo "sops-nix secrets (including the login password) will NOT decrypt on first boot."
echo "Run scripts/secrets/prepare-host-key.sh for host ${choice} on your admin workstation first,"
echo "then either rebuild this image with NIXOS_HOST_KEYS_DIR set, scp the result to"
echo "/root/host-keys/ on this machine, or point at it when prompted above."
read -rp "Continue without a pre-seeded key anyway? (y/N): " skip_key
if [[ ! "$skip_key" =~ ^[Yy]$ ]]; then
echo "Aborted."
exit 1
fi
fi
fi
mkdir -p /mnt/install-tmp
export TMPDIR=/mnt/install-tmp
nixos-install \
--flake "${FLAKE_BASE_URL}#${choice}" \
"${nix_extra_opts[@]}" \
--no-root-password
rm -rf /mnt/install-tmp
# Redundant copy of the host's private key — the real one is now at
# /etc/ssh/ssh_host_ed25519_key. Nothing NixOS-managed ever cleans this
# up on its own since it was written imperatively, not declaratively.
rm -rf /root/host-keys
# disko's --mode ...,mount left any ZFS root pool imported (that's what
# let nixos-install write into /mnt). If we reboot with it still
# imported, it isn't just "not exported" -- it's stamped with *this*
# live installer environment's hostid, which almost never matches the
# target's own networking.hostId (see hosts/*/host.nix; the installer
# itself sets none). modules/services/zfs/enable-service.nix and
# modules/common/configuration.nix both set boot.zfs.forceImportRoot =
# false deliberately (the safe option per that setting's own docs), so
# the freshly-installed system's first real boot sees a pool "in use by
# another system" and refuses to import it without -f -- which is what
# makes boot stall waiting on the ZFS import. Exporting here (a no-op
# if the chosen host has no ZFS root, e.g. proxmox-*/linode-*) clears
# that in-use state so the next import, from any hostid, succeeds.
#
# Anything still mounted under /mnt -- nixos-install's own leftover
# chroot bind mounts for running the target's activation script
# (/mnt/dev, /mnt/proc, /mnt/sys, /mnt/run), and disko's own /mnt/boot
# ESP mount (modules/disko/baremetal.nix) -- blocks ZFS from unmounting
# its root dataset at /mnt, the same way any nested mount blocks
# unmounting its parent. Confirmed live: zpool export failed with
# "cannot unmount '/mnt': pool or dataset busy" even after handling the
# chroot mounts alone, because /mnt/boot was still mounted too. Because
# of this script's `set -e`, that killed the script before it ever
# reached reboot, silently defeating the whole point of exporting first.
# Unmounting everything under /mnt up front (recursively, so nested
# mounts like /mnt/dev/pts come along for free) sidesteps needing to
# enumerate every mount disko/nixos-install might leave behind.
if mountpoint -q /mnt; then
umount -R /mnt
fi
if [[ -n "$(zpool list -H -o name 2>/dev/null)" ]]; then
echo "Exporting ZFS pool(s) before reboot..."
zpool export -a
fi
sleep 10
reboot
@@ -1,303 +0,0 @@
#!/usr/bin/env bash
# Add a NixOS host to the FreeIPA domain and produce a sops-encrypted keytab
# at secrets/<hostname>.keytab, ready for modules/ipa/client.nix.
#
# One command replaces three error-prone manual steps:
# 1. ipa host-add on the domain controller
# 2. ipa-getkeytab on the domain controller + SCP back
# 3. sops encrypt in-place (must be at secrets/<hostname>.keytab for
# the creation rule to match -- the common mistake that breaks sops)
#
# Usage:
# scripts/ipa/create-nixos-ipa-host-account.sh [options] <hostname>
#
# Arguments:
# <hostname> Short hostname, e.g. "tailscale-router". The FQDN is
# derived as <hostname>.<HOME_DOMAIN>.
#
# Options:
# --ip <addr> Register this IP with the IPA host record (optional).
# --dc <host> SSH to this host to run IPA commands.
# Default: $IPA_SERVER (from env.sh / environment).
# --dc-user <u> SSH user on the domain controller. Default: wayne.
# --dry-run Print what would be done without making any changes.
# -h, --help Show this message.
#
# Prereqs:
# 1. Run from the repo root (so .sops.yaml and secrets/ are found).
# 2. SSH access to the domain controller as --dc-user (default: wayne)
# with passwordless sudo (or sudo cached). IPA commands and kinit run
# as root via sudo so the Kerberos ticket is in root's cache where all
# ipa tools expect it. If there's no valid ticket, the script runs
# `sudo kinit admin` interactively — you'll be prompted for the IPA
# admin password once. The password never touches this script.
# 3. The host's age key(s) must already be in .sops.yaml. Run
# scripts/secrets/sync-host-keys.sh <flake-target> first so the host
# can decrypt its own keytab on boot. This script adds the .sops.yaml
# creation rule for secrets/<hostname>.keytab automatically, but the
# host age key anchor (&lxc-<hostname> etc.) must already exist —
# otherwise only the admin key can decrypt the keytab and the deployed
# host will fail to read it.
# 4. sops in PATH, or Nix available to run it via `nix run`.
set -euo pipefail
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
REPO_ROOT="$(cd "$SCRIPT_DIR/../.." && pwd)"
# shellcheck source=../env.sh
source "${SCRIPT_DIR}/../env.sh"
# --- Argument parsing ---
DC_HOST="${IPA_SERVER}"
DC_USER="wayne"
IP_ADDR=""
DRY_RUN=false
TARGET=""
usage() {
sed -n '/^# Usage:/,/^[^#]/{ /^#/{ s/^# \?//; p } }' "$0"
exit "${1:-0}"
}
while [[ $# -gt 0 ]]; do
case "$1" in
--ip) IP_ADDR="$2"; shift 2 ;;
--dc) DC_HOST="$2"; shift 2 ;;
--dc-user) DC_USER="$2"; shift 2 ;;
--dry-run) DRY_RUN=true; shift ;;
-h|--help) usage 0 ;;
-*) echo "Unknown flag: $1" >&2; usage 1 ;;
*)
if [[ -n "${TARGET}" ]]; then echo "Unexpected argument: $1" >&2; usage 1; fi
TARGET="$1"; shift
;;
esac
done
if [[ -z "${TARGET}" ]]; then
echo "Error: hostname required." >&2
usage 1
fi
# Reject FQDNs passed by mistake — the script appends HOME_DOMAIN itself.
# "nixos.sweet.home" → FQDN would become "nixos.sweet.home.sweet.home".
if [[ "${TARGET}" == *"."* ]]; then
echo "Error: <hostname> must be the short name (e.g. 'nixos'), not a FQDN." >&2
echo " The FQDN is derived automatically as ${TARGET}.${HOME_DOMAIN}." >&2
exit 1
fi
FQDN="${TARGET}.${HOME_DOMAIN}"
KEYTAB_SECRET="${REPO_ROOT}/secrets/${TARGET}.keytab"
# Temp path on the domain controller — use a name that won't collide.
DC_TMP="/tmp/nixos-keytab-${TARGET}-$$.keytab"
# --- Helpers ---
log() { echo "==> $*"; }
logn() { echo " $*"; }
run() {
if $DRY_RUN; then
echo "[dry-run] $*"
else
"$@"
fi
}
dc_run() {
# Run a command string on the domain controller via SSH.
if $DRY_RUN; then
echo "[dry-run] ssh ${DC_USER}@${DC_HOST} $*"
else
ssh "${DC_USER}@${DC_HOST}" "$@"
fi
}
# --- Locate sops ---
if command -v sops &>/dev/null; then
SOPS_CMD=(sops)
else
log "sops not in PATH — will use 'nix run nixpkgs#sops'"
SOPS_CMD=(nix run "nixpkgs#sops" --)
fi
# --- Preflight checks ---
cd "${REPO_ROOT}"
[[ -f .sops.yaml ]] || { echo "Error: .sops.yaml not found — run from repo root." >&2; exit 1; }
[[ -d secrets ]] || { echo "Error: secrets/ not found — run from repo root." >&2; exit 1; }
# --- Step 1: Ensure .sops.yaml has a creation rule for this keytab ---
#
# sops matches creation rules against the PATH of the file being encrypted,
# not the output path. To match secrets/<hostname>.keytab, the file must
# already be at that path when sops -e -i is called. The creation rule must
# also exist at that point or sops will refuse with "no matching creation
# rules found."
log "Checking .sops.yaml for creation rule: secrets/${TARGET}.keytab"
RULE_EXISTS=false
# Match "path_regex: secrets/<hostname>...keytab" — using .*keytab rather
# than \.keytab because the file stores the regex verbatim (\.keytab = two
# chars: backslash + dot), which a BRE \. (= escaped literal dot) won't span.
if grep -q "path_regex: secrets/${TARGET}.*keytab" .sops.yaml 2>/dev/null; then
RULE_EXISTS=true
logn "Rule already exists — skipping addition."
fi
if ! $RULE_EXISTS; then
# Collect which platform-variant age anchors exist in .sops.yaml for this
# hostname. The keytab is platform-agnostic (same FQDN regardless of
# whether lxc/proxmox/linode variant is deployed), so all platform anchors
# that have been registered get added as recipients.
RECIPIENTS=("*admin")
for platform in lxc proxmox linode; do
anchor="${platform}-${TARGET}"
if grep -q "^ - &${anchor} " .sops.yaml; then
RECIPIENTS+=("*${anchor}")
fi
done
if [[ ${#RECIPIENTS[@]} -eq 1 ]]; then
echo "Warning: no platform age keys found for '${TARGET}' in .sops.yaml." >&2
echo " Run scripts/secrets/sync-host-keys.sh <flake-target> first," >&2
echo " otherwise only the admin key can decrypt the keytab and the" >&2
echo " deployed host won't be able to read it at boot." >&2
echo " Continuing with admin-only encryption..." >&2
fi
# Build the indented recipient list for the YAML block.
RECIPIENT_YAML=""
for r in "${RECIPIENTS[@]}"; do
RECIPIENT_YAML+=" - ${r}"$'\n'
done
RECIPIENT_YAML="${RECIPIENT_YAML%$'\n'}" # strip trailing newline
NEW_RULE="
# Host keytab for ${TARGET} FreeIPA enrollment (binary sops file).
# Generated by scripts/ipa/create-nixos-ipa-host-account.sh.
- path_regex: secrets/${TARGET}\\.keytab\$
key_groups:
- age:
${RECIPIENT_YAML}"
if $DRY_RUN; then
echo "[dry-run] Would append to .sops.yaml:"
echo "${NEW_RULE}"
else
logn "Adding creation rule (recipients: ${RECIPIENTS[*]})"
printf '%s\n' "${NEW_RULE}" >> .sops.yaml
logn "Added."
fi
fi
# --- Step 2: Add IPA host account (idempotent) ---
log "Adding FreeIPA host account: ${FQDN}"
# Ensure there's a valid admin Kerberos ticket on the DC.
# ipa host-add and ipa-getkeytab both need one. All IPA commands run via
# sudo so the ticket must be in root's cache — check and refresh as root.
# ssh -t allocates a PTY so kinit (and sudo if needed) can prompt normally;
# no password ever touches this script or the shell history.
if ! $DRY_RUN; then
if ! ssh "${DC_USER}@${DC_HOST}" "sudo klist -s" &>/dev/null; then
log "No valid Kerberos ticket on ${DC_HOST} — running sudo kinit admin"
ssh -t "${DC_USER}@${DC_HOST}" "sudo kinit admin"
if ! ssh "${DC_USER}@${DC_HOST}" "sudo klist -s" &>/dev/null; then
echo "Error: kinit admin failed or produced no valid ticket." >&2
exit 1
fi
else
logn "Kerberos ticket on ${DC_HOST} is valid."
fi
fi
IP_FLAG=""
[[ -n "${IP_ADDR}" ]] && IP_FLAG="--ip-address=${IP_ADDR}"
# --force: create the host record even if DNS doesn't resolve it yet.
if $DRY_RUN; then
echo "[dry-run] ssh ${DC_USER}@${DC_HOST} sudo ipa host-add '${FQDN}' ${IP_FLAG} --force"
else
HOST_ADD_OUT=$(ssh "${DC_USER}@${DC_HOST}" "sudo ipa host-add '${FQDN}' ${IP_FLAG} --force 2>&1") \
&& HOST_ADD_RC=0 || HOST_ADD_RC=$?
if [[ $HOST_ADD_RC -eq 0 ]]; then
echo "${HOST_ADD_OUT}"
elif echo "${HOST_ADD_OUT}" | grep -q "already exists"; then
logn "(host already registered)"
else
echo "Error: ipa host-add failed (exit ${HOST_ADD_RC}):" >&2
echo "${HOST_ADD_OUT}" >&2
exit 1
fi
fi
# --- Step 3: Fetch the keytab from the domain controller ---
log "Fetching keytab for host/${FQDN}"
# Remove the plaintext keytab if the script aborts before encryption completes.
# The trap is cleared at the end of step 4 once sops has encrypted it in-place.
trap 'rm -f "${KEYTAB_SECRET}"' EXIT
dc_run "sudo ipa-getkeytab -s '${IPA_SERVER}' -p 'host/${FQDN}' -k '${DC_TMP}'"
if $DRY_RUN; then
echo "[dry-run] Would stream ${DC_USER}@${DC_HOST}:${DC_TMP} → secrets/${TARGET}.keytab"
else
logn "Streaming keytab from ${DC_HOST}:${DC_TMP} → secrets/${TARGET}.keytab"
# scp can't read a root-owned temp file as ${DC_USER}; pipe through sudo cat instead.
ssh "${DC_USER}@${DC_HOST}" "sudo cat '${DC_TMP}'" > "${KEYTAB_SECRET}"
logn "Removing temp file on ${DC_HOST}"
dc_run "sudo rm -f '${DC_TMP}'"
fi
# --- Step 4: Encrypt in-place ---
#
# The file must already be at secrets/<hostname>.keytab (done above) so
# sops matches the creation rule by path. Using -i (in-place) rather than
# stdout redirect keeps the path intact through the encrypt call.
log "Encrypting secrets/${TARGET}.keytab in-place with sops"
run "${SOPS_CMD[@]}" -e --input-type binary -i "${KEYTAB_SECRET}"
# Encryption succeeded — the file is now sops-encrypted; cancel the cleanup trap.
trap - EXIT
# --- Done ---
if ! $DRY_RUN; then
echo ""
echo "Done. secrets/${TARGET}.keytab is sops-encrypted and ready."
echo ""
echo "Next steps:"
echo " 1. Verify: grep '\"data\": \"ENC' secrets/${TARGET}.keytab"
echo " 2. Stage and commit:"
echo " git add secrets/${TARGET}.keytab .sops.yaml"
echo " git commit -m 'secrets: add IPA keytab for ${TARGET}'"
echo " 3. Add to hosts/${TARGET}/host.nix (networking block and imports):"
echo ""
echo " networking = {"
echo " hostName = \"${TARGET}\";"
echo " domain = vars.homeDomain; # required for Kerberos FQDN"
echo " nameservers = [ vars.domainControllerIp ]; # IPA DNS"
echo " ..."
echo " };"
echo ""
echo " imports = ["
echo " (import ../../modules/ipa/client.nix {"
echo " keytabSopsFile = ../../secrets/${TARGET}.keytab;"
echo " caCertFile = ../../certs/ipa-ca.crt;"
echo " })"
echo " ];"
echo ""
echo " 4. Deploy: nixos-rebuild switch (or create-proxmox-resource.sh)"
fi
-106
View File
@@ -1,106 +0,0 @@
#!/usr/bin/env bash
# Clan vars helpers: manage SSH host keys stored as clan vars (sops-encrypted
# binary files under vars/per-machine/<target>/openssh/) instead of the
# gitignored host-keys/ directory.
#
# Layout (per clan's convention):
# vars/per-machine/<target>/openssh/ssh_host_ed25519_key/secret -- sops binary (admin-encrypted)
# vars/per-machine/<target>/openssh/ssh_host_ed25519_key.pub/value -- plaintext SSH pubkey
#
# Sourced by create-proxmox-resource.sh and sync-host-keys.sh.
# Depends on sops-age.sh and ssh-host-keys.sh being sourced first (for
# sops_yaml_admin_pubkey, ssh_pubkey_to_age, and NIX_OPTS).
if ! declare -p NIX_OPTS >/dev/null 2>&1; then
declare -a NIX_OPTS=()
fi
# clan_ssh_key_exists <target> <repo_root>
# Returns 0 if clan vars hold a SSH host key for <target>, 1 otherwise.
clan_ssh_key_exists() {
local target="$1" repo_root="$2"
[[ -f "${repo_root}/vars/per-machine/${target}/openssh/ssh_host_ed25519_key/secret" ]]
}
# clan_ssh_pubkey_path <target> <repo_root>
# Prints the path to the plaintext SSH public key value file.
clan_ssh_pubkey_path() {
local target="$1" repo_root="$2"
echo "${repo_root}/vars/per-machine/${target}/openssh/ssh_host_ed25519_key.pub/value"
}
# clan_decrypt_ssh_key <target> <repo_root> <dest_dir>
# Decrypts the sops-encrypted SSH host private key for <target> into <dest_dir>,
# naming it <target>_ssh_host_ed25519_key (to match NIXOS_HOST_KEYS_DIR
# conventions that lxc.nix and the disko build already expect). Also copies
# the plaintext public key. The caller is responsible for protecting and
# cleaning up <dest_dir>.
clan_decrypt_ssh_key() {
local target="$1" repo_root="$2" dest_dir="$3"
local secret="${repo_root}/vars/per-machine/${target}/openssh/ssh_host_ed25519_key/secret"
local pubval="${repo_root}/vars/per-machine/${target}/openssh/ssh_host_ed25519_key.pub/value"
local dest_priv="${dest_dir}/${target}_ssh_host_ed25519_key"
local dest_pub="${dest_dir}/${target}_ssh_host_ed25519_key.pub"
nix-shell "${NIX_OPTS[@]}" -p sops --run \
"sops -d --output-type binary '${secret}'" > "$dest_priv"
chmod 0600 "$dest_priv"
cp "$pubval" "$dest_pub"
}
# clan_generate_ssh_key <target> <repo_root>
# Generates a new SSH host key pair and stores it in clan vars format:
# - private key: sops binary-encrypted for the admin age key
# - public key: plaintext value file
# Idempotent: if the secret already exists, prints a note and returns 0.
# Requires sops_yaml_admin_pubkey (from sops-age.sh) to be available.
clan_generate_ssh_key() {
local target="$1" repo_root="$2"
local var_base="${repo_root}/vars/per-machine/${target}/openssh"
local secret_dir="${var_base}/ssh_host_ed25519_key"
local pubval_dir="${var_base}/ssh_host_ed25519_key.pub"
if [[ -f "${secret_dir}/secret" ]]; then
echo "Clan SSH host key for ${target} already exists -- skipping generation."
return 0
fi
# Resolve admin age public key from .sops.yaml
local admin_pubkey
admin_pubkey="$(sops_yaml_admin_pubkey "${repo_root}/.sops.yaml")"
if [[ -z "$admin_pubkey" ]]; then
echo "ERROR: Could not find &admin age key in ${repo_root}/.sops.yaml" >&2
return 1
fi
# Generate the SSH key pair in a secure temp directory
local tmpdir
tmpdir="$(mktemp -d)"
local priv_tmp="${tmpdir}/ssh_host_ed25519_key"
# shellcheck disable=SC2064
trap "rm -rf '${tmpdir}'" RETURN
nix-shell "${NIX_OPTS[@]}" -p openssh --run \
"ssh-keygen -t ed25519 -N '' -C '${target}' -f '${priv_tmp}'" >/dev/null
# Create a minimal sops config that uses only the admin age key -- this
# prevents sops from merging in ALL recipients from .sops.yaml (which
# would unnecessarily encrypt for every host's key, not just admin).
local sops_cfg="${tmpdir}/sops-config.json"
printf '{"creation_rules":[{"key_groups":[{"age":["%s"]}]}]}\n' \
"$admin_pubkey" > "$sops_cfg"
# Encrypt the private key in sops binary format (admin-only recipient)
mkdir -p "$secret_dir" "$pubval_dir"
nix-shell "${NIX_OPTS[@]}" -p sops --run \
"sops -e --config '${sops_cfg}' --input-type binary '${priv_tmp}'" \
> "${secret_dir}/secret"
# Store the public key as a plaintext value file
cp "${priv_tmp}.pub" "${pubval_dir}/value"
echo "Generated and stored clan SSH host key for ${target}."
echo " Private key: ${secret_dir}/secret (sops binary, admin-key encrypted)"
echo " Public key: ${pubval_dir}/value"
}
-95
View File
@@ -1,95 +0,0 @@
#!/usr/bin/env bash
# Shared parallel-nix-invocation helper for scripts/codex-maintenance.sh.
# Source alongside nix-eval.sh:
# source "$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)/lib/nix-parallel.sh"
#
# The per-host/per-package `nix eval`/`nix build --dry-run` calls in
# codex-maintenance.sh are independent of each other, so running them one at
# a time leaves most cores idle for most of the sweep -- run_nix_parallel
# fans a batch of them out across up to NIX_PARALLEL_JOBS processes instead.
# NIX_PARALLEL_JOBS: how many `nix` invocations run_nix_parallel runs at
# once. Defaults to core count capped by available memory (~1GB/job,
# floor 1) rather than plain `nproc` -- each concurrent `nix eval` here
# evaluates a whole NixOS system closure from scratch, and on a small/
# memory-constrained CI runner, `nproc` concurrent evals can OOM-kill each
# other (confirmed empirically: on a 4GB/6-core box, 5-6 concurrent evals
# started getting killed while 3-4 ran clean and were still ~2x faster than
# serial). Override via env if a given machine/CI runner has room to spare
# or needs a tighter cap.
default_nix_parallel_jobs() {
local cores mem_avail_kb mem_cap
cores="$(nproc 2>/dev/null || echo 4)"
mem_avail_kb="$(awk '/^MemAvailable:/ {print $2}' /proc/meminfo 2>/dev/null)"
if [[ -z "$mem_avail_kb" ]]; then
echo "$cores"
return
fi
mem_cap=$((mem_avail_kb / 1024 / 1024))
((mem_cap < 1)) && mem_cap=1
((mem_cap < cores)) && echo "$mem_cap" || echo "$cores"
}
NIX_PARALLEL_JOBS="${NIX_PARALLEL_JOBS:-$(default_nix_parallel_jobs)}"
# Separator between a job's label and its flake attr in the arrays
# run_nix_parallel takes -- a control character so it can't collide with
# anything a label or attr path would plausibly contain.
NIX_PARALLEL_SEP=$'\x1f'
# run_nix_parallel <jobs_array_name> <nix subcommand + flags...>
#
# jobs_array_name: name of an already-populated bash array whose entries are
# "<label>${NIX_PARALLEL_SEP}<attr>" pairs, e.g.
# jobs=("proxmox-docker${NIX_PARALLEL_SEP}.#nixosConfigurations.proxmox-docker...drvPath")
# Remaining args are passed to `nix` before the attr, e.g.:
# run_nix_parallel jobs eval --raw "${NIX_EVAL_FLAGS[@]}"
# run_nix_parallel jobs build --dry-run --no-link "${NIX_EVAL_FLAGS[@]}"
#
# Prints "==> <label>" followed by that job's stdout+stderr for every job,
# in submission order (not completion order) so a run stays readable and
# diffable across invocations even though the work itself doesn't finish in
# that order. Returns non-zero if any job failed, only after every job has
# finished and been printed -- same "surface everything, then fail" contract
# a `set -e` caller gets, just parallelized instead of stopping at the first
# failure.
run_nix_parallel() {
local -n jobs_ref="$1"
shift
local -a nix_args=("$@")
local n=${#jobs_ref[@]}
[[ $n -eq 0 ]] && return 0
local tmp_dir
tmp_dir="$(mktemp -d)"
local i=0 running=0
for job in "${jobs_ref[@]}"; do
local attr="${job#*"${NIX_PARALLEL_SEP}"}"
printf '%s\n' "${job%%"${NIX_PARALLEL_SEP}"*}" >"${tmp_dir}/${i}.label"
(
if nix "${nix_args[@]}" "$attr" >"${tmp_dir}/${i}.out" 2>&1; then
echo 0 >"${tmp_dir}/${i}.status"
else
echo 1 >"${tmp_dir}/${i}.status"
fi
) &
i=$((i + 1))
running=$((running + 1))
if ((running >= NIX_PARALLEL_JOBS)); then
wait -n
running=$((running - 1))
fi
done
wait
local failed=0 j
for ((j = 0; j < n; j++)); do
echo "==> $(cat "${tmp_dir}/${j}.label")"
cat "${tmp_dir}/${j}.out"
[[ "$(cat "${tmp_dir}/${j}.status")" -ne 0 ]] && failed=1
done
rm -rf "$tmp_dir"
return $failed
}
-221
View File
@@ -1,221 +0,0 @@
#!/usr/bin/env bash
# Ad hoc clone of a single VM/CT from pve1 (production) to pve-test
# (sandbox), via vzdump + qmrestore/pct restore -- not a general-purpose
# backup tool, just a quick "give me a disposable copy of this thing on
# pve-test" for testing against real-ish data without touching prod.
#
# Flow:
# 1. vzdump the resource on pve1 into its "local" storage (--mode
# snapshot by default, so the source keeps running throughout --
# see --mode below for when that's not possible).
# 2. Stream the resulting archive straight from pve1 to pve-test
# (ssh pve1 cat ... | ssh pve-test cat > ...) -- this machine is
# just the relay, no separate on-disk staging copy here.
# 3. qmrestore / pct restore it on pve-test under --new-vmid (default:
# same VMID as the source -- pve-test is a separate node/cluster, so
# no collision unless that VMID is already in use there too).
# Always restored with --unique 1 (fresh MAC addresses) since the
# source is typically still running on the same LAN -- restoring
# with the *same* MAC would put two live guests on the wire with
# identical hardware addresses.
# 4. Delete the vzdump archive from pve1's local storage and the
# relayed copy on pve-test, so neither node accumulates ad hoc
# backup files from this script. Only the pve1 original is
# preserved on any failure after step 1, so a failed
# transfer/restore can be retried without re-running the backup.
#
# This script's own defaults are pve1 -> pve-test, unlike
# create-proxmox-resource.sh's --node (which defaults to production) --
# see CLAUDE.md's "Two Proxmox nodes" section. pve1 is only ever touched
# here after typing the source VMID back to confirm; pve-test is treated
# as disposable, matching this repo's usual policy for that node.
#
# See --help for the full option list.
set -euo pipefail
repo_root="$(cd "$(dirname "$0")/../.." && pwd)"
# shellcheck source=../env.sh
source "${repo_root}/scripts/env.sh"
# shellcheck source=../lib/confirm.sh
source "${repo_root}/scripts/lib/confirm.sh"
usage() {
cat <<EOF
Usage: $0 --vmid <n> [options]
--vmid <n> Required: VMID on the source node to clone.
Kind (qemu VM vs LXC CT) is auto-detected.
--new-vmid <n> VMID to restore as on the target node
(default: same as --vmid).
--mode snapshot|suspend|stop
vzdump backup mode (default: snapshot -- the
source resource keeps running throughout;
requires snapshot-capable storage, e.g.
ZFS/LVM-thin/Ceph/qcow2). Fall back to
"suspend" (brief pause) or "stop" (source
goes down for the duration) if the source's
storage doesn't support live snapshots --
vzdump's own error will say so.
--source-node <host> (default: \$PVE1_HOST, ${PVE1_HOST})
--target-node <host> (default: \$PVE_TEST_HOST, ${PVE_TEST_HOST})
--source-storage <pool> Where vzdump writes the backup on the
source node (default: local).
--target-storage <pool> Where the restored disk/rootfs lands on
the target node (default: \$PROXMOX_STORAGE, ${PROXMOX_STORAGE}).
--keep-backup Don't delete the vzdump archive from
either node afterward (debugging aid).
--yes Skip the typed VMID confirmation
before touching the source node.
--dry-run Print the full plan and skip every
mutating step (vzdump, transfer,
restore, delete) and the confirm
prompt. Still makes read-only SSH
calls to look up the source kind
and check the target VMID is free
-- harmless on either node.
-h, --help
EOF
}
vmid=""
new_vmid=""
mode="snapshot"
source_node="$PVE1_HOST"
target_node="$PVE_TEST_HOST"
source_storage="local"
target_storage="$PROXMOX_STORAGE"
keep_backup=0
skip_confirm=0
dry_run=0
while [[ $# -gt 0 ]]; do
case "$1" in
--vmid) vmid="$2"; shift 2 ;;
--new-vmid) new_vmid="$2"; shift 2 ;;
--mode) mode="$2"; shift 2 ;;
--source-node) source_node="$2"; shift 2 ;;
--target-node) target_node="$2"; shift 2 ;;
--source-storage) source_storage="$2"; shift 2 ;;
--target-storage) target_storage="$2"; shift 2 ;;
--keep-backup) keep_backup=1; shift ;;
--yes) skip_confirm=1; shift ;;
--dry-run) dry_run=1; shift ;;
-h | --help) usage; exit 0 ;;
*) echo "Unknown option: $1" >&2; usage >&2; exit 1 ;;
esac
done
if [[ -z "$vmid" ]]; then
echo "ERROR: --vmid is required." >&2
usage >&2
exit 1
fi
if [[ "$mode" != "snapshot" && "$mode" != "suspend" && "$mode" != "stop" ]]; then
echo "ERROR: --mode must be snapshot, suspend, or stop." >&2
exit 1
fi
[[ -z "$new_vmid" ]] && new_vmid="$vmid"
source_target="${PROXMOX_SSH_USER}@${source_node}"
target_target="${PROXMOX_SSH_USER}@${target_node}"
# No dry-run wrapper needed for the calls below: every mutating step
# (vzdump, transfer, restore, delete) is reached only after the --dry-run
# early-exit further down, so a plain `ssh` call is never in the dry-run
# path.
# --- identify the resource kind on the source node -----------------------
echo "==> Looking up VMID ${vmid} on ${source_node}..."
kind=""
if ssh "$source_target" "qm status ${vmid}" >/dev/null 2>&1; then
kind="vm"
elif ssh "$source_target" "pct status ${vmid}" >/dev/null 2>&1; then
kind="lxc"
else
echo "ERROR: VMID ${vmid} doesn't exist on ${source_node} as either a VM or CT." >&2
exit 1
fi
echo "VMID ${vmid} on ${source_node} is a ${kind}."
# --- refuse to clobber an existing resource on the target node -----------
if ssh "$target_target" "qm status ${new_vmid}" >/dev/null 2>&1 \
|| ssh "$target_target" "pct status ${new_vmid}" >/dev/null 2>&1; then
echo "ERROR: VMID ${new_vmid} already exists on ${target_node}. Pass --new-vmid" >&2
echo "with a free ID, or remove the existing resource there first." >&2
exit 1
fi
echo
echo "Plan:"
echo " source: ${kind} VMID ${vmid} on ${source_node} (storage: ${source_storage}, mode: ${mode})"
echo " target: VMID ${new_vmid} on ${target_node} (storage: ${target_storage}, fresh MAC via --unique)"
[[ "$keep_backup" -eq 1 ]] && echo " backup archives are kept on both nodes afterward (--keep-backup)"
if [[ "$dry_run" -eq 1 ]]; then
echo
echo "[dry-run] No backup, transfer, restore, or delete was performed."
exit 0
fi
if [[ "$skip_confirm" -ne 1 ]]; then
echo
if ! confirm_typed "$vmid" "Type the source VMID (${vmid}) to confirm backing it up from ${source_node}: "; then
echo "Cancelled -- input didn't match ${vmid}." >&2
exit 1
fi
fi
# --- vzdump on the source node --------------------------------------------
echo
echo "==> Backing up VMID ${vmid} on ${source_node} (mode=${mode}, storage=${source_storage})..."
vzdump_log="$(ssh "$source_target" \
"vzdump ${vmid} --mode ${mode} --storage ${source_storage} --compress zstd" 2>&1)" \
|| {
echo "$vzdump_log" >&2
echo "ERROR: vzdump failed on ${source_node}." >&2
exit 1
}
echo "$vzdump_log"
archive="$(echo "$vzdump_log" | grep -oP "creating vzdump archive '\K[^']+" | tail -n1)"
if [[ -z "$archive" ]]; then
echo "ERROR: couldn't find the archive path in vzdump's output above." >&2
exit 1
fi
archive_basename="$(basename "$archive")"
target_tmp_archive="/var/tmp/${archive_basename}"
echo "Archive: ${archive}"
# Always clean up the relayed copy on the target node, success or failure
# -- it's only ever a working copy, restored or not.
cleanup_target_tmp() {
if [[ "$keep_backup" -ne 1 ]]; then
ssh "$target_target" "rm -f '${target_tmp_archive}'" >/dev/null 2>&1 || true
fi
}
trap cleanup_target_tmp EXIT
# --- relay the archive from source to target ------------------------------
echo
echo "==> Transferring archive to ${target_node}..."
ssh "$source_target" "cat '${archive}'" | ssh "$target_target" "cat > '${target_tmp_archive}'"
# --- restore on the target node --------------------------------------------
echo
echo "==> Restoring as VMID ${new_vmid} on ${target_node} (storage=${target_storage})..."
if [[ "$kind" == "vm" ]]; then
ssh "$target_target" "qmrestore '${target_tmp_archive}' ${new_vmid} --storage ${target_storage} --unique 1"
else
ssh "$target_target" "pct restore ${new_vmid} '${target_tmp_archive}' --storage ${target_storage} --unique 1"
fi
# --- clean up the source backup now that the restore succeeded -----------
if [[ "$keep_backup" -ne 1 ]]; then
echo
echo "==> Deleting backup archive from ${source_node}'s ${source_storage} storage..."
ssh "$source_target" "rm -f '${archive}' '${archive}.notes' '${archive}.log'" >/dev/null 2>&1 || true
fi
echo
echo "Done. VMID ${new_vmid} (${kind}) is now on ${target_node}, cloned from" \
"VMID ${vmid} on ${source_node}."
+27 -44
View File
@@ -6,31 +6,27 @@
# #
# This is the non-NixOS equivalent of modules/nix-cache/client.nix + # This is the non-NixOS equivalent of modules/nix-cache/client.nix +
# modules/nix-cache/remote-builder-client.nix -- those two only apply to # modules/nix-cache/remote-builder-client.nix -- those two only apply to
# hosts built from this flake. A plain Debian box with Nix installed has no # hosts built from this flake. A plain Debian box with Nix installed
# NixOS module system to pick that config up, so this edits nix.conf by hand. # (single- or multi-user install, nix-daemon running) has no NixOS module
# # system to pick that config up, so this edits /etc/nix/nix.conf by hand
# Two modes depending on who runs it: # instead. Run this ON the target Debian machine, as root.
#
# root (multi-user / daemon install):
# Writes /etc/nix/nix.conf, /etc/ssh/ssh_known_hosts, restarts nix-daemon.
# Requires /etc/nix/nix.conf to already exist (i.e. nix-daemon is set up).
# Run as: sudo ./configure-nix-cache-client.sh [options]
#
# non-root (single-user install):
# Writes ~/.config/nix/nix.conf, ~/.ssh/known_hosts. No daemon to restart.
# Run as: ./configure-nix-cache-client.sh [options]
# #
# The values below mirror variables.nix / modules/nix-cache/client.nix in # The values below mirror variables.nix / modules/nix-cache/client.nix in
# this repo -- update both if nix-cache is ever rebuilt with a new host # this repo -- update both if nix-cache is ever rebuilt with a new host
# key or the cache signing key is rotated (see docs/nix-cache.md). # key or the cache signing key is rotated (see docs/nix-cache.md).
# #
# REMOTE_BUILDER_KEY defaults to the running user's default SSH identity # REMOTE_BUILDER_KEY defaults to this machine's own default root SSH
# (root: /root/.ssh/id_ed25519, other user: ~/.ssh/id_ed25519). That key # identity (matches modules/nix-cache/remote-builder-client.nix's
# must be listed in vars.remoteBuilderAuthorizedKeys in this repo and # convention for real NixOS clients: authenticate as nixremote with the
# nix-cache rebuilt before remote building works. # host's own default key, added individually to
# vars.remoteBuilderAuthorizedKeys, rather than a separately-named or
# shared keypair) -- generate one with
# `ssh-keygen -t ed25519 -N '' -f /root/.ssh/id_ed25519` if this machine
# doesn't have one yet, then add its .pub to vars.remoteBuilderAuthorizedKeys
# and rebuild nix-cache.
# #
# Usage: # Usage:
# ./configure-nix-cache-client.sh [--dry-run] [--no-remote-builder] [--no-restart] # sudo ./configure-nix-cache-client.sh [--dry-run] [--no-remote-builder] [--no-restart]
# #
# Env overrides (defaults match variables.nix): # Env overrides (defaults match variables.nix):
# NIX_CACHE_HOST, NIX_CACHE_HOST_KEY, REMOTE_BUILDER_USER, REMOTE_BUILDER_KEY # NIX_CACHE_HOST, NIX_CACHE_HOST_KEY, REMOTE_BUILDER_USER, REMOTE_BUILDER_KEY
@@ -38,30 +34,19 @@
set -euo pipefail set -euo pipefail
: "${NIX_CACHE_HOST:=nix-cache}" : "${NIX_CACHE_HOST:=nix-cache}"
: "${NIX_CACHE_HOST_KEY:=ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAICuHUxGNH6ei3BZD+EfZs3l4X8uJNcjQiOsM/G4yo4O/ lxc-nix-cache}" : "${NIX_CACHE_HOST_KEY:=ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIPeWgMsdaiz4axT/deFc1+0B5bN+GX/NOeW9bbQ0c/IT lxc-nix-cache}"
: "${REMOTE_BUILDER_USER:=nixremote}" : "${REMOTE_BUILDER_USER:=nixremote}"
: "${REMOTE_BUILDER_KEY:=/root/.ssh/id_ed25519}"
CACHE_PUB_KEY="cache.local-1:usoWYanY3Kpq2+kDIS2nhWoLZiRxanmdysdzqCFBHW4=" CACHE_PUB_KEY="cache.local-1:usoWYanY3Kpq2+kDIS2nhWoLZiRxanmdysdzqCFBHW4="
FALLBACK_URL="https://cache.nixos.org/" FALLBACK_URL="https://cache.nixos.org/"
FALLBACK_PUB_KEY="cache.nixos.org-1:6NCHdD59X431o0gWypbMrAURkbJ16ZPMQFGspcDShjY=" FALLBACK_PUB_KEY="cache.nixos.org-1:6NCHdD59X431o0gWypbMrAURkbJ16ZPMQFGspcDShjY="
NIX_CONF="/etc/nix/nix.conf"
KNOWN_HOSTS="/etc/ssh/ssh_known_hosts"
MARKER_BEGIN="# BEGIN nix-cache client config (configure-nix-cache-client.sh)" MARKER_BEGIN="# BEGIN nix-cache client config (configure-nix-cache-client.sh)"
MARKER_END="# END nix-cache client config" MARKER_END="# END nix-cache client config"
# Mode: root uses system-wide paths and restarts the daemon; non-root uses
# user-level paths and has no daemon to restart.
if [[ "$EUID" -eq 0 ]]; then
install_mode="multi"
NIX_CONF="/etc/nix/nix.conf"
KNOWN_HOSTS="/etc/ssh/ssh_known_hosts"
: "${REMOTE_BUILDER_KEY:=/root/.ssh/id_ed25519}"
else
install_mode="single"
NIX_CONF="${XDG_CONFIG_HOME:-$HOME/.config}/nix/nix.conf"
KNOWN_HOSTS="$HOME/.ssh/known_hosts"
: "${REMOTE_BUILDER_KEY:=$HOME/.ssh/id_ed25519}"
fi
dry_run=0 dry_run=0
with_remote_builder=1 with_remote_builder=1
restart_daemon=1 restart_daemon=1
@@ -72,7 +57,7 @@ for arg in "$@"; do
--no-remote-builder) with_remote_builder=0 ;; --no-remote-builder) with_remote_builder=0 ;;
--no-restart) restart_daemon=0 ;; --no-restart) restart_daemon=0 ;;
-h|--help) -h|--help)
sed -n '2,37p' "$0" sed -n '2,20p' "$0"
exit 0 exit 0
;; ;;
*) *)
@@ -82,22 +67,21 @@ for arg in "$@"; do
esac esac
done done
if [[ "$dry_run" -eq 0 && "$EUID" -ne 0 ]]; then
echo "ERROR: must run as root (writes $NIX_CONF and, unless --no-remote-builder, $KNOWN_HOSTS)." >&2
exit 1
fi
if ! command -v nix >/dev/null 2>&1; then if ! command -v nix >/dev/null 2>&1; then
echo "ERROR: no 'nix' binary on PATH -- install the Nix package manager first." >&2 echo "ERROR: no 'nix' binary on PATH -- install the Nix package manager first." >&2
exit 1 exit 1
fi fi
if [[ "$install_mode" == "multi" && ! -f "$NIX_CONF" ]]; then if [[ ! -f "$NIX_CONF" ]]; then
echo "ERROR: $NIX_CONF not found -- expected an existing multi-user Nix install." >&2 echo "ERROR: $NIX_CONF not found -- expected an existing multi-user Nix install." >&2
exit 1 exit 1
fi fi
# Single-user: create the config file if it doesn't exist yet.
if [[ "$install_mode" == "single" && "$dry_run" -eq 0 ]]; then
mkdir -p "$(dirname "$NIX_CONF")"
[[ -f "$NIX_CONF" ]] || touch "$NIX_CONF"
fi
builder_line="" builder_line=""
if [[ "$with_remote_builder" -eq 1 ]]; then if [[ "$with_remote_builder" -eq 1 ]]; then
if [[ -f "$REMOTE_BUILDER_KEY" ]]; then if [[ -f "$REMOTE_BUILDER_KEY" ]]; then
@@ -133,7 +117,7 @@ fi
block="${block} block="${block}
$MARKER_END" $MARKER_END"
echo "== nix.conf block to install ($NIX_CONF) ==" echo "== nix.conf block to install =="
echo "$block" echo "$block"
echo "================================" echo "================================"
@@ -175,8 +159,7 @@ if [[ "$with_remote_builder" -eq 1 ]]; then
fi fi
fi fi
# Only restart the daemon for multi-user installs -- single-user has no daemon. if [[ "$dry_run" -eq 0 && "$restart_daemon" -eq 1 ]]; then
if [[ "$dry_run" -eq 0 && "$restart_daemon" -eq 1 && "$install_mode" == "multi" ]]; then
if command -v systemctl >/dev/null 2>&1 && systemctl is-active --quiet nix-daemon 2>/dev/null; then if command -v systemctl >/dev/null 2>&1 && systemctl is-active --quiet nix-daemon 2>/dev/null; then
systemctl restart nix-daemon systemctl restart nix-daemon
echo "Restarted nix-daemon to pick up the new config." echo "Restarted nix-daemon to pick up the new config."
+93 -206
View File
@@ -8,12 +8,9 @@
# script -- there's no multi-gigabyte image to transfer afterward. The first # script -- there's no multi-gigabyte image to transfer afterward. The first
# time a node doesn't have that repo path yet, it's bootstrapped: cloned from # time a node doesn't have that repo path yet, it's bootstrapped: cloned from
# this checkout's own `origin` remote, then scripts/codex-setup.sh installs # this checkout's own `origin` remote, then scripts/codex-setup.sh installs
# the build tooling (Nix, etc.). Every run after that just `git pull`s it. # the build tooling (Nix, etc.). Every run after that just `git pull`s it and
# SSH host keys are stored as clan vars (vars/per-machine/<target>/openssh/, # copies over the locally-managed host-keys/ (gitignored, so a git pull
# committed and sops-encrypted) -- the script decrypts them locally and # alone wouldn't carry it) before building.
# copies only the two files for this target to the node's host-keys/ before
# building. A target with no clan var is an error (generate one first with
# scripts/secrets/sync-host-keys.sh <target>).
# #
# --node (default: $PROXMOX_HOST, see scripts/env.sh) picks which of the two # --node (default: $PROXMOX_HOST, see scripts/env.sh) picks which of the two
# LAN Proxmox nodes this runs against: production, pve1.sweet.home # LAN Proxmox nodes this runs against: production, pve1.sweet.home
@@ -62,10 +59,6 @@ source "${repo_root}/scripts/env.sh"
source "${repo_root}/scripts/lib/nix-eval.sh" source "${repo_root}/scripts/lib/nix-eval.sh"
# shellcheck source=../lib/confirm.sh # shellcheck source=../lib/confirm.sh
source "${repo_root}/scripts/lib/confirm.sh" source "${repo_root}/scripts/lib/confirm.sh"
# shellcheck source=../lib/sops-age.sh
source "${repo_root}/scripts/lib/sops-age.sh"
# shellcheck source=../lib/clan-vars.sh
source "${repo_root}/scripts/lib/clan-vars.sh"
sync_keys="${repo_root}/scripts/secrets/sync-host-keys.sh" sync_keys="${repo_root}/scripts/secrets/sync-host-keys.sh"
@@ -200,16 +193,6 @@ done
ssh_target="${PROXMOX_SSH_USER}@${node}" ssh_target="${PROXMOX_SSH_USER}@${node}"
# Proxmox tools (pvesh, qm, pct) require root access to the cluster IPC
# socket. When SSH-ing as a non-root user with sudo, prefix every remote
# Proxmox command with sudo.
sudo_prefix=""
sudo_display=""
if [[ "$PROXMOX_SSH_USER" != "root" ]]; then
sudo_prefix="sudo"
sudo_display="sudo "
fi
remote() { remote() {
if [[ "$dry_run" -eq 1 ]]; then if [[ "$dry_run" -eq 1 ]]; then
echo "[dry-run] ssh ${ssh_target} -- $*" echo "[dry-run] ssh ${ssh_target} -- $*"
@@ -231,10 +214,10 @@ cmd_modify() {
echo "Looking up VMID ${vmid} on ${node}..." echo "Looking up VMID ${vmid} on ${node}..."
local kind current_cores current_memory disk_key local kind current_cores current_memory disk_key
if ssh "$ssh_target" "${sudo_prefix} qm status ${vmid}" >/dev/null 2>&1; then if ssh "$ssh_target" "qm status ${vmid}" >/dev/null 2>&1; then
kind="vm" kind="vm"
disk_key="scsi0" disk_key="scsi0"
elif ssh "$ssh_target" "${sudo_prefix} pct status ${vmid}" >/dev/null 2>&1; then elif ssh "$ssh_target" "pct status ${vmid}" >/dev/null 2>&1; then
kind="lxc" kind="lxc"
disk_key="rootfs" disk_key="rootfs"
else else
@@ -242,8 +225,8 @@ cmd_modify() {
exit 1 exit 1
fi fi
local config_cmd="${sudo_prefix} qm config ${vmid}" local config_cmd="qm config ${vmid}"
[[ "$kind" == "lxc" ]] && config_cmd="${sudo_prefix} pct config ${vmid}" [[ "$kind" == "lxc" ]] && config_cmd="pct config ${vmid}"
local current_config local current_config
current_config="$(ssh "$ssh_target" "$config_cmd")" current_config="$(ssh "$ssh_target" "$config_cmd")"
current_cores="$(echo "$current_config" | grep -oP '^cores:\s*\K\S+' || echo '?')" current_cores="$(echo "$current_config" | grep -oP '^cores:\s*\K\S+' || echo '?')"
@@ -267,9 +250,9 @@ cmd_modify() {
exit 1 exit 1
fi fi
local set_cmd="${sudo_prefix} qm set" local set_cmd="qm set"
local resize_cmd="${sudo_prefix} qm resize" local resize_cmd="qm resize"
[[ "$kind" == "lxc" ]] && set_cmd="${sudo_prefix} pct set" && resize_cmd="${sudo_prefix} pct resize" [[ "$kind" == "lxc" ]] && set_cmd="pct set" && resize_cmd="pct resize"
if [[ -n "$cores" || -n "$memory" ]]; then if [[ -n "$cores" || -n "$memory" ]]; then
local args="" local args=""
@@ -302,12 +285,6 @@ platform_prefix="lxc"
[[ -z "$cores" ]] && cores="$PROXMOX_DEFAULT_CORES" [[ -z "$cores" ]] && cores="$PROXMOX_DEFAULT_CORES"
[[ -z "$memory" ]] && memory="$PROXMOX_DEFAULT_MEMORY_MB" [[ -z "$memory" ]] && memory="$PROXMOX_DEFAULT_MEMORY_MB"
if [[ "$type" == "vm" && -n "$disk_size" ]]; then
echo "WARNING: --disk-size is LXC-only for create mode and is ignored for VMs." >&2
echo " VM disk size comes from proxmoxImageSize in variables.nix (currently ${disk_size}G was requested)." >&2
echo " To expand after creation, use: --modify --vmid <n> --grow-disk <GB>" >&2
fi
# --- discover / resolve the flake target from --host -------------------- # --- discover / resolve the flake target from --host --------------------
# Emits "<target>\t<hostName>" pairs for every ${platform_prefix}-* flake # Emits "<target>\t<hostName>" pairs for every ${platform_prefix}-* flake
# target -- the one source both --list and the --host lookup below read # target -- the one source both --list and the --host lookup below read
@@ -360,14 +337,6 @@ fi
# feeds straight into the guest's real hostname) disagree with host.nix. # feeds straight into the guest's real hostname) disagree with host.nix.
[[ -z "$name" ]] && name="$host" [[ -z "$name" ]] && name="$host"
# For VM builds: the diskoImagesScript (run via QEMU on the node) writes the
# raw disk image as <hostname>.raw into the CWD it was called from (the remote
# repo dir), not to /var/lib/vz/import/ or anywhere else. Import directly from
# there -- no intermediate mv that can fail crossing filesystem boundaries or
# leave a stale file on error.
vm_built_raw=""
[[ "$type" == "vm" ]] && vm_built_raw="${remote_repo_dir}/${host}.raw"
# --- refuse to duplicate a host that's already live on the node --------- # --- refuse to duplicate a host that's already live on the node ---------
# Queries the node itself (qm/pct's own name/hostname config), not any # Queries the node itself (qm/pct's own name/hostname config), not any
# static list in this repo -- a file can't track whether a resource still # static list in this repo -- a file can't track whether a resource still
@@ -390,15 +359,14 @@ else
echo echo
echo "==> Checking ${node} for an existing VM/CT identified as '${host}'..." echo "==> Checking ${node} for an existing VM/CT identified as '${host}'..."
ssh_check_status=0 ssh_check_status=0
existing="$(ssh "$ssh_target" bash -s -- "$host" "$sudo_prefix" <<'REMOTE_SCRIPT' existing="$(ssh "$ssh_target" bash -s -- "$host" <<'REMOTE_SCRIPT'
target="$1" target="$1"
sudo_pfx="$2" for id in $(qm list 2>/dev/null | awk 'NR>1{print $1}'); do
for id in $($sudo_pfx qm list 2>/dev/null | awk 'NR>1{print $1}'); do n="$(qm config "$id" 2>/dev/null | grep -oP '^name:\s*\K\S+' || true)"
n="$($sudo_pfx qm config "$id" 2>/dev/null | grep -oP '^name:\s*\K\S+' || true)"
[[ "$n" == "$target" ]] && echo "vm ${id} ${n}" [[ "$n" == "$target" ]] && echo "vm ${id} ${n}"
done done
for id in $($sudo_pfx pct list 2>/dev/null | awk 'NR>1{print $1}'); do for id in $(pct list 2>/dev/null | awk 'NR>1{print $1}'); do
n="$($sudo_pfx pct config "$id" 2>/dev/null | grep -oP '^hostname:\s*\K\S+' || true)" n="$(pct config "$id" 2>/dev/null | grep -oP '^hostname:\s*\K\S+' || true)"
[[ "$n" == "$target" ]] && echo "lxc ${id} ${n}" [[ "$n" == "$target" ]] && echo "lxc ${id} ${n}"
done done
exit 0 exit 0
@@ -485,12 +453,12 @@ REMOTE_SCRIPT
if [[ "$kind" == "vm" ]]; then if [[ "$kind" == "vm" ]]; then
# qm destroy has no --force to stop-then-destroy in one call (pct's # qm destroy has no --force to stop-then-destroy in one call (pct's
# does) -- stop explicitly first if it's running. # does) -- stop explicitly first if it's running.
if ssh "$ssh_target" "${sudo_prefix} qm status ${id}" 2>/dev/null | grep -q running; then if ssh "$ssh_target" "qm status ${id}" 2>/dev/null | grep -q running; then
ssh "$ssh_target" "${sudo_prefix} qm stop ${id}" ssh "$ssh_target" "qm stop ${id}"
fi fi
ssh "$ssh_target" "${sudo_prefix} qm destroy ${id} --purge 1" ssh "$ssh_target" "qm destroy ${id} --purge 1"
else else
ssh "$ssh_target" "${sudo_prefix} pct destroy ${id} --force 1 --purge 1" ssh "$ssh_target" "pct destroy ${id} --force 1 --purge 1"
fi fi
done done
fi fi
@@ -506,37 +474,12 @@ echo "Target: ${flake_target} (host=${host}, type=${type}) -> Proxmox resource '
nix_extra_opts nix_extra_opts
# --- make sure this target has a registered host key -------------------- # --- make sure this target has a registered host key --------------------
# sync-host-keys.sh is idempotent and generates the key (via clan vars) if
# no key exists yet -- the old inline prepare-host-key.sh call is gone.
echo echo
echo "==> Ensuring host key exists and is registered..." echo "==> Ensuring host key exists and is registered..."
sync_args=("$flake_target") sync_args=("$flake_target")
[[ "$dry_run" -eq 1 ]] && sync_args+=(--dry-run) [[ "$dry_run" -eq 1 ]] && sync_args+=(--dry-run)
bash "$sync_keys" "${sync_args[@]}" bash "$sync_keys" "${sync_args[@]}"
# If sync-host-keys.sh changed .sops.yaml, secrets/, or vars/per-machine/,
# those changes must be committed and pushed before the remote `git pull`
# below picks them up -- the PVE node builds from whatever HEAD is checked
# out there, not the local working tree. Uncommitted clan vars or sops
# recipients mean the image builds fine but the host cannot decrypt its
# secrets on first boot. Block until the operator confirms they've pushed.
if [[ "$dry_run" -eq 0 ]]; then
_dirty="$(git -C "$repo_root" status --porcelain -- .sops.yaml secrets/ vars/per-machine/ 2>/dev/null || true)"
if [[ -n "$_dirty" ]]; then
echo
echo "==> COMMIT + PUSH REQUIRED before the remote build can succeed:"
echo " Uncommitted changes in .sops.yaml, secrets/, or vars/per-machine/."
echo " The PVE node builds from the git-tracked flake, so these changes"
echo " must be committed and pushed first -- otherwise the image build will"
echo " succeed but the host cannot decrypt its secrets on first boot."
echo
git -C "$repo_root" status --short -- .sops.yaml secrets/ vars/per-machine/ || true
echo
read -rp " Commit and push those changes, then press Enter to continue (Ctrl-C to abort): "
fi
unset _dirty
fi
# --- VMID: pick one, and refuse to touch anything that already exists --- # --- VMID: pick one, and refuse to touch anything that already exists ---
echo echo
if [[ -z "$vmid" ]]; then if [[ -z "$vmid" ]]; then
@@ -544,7 +487,7 @@ if [[ -z "$vmid" ]]; then
vmid="<next-free-vmid>" vmid="<next-free-vmid>"
echo "[dry-run] would ask ${node} for the next free VMID (pvesh get /cluster/nextid)" echo "[dry-run] would ask ${node} for the next free VMID (pvesh get /cluster/nextid)"
else else
vmid="$(ssh "$ssh_target" "${sudo_prefix} pvesh get /cluster/nextid" | tr -d '[:space:]')" vmid="$(ssh "$ssh_target" "pvesh get /cluster/nextid" | tr -d '[:space:]')"
echo "Auto-assigned VMID: ${vmid}" echo "Auto-assigned VMID: ${vmid}"
fi fi
else else
@@ -558,8 +501,8 @@ if [[ "$dry_run" -eq 0 ]]; then
# both. Any success here means something is already using this ID -- # both. Any success here means something is already using this ID --
# refuse to go anywhere near it. (Reconfiguring an existing resource is # refuse to go anywhere near it. (Reconfiguring an existing resource is
# --modify's job, not this one's.) # --modify's job, not this one's.)
if ssh "$ssh_target" "${sudo_prefix} qm status ${vmid}" >/dev/null 2>&1 \ if ssh "$ssh_target" "qm status ${vmid}" >/dev/null 2>&1 \
|| ssh "$ssh_target" "${sudo_prefix} pct status ${vmid}" >/dev/null 2>&1; then || ssh "$ssh_target" "pct status ${vmid}" >/dev/null 2>&1; then
echo "ERROR: VMID ${vmid} already exists on ${node}. Refusing to touch an" >&2 echo "ERROR: VMID ${vmid} already exists on ${node}. Refusing to touch an" >&2
echo "existing resource here -- use --modify to reconfigure it, pick a" >&2 echo "existing resource here -- use --modify to reconfigure it, pick a" >&2
echo "different --vmid, or omit it to auto-assign." >&2 echo "different --vmid, or omit it to auto-assign." >&2
@@ -659,35 +602,21 @@ ensure_remote_repo() {
fi fi
} }
# --- sync host key to the node --------------------------------------------- # --- sync locally-managed host-keys/ to the node ---------------------------
# SSH host keys are stored as clan vars (vars/per-machine/<target>/openssh/). # Gitignored (see .gitignore), so `git pull` above never carries it -- both
# Decrypt locally and scp just the two files for this target to the node's # build paths need it present as NIXOS_HOST_KEYS_DIR / --pre-format-files
# host-keys/ directory, where the remote build script picks them up via # input on the node itself now that the build runs there. scp (not rsync,
# NIXOS_HOST_KEYS_DIR (LXC) or --pre-format-files (VM). A target with no # not already a dependency anywhere else in this repo) mirrors how this
# clan var is an error -- generate one first with sync-host-keys.sh. # script already transfers the --image case below.
sync_remote_host_keys() { sync_remote_host_keys() {
echo echo
echo "==> Syncing host key for ${flake_target} to ${node}..." echo "==> Syncing host-keys/ to ${node}..."
if [[ "$dry_run" -eq 1 ]]; then if [[ "$dry_run" -eq 1 ]]; then
echo "[dry-run] would decrypt clan SSH key for ${flake_target} and copy to ${ssh_target}:${remote_repo_dir}/host-keys/" echo "[dry-run] would copy ${repo_root}/host-keys/ to ${ssh_target}:${remote_repo_dir}/host-keys/"
return return
fi fi
if ! clan_ssh_key_exists "$flake_target" "$repo_root"; then
echo "ERROR: no clan SSH key found for ${flake_target}" >&2
echo " (expected: ${repo_root}/vars/per-machine/${flake_target}/openssh/ssh_host_ed25519_key/secret)" >&2
echo " Generate one first: bash scripts/secrets/sync-host-keys.sh ${flake_target}" >&2
exit 1
fi
local tmpdir
tmpdir="$(mktemp -d)"
# shellcheck disable=SC2064
trap "rm -rf '${tmpdir}'" RETURN
echo " Decrypting clan SSH key for ${flake_target}..."
clan_decrypt_ssh_key "$flake_target" "$repo_root" "$tmpdir"
ssh "$ssh_target" "mkdir -p '${remote_repo_dir}/host-keys'" ssh "$ssh_target" "mkdir -p '${remote_repo_dir}/host-keys'"
scp -p "${tmpdir}/${flake_target}_ssh_host_ed25519_key" \ scp -pr "${repo_root}/host-keys/." "${ssh_target}:${remote_repo_dir}/host-keys/"
"${tmpdir}/${flake_target}_ssh_host_ed25519_key.pub" \
"${ssh_target}:${remote_repo_dir}/host-keys/"
} }
# --- build (or reuse an image already on the node) ------------------------ # --- build (or reuse an image already on the node) ------------------------
@@ -702,14 +631,10 @@ if [[ -n "$image" ]]; then
elif [[ "$force_rebuild" -eq 1 ]]; then elif [[ "$force_rebuild" -eq 1 ]]; then
echo "--force-rebuild: skipping the existing-image check on ${node}." echo "--force-rebuild: skipping the existing-image check on ${node}."
else else
# VMs: check for the raw image in the remote repo dir (where disko writes it). echo "==> Checking whether ${node} already has ${remote_path}..."
# LXC: check for the tarball in iso_storage (where the LXC build stages it).
_check_path="$remote_path"
[[ "$type" == "vm" ]] && _check_path="$vm_built_raw"
echo "==> Checking whether ${node} already has ${_check_path}..."
if [[ "$dry_run" -eq 1 ]]; then if [[ "$dry_run" -eq 1 ]]; then
echo "[dry-run] would check: ssh ${ssh_target} -- test -f ${_check_path}" echo "[dry-run] would check: ssh ${ssh_target} -- test -f ${remote_path}"
elif ssh "$ssh_target" "test -f '${_check_path}'" 2>/dev/null; then elif ssh "$ssh_target" "test -f '${remote_path}'" 2>/dev/null; then
echo "Found it -- reusing, skipping build (use --force-rebuild to override)." echo "Found it -- reusing, skipping build (use --force-rebuild to override)."
image_already_remote=1 image_already_remote=1
else else
@@ -747,11 +672,11 @@ if [[ "$image_already_remote" -eq 0 && -z "$local_image" ]]; then
# hands the result to the remote shell to re-split, which would # hands the result to the remote shell to re-split, which would
# otherwise scatter NIX_EXTRA_OPTS (itself several space-separated, # otherwise scatter NIX_EXTRA_OPTS (itself several space-separated,
# %q-quoted tokens) across the wrong positional parameters below. # %q-quoted tokens) across the wrong positional parameters below.
printf -v remote_cmd 'bash -s -- %q %q %q %q %q %q' \ printf -v remote_cmd 'bash -s -- %q %q %q %q %q' \
"$remote_repo_dir" "$flake_target" "$remote_dir" "$remote_filename" "$NIX_EXTRA_OPTS" "$sudo_prefix" "$remote_repo_dir" "$flake_target" "$remote_dir" "$remote_filename" "$NIX_EXTRA_OPTS"
ssh "$ssh_target" "$remote_cmd" <<'REMOTE_SCRIPT' ssh "$ssh_target" "$remote_cmd" <<'REMOTE_SCRIPT'
set -euo pipefail set -euo pipefail
repo_dir="$1"; target="$2"; dest_dir="$3"; dest_name="$4"; nix_extra_opts_str="$5"; sudo_pfx="$6" repo_dir="$1"; target="$2"; dest_dir="$3"; dest_name="$4"; nix_extra_opts_str="$5"
declare -a NIX_OPTS=() declare -a NIX_OPTS=()
[[ -n "$nix_extra_opts_str" ]] && eval "NIX_OPTS=(${nix_extra_opts_str})" [[ -n "$nix_extra_opts_str" ]] && eval "NIX_OPTS=(${nix_extra_opts_str})"
cd "$repo_dir" cd "$repo_dir"
@@ -760,12 +685,6 @@ cd "$repo_dir"
# right after a successful install. # right after a successful install.
. scripts/lib/nix-bootstrap.sh . scripts/lib/nix-bootstrap.sh
ensure_nix_profile ensure_nix_profile
if [[ ! -f "host-keys/${target}_ssh_host_ed25519_key" ]]; then
echo "ERROR: host-keys/${target}_ssh_host_ed25519_key not found in ${repo_dir}." >&2
echo "Generate the key locally (scripts/secrets/sync-host-keys.sh ${target})" >&2
echo "and ensure it was synced here before starting the build." >&2
exit 1
fi
NIXOS_HOST_KEYS_DIR="$(pwd)/host-keys" nix build --impure \ NIXOS_HOST_KEYS_DIR="$(pwd)/host-keys" nix build --impure \
--no-use-registries --no-accept-flake-config "${NIX_OPTS[@]}" \ --no-use-registries --no-accept-flake-config "${NIX_OPTS[@]}" \
".#nixosConfigurations.${target}.config.system.build.tarball" \ ".#nixosConfigurations.${target}.config.system.build.tarball" \
@@ -775,69 +694,64 @@ if [[ -z "$built" ]]; then
echo "ERROR: no tarball found under result-${target}/tarball after build." >&2 echo "ERROR: no tarball found under result-${target}/tarball after build." >&2
exit 1 exit 1
fi fi
$sudo_pfx mkdir -p "$dest_dir" mkdir -p "$dest_dir"
$sudo_pfx cp "$built" "${dest_dir}/${dest_name}" cp "$built" "${dest_dir}/${dest_name}"
echo "Built and staged: ${dest_dir}/${dest_name}" echo "Built and staged: ${dest_dir}/${dest_name}"
REMOTE_SCRIPT REMOTE_SCRIPT
local_image="$remote_path" local_image="$remote_path"
echo "Built on ${node}: ${remote_path}" echo "Built on ${node}: ${remote_path}"
fi fi
else else
# PROXMOX_SSH_USER defaults to root (env.sh), which needs no sudo and
# can't assume it's even installed on a minimal node -- only shell out
# through sudo when actually running as a non-root SSH user.
sudo_prefix="sudo"
sudo_display="sudo "
if [[ "$PROXMOX_SSH_USER" == "root" ]]; then
sudo_prefix=""
sudo_display=""
fi
if [[ "$dry_run" -eq 1 ]]; then if [[ "$dry_run" -eq 1 ]]; then
echo "[dry-run] would build on ${node}: NIXOS_HOST_KEYS_DIR=\$(pwd)/host-keys nix build --impure --no-use-registries --no-accept-flake-config${nix_opts_display} \\" echo "[dry-run] would build on ${node}: nix build --no-use-registries --no-accept-flake-config${nix_opts_display} \\"
echo "[dry-run] .#nixosConfigurations.${flake_target}.config.system.build.diskoImagesScript" echo "[dry-run] .#nixosConfigurations.${flake_target}.config.system.build.diskoImagesScript"
echo "[dry-run] would run: ${sudo_display}./result-${flake_target} --build-memory 2048" echo "[dry-run] would run: ${sudo_display}./result-${flake_target} \\"
echo "[dry-run] image will be at ${vm_built_raw} (imported from there; no mv to /var/lib/vz/import/)" echo "[dry-run] --pre-format-files host-keys/${flake_target}_ssh_host_ed25519_key /etc/ssh/ssh_host_ed25519_key \\"
echo "[dry-run] --pre-format-files host-keys/${flake_target}_ssh_host_ed25519_key.pub /etc/ssh/ssh_host_ed25519_key.pub \\"
echo "[dry-run] --build-memory 2048"
echo "[dry-run] would stage the result at ${remote_path}"
local_image="<built-image>.raw" local_image="<built-image>.raw"
else else
echo "==> Building Disko image for ${flake_target} on ${node}..." echo "==> Building Disko image for ${flake_target} on ${node}..."
# See the LXC branch above for why this is one %q-quoted command # See the LXC branch above for why this is one %q-quoted command
# string rather than separate ssh argv elements. # string rather than separate ssh argv elements.
# $7 = image_name (hostname, the diskoImagesScript's own output filename). printf -v remote_cmd 'bash -s -- %q %q %q %q %q %q' \
# "$remote_repo_dir" "$flake_target" "$remote_dir" "$remote_filename" "$NIX_EXTRA_OPTS" "$sudo_prefix"
# NIXOS_HOST_KEYS_DIR + --impure: modules/platforms/proxmox.nix reads
# this env var at eval time (like lxc.nix) to embed the clan SSH host
# key in environment.etc. nixos-install's own activation then places the
# key on the target disk, so sshd-keygen finds it already present and
# skips generation. --pre-format-files put the key on the QEMU builder
# VM's rootfs (not the target disk), so sshd-keygen regenerated a fresh
# key -- one not registered in .sops.yaml -- and sops could never decrypt.
printf -v remote_cmd 'bash -s -- %q %q %q %q %q %q %q' \
"$remote_repo_dir" "$flake_target" "$remote_dir" "$remote_filename" "$NIX_EXTRA_OPTS" "$sudo_prefix" "$host"
ssh "$ssh_target" "$remote_cmd" <<'REMOTE_SCRIPT' ssh "$ssh_target" "$remote_cmd" <<'REMOTE_SCRIPT'
set -euo pipefail set -euo pipefail
repo_dir="$1"; target="$2"; dest_dir="$3"; dest_name="$4"; nix_extra_opts_str="$5"; sudo_pfx="$6"; image_name="$7" repo_dir="$1"; target="$2"; dest_dir="$3"; dest_name="$4"; nix_extra_opts_str="$5"; sudo_prefix="$6"
declare -a NIX_OPTS=() declare -a NIX_OPTS=()
[[ -n "$nix_extra_opts_str" ]] && eval "NIX_OPTS=(${nix_extra_opts_str})" [[ -n "$nix_extra_opts_str" ]] && eval "NIX_OPTS=(${nix_extra_opts_str})"
cd "$repo_dir" cd "$repo_dir"
. scripts/lib/nix-bootstrap.sh . scripts/lib/nix-bootstrap.sh
ensure_nix_profile ensure_nix_profile
if [[ ! -f "host-keys/${target}_ssh_host_ed25519_key" ]]; then nix build --no-use-registries --no-accept-flake-config "${NIX_OPTS[@]}" \
echo "ERROR: host-keys/${target}_ssh_host_ed25519_key not found in ${repo_dir}." >&2
echo "Generate the key locally (scripts/secrets/sync-host-keys.sh ${target})" >&2
echo "and ensure it was synced here before starting the build." >&2
exit 1
fi
# Build diskoImagesScript with NIXOS_HOST_KEYS_DIR so proxmox.nix embeds the
# clan SSH key in environment.etc (same as lxc.nix). This causes nixos-install
# to place the key on the target disk, so sshd-keygen finds it and skips
# generation -- the disk image boots with the registered key, sops decrypts.
NIXOS_HOST_KEYS_DIR="$(pwd)/host-keys" nix build --impure \
--no-use-registries --no-accept-flake-config "${NIX_OPTS[@]}" \
".#nixosConfigurations.${target}.config.system.build.diskoImagesScript" \ ".#nixosConfigurations.${target}.config.system.build.diskoImagesScript" \
--out-link "result-${target}" --out-link "result-${target}"
# Remove any stale .raw from a previous failed build so the post-build check $sudo_prefix "./result-${target}" \
# below is unambiguous (diskoImagesScript writes to CWD as ${image_name}.raw). --pre-format-files "host-keys/${target}_ssh_host_ed25519_key" /etc/ssh/ssh_host_ed25519_key \
$sudo_pfx rm -f "${image_name}.raw" 2>/dev/null || true --pre-format-files "host-keys/${target}_ssh_host_ed25519_key.pub" /etc/ssh/ssh_host_ed25519_key.pub \
$sudo_pfx "./result-${target}" --build-memory 2048 --build-memory 2048
if [[ ! -f "${image_name}.raw" ]]; then built="$(find . -maxdepth 1 -name '*.raw' -newer "result-${target}" | head -1)"
echo "ERROR: ${image_name}.raw not found in ${repo_dir} after build -- disko/QEMU may have failed." >&2 if [[ -z "$built" ]]; then
echo "ERROR: no .raw image found in ${repo_dir} after build." >&2
exit 1 exit 1
fi fi
echo "Built image: ${repo_dir}/${image_name}.raw" mkdir -p "$dest_dir"
mv "$built" "${dest_dir}/${dest_name}"
echo "Built and staged: ${dest_dir}/${dest_name}"
REMOTE_SCRIPT REMOTE_SCRIPT
local_image="$vm_built_raw" local_image="$remote_path"
echo "Built on ${node}: ${vm_built_raw}" echo "Built on ${node}: ${remote_path}"
fi fi
fi fi
fi fi
@@ -867,17 +781,17 @@ if [[ "$type" == "lxc" ]]; then
local_swap="${swap:-$memory}" local_swap="${swap:-$memory}"
# --unprivileged: read back from modules/platforms/lxc.nix's own # --unprivileged: read back from modules/platforms/lxc.nix's own
# proxmoxLXC.privileged (via flake_target_lxc_privileged) rather than # proxmoxLXC.privileged (via flake_target_lxc_privileged) rather than
# hardcoded. lxc.nix derives this automatically: any lxc-* host whose # hardcoded, since that's no longer the same for every lxc-* target --
# config.fileSystems has an NFS entry gets privileged=true, because the # lxc-docker sets it true so the container's NFS mounts work at all (the
# kernel's NFS client (FS_USERNS_MOUNT not set) rejects NFS mounts from # kernel's NFS client can't mount from inside any unprivileged
# inside any non-init user namespace -- exactly what an unprivileged # container's user namespace, no matter what AppArmor allows -- see that
# container's UID-mapped root lives in -- with EPERM at the VFS layer, # option's own comment). The NixOS config inside the image bakes in
# regardless of AppArmor (see lxc.nix's own comment). The NixOS config # cgroup/capability/mount expectations matching whichever value it was
# bakes in cgroup/capability/mount expectations matching whichever value # built with, so this must stay in sync with it -- `pct create`'s own
# it was built with, so this must stay in sync -- `pct create`'s CLI # CLI default for this flag is privileged (unlike the web UI, which
# default is privileged (unlike the web UI, which defaults the other # defaults its checkbox the other way), so leaving it unset would create
# way), so leaving it unset would create a privileged container running # a privileged container running a NixOS config that assumes
# a NixOS config that assumes unprivileged, a real mismatch. # unprivileged for every target except lxc-docker, a real mismatch.
privileged_eval="$(flake_target_lxc_privileged "$repo_root" "$flake_target")" privileged_eval="$(flake_target_lxc_privileged "$repo_root" "$flake_target")"
unprivileged_flag=1 unprivileged_flag=1
[[ "$privileged_eval" == "true" ]] && unprivileged_flag=0 [[ "$privileged_eval" == "true" ]] && unprivileged_flag=0
@@ -898,9 +812,9 @@ if [[ "$type" == "lxc" ]]; then
# hands the whole string to `ssh` as a single command for the *remote* # hands the whole string to `ssh` as a single command for the *remote*
# shell to parse -- unquoted, that `;` would be read as a remote # shell to parse -- unquoted, that `;` would be read as a remote
# command separator and silently truncate this into two commands. # command separator and silently truncate this into two commands.
create_cmd="${sudo_prefix} pct create ${vmid} ${iso_storage}:vztmpl/${remote_filename} --unprivileged ${unprivileged_flag} --features '${PROXMOX_DEFAULT_LXC_FEATURES}' --rootfs ${storage}:${local_disk_size} --hostname ${name} --cores ${cores} --memory ${memory} --swap ${local_swap} --net0 name=eth0,bridge=${bridge},ip=dhcp" create_cmd="pct create ${vmid} ${iso_storage}:vztmpl/${remote_filename} --unprivileged ${unprivileged_flag} --features '${PROXMOX_DEFAULT_LXC_FEATURES}' --rootfs ${storage}:${local_disk_size} --hostname ${name} --cores ${cores} --memory ${memory} --swap ${local_swap} --net0 name=eth0,bridge=${bridge},ip=dhcp"
remote "$create_cmd" remote "$create_cmd"
remote "${sudo_prefix} pct start ${vmid}" remote "pct start ${vmid}"
else else
echo "==> Creating VM ${vmid} (${name})..." echo "==> Creating VM ${vmid} (${name})..."
# pre-enrolled-keys=0 disables OVMF's Secure Boot key pre-enrollment -- # pre-enrolled-keys=0 disables OVMF's Secure Boot key pre-enrollment --
@@ -911,56 +825,29 @@ else
# without this flag Proxmox never creates the channel it listens on, so # without this flag Proxmox never creates the channel it listens on, so
# `qm guest exec`/`qm agent` and the UI's IP-address display silently # `qm guest exec`/`qm agent` and the UI's IP-address display silently
# never work for any VM this script creates. # never work for any VM this script creates.
remote "${sudo_prefix} qm create ${vmid} --name ${name} --memory ${memory} --cores ${cores} \ remote "qm create ${vmid} --name ${name} --memory ${memory} --cores ${cores} \
--net0 virtio,bridge=${bridge} --bios ovmf --machine q35 --scsihw virtio-scsi-pci \ --net0 virtio,bridge=${bridge} --bios ovmf --machine q35 --scsihw virtio-scsi-pci \
--efidisk0 ${storage}:1,efitype=4m,pre-enrolled-keys=0 --agent enabled=1" --efidisk0 ${storage}:1,efitype=4m,pre-enrolled-keys=0 --agent enabled=1"
# VMs built on the node: import from the repo dir (where disko/QEMU wrote it).
# VMs from --image: import from remote_path (where scp uploaded it).
_import_path="${remote_path}"
[[ -z "$image" ]] && _import_path="${vm_built_raw}"
if [[ "$dry_run" -eq 1 ]]; then if [[ "$dry_run" -eq 1 ]]; then
echo "[dry-run] ssh ${ssh_target} -- ${sudo_display}qm importdisk ${vmid} ${_import_path} ${storage}" echo "[dry-run] ssh ${ssh_target} -- qm importdisk ${vmid} ${remote_path} ${storage}"
echo "[dry-run] (would parse the resulting disk identifier from that output)" echo "[dry-run] (would parse the resulting disk identifier from that output)"
echo "[dry-run] ssh ${ssh_target} -- ${sudo_display}qm set ${vmid} --scsi0 ${storage}:<parsed-disk-id>" echo "[dry-run] ssh ${ssh_target} -- qm set ${vmid} --scsi0 ${storage}:<parsed-disk-id>"
else else
if ! importdisk_output="$(ssh "$ssh_target" "${sudo_prefix} qm importdisk ${vmid} ${_import_path} ${storage}" 2>&1)"; then importdisk_output="$(ssh "$ssh_target" "qm importdisk ${vmid} ${remote_path} ${storage}")"
echo "ERROR: qm importdisk failed:" >&2
echo "${importdisk_output}" >&2
exit 1
fi
echo "$importdisk_output" echo "$importdisk_output"
# PVE output format: "unusedN: successfully imported disk '<storage>:<vol>'" disk_id="$(echo "$importdisk_output" | grep -oP "(?<=Successfully imported disk as ')[^']+" | sed 's/^unused[0-9]*://')"
# (lowercase "successfully", no "as"; the primary regex targets this form; the
# || true inside the substitution prevents set -e from aborting when grep finds
# no match -- without it the script would silently exit before reaching the
# fallback whenever the PVE format doesn't match).
disk_id="$(echo "$importdisk_output" | grep -oP "successfully imported disk '\\K[^']+" || true)"
if [[ -z "$disk_id" ]]; then
# Fallback for other PVE output variants: read qm config directly.
unused_line="$(ssh "$ssh_target" "${sudo_prefix} qm config ${vmid}" | grep '^unused[0-9]*:' | head -1 || true)"
if [[ -n "$unused_line" ]]; then
disk_id="${unused_line#*: }"
echo "Note: disk ID resolved from qm config: ${disk_id}"
fi
fi
if [[ -z "$disk_id" ]]; then if [[ -z "$disk_id" ]]; then
echo "ERROR: couldn't parse the imported disk identifier from qm importdisk's output above." >&2 echo "ERROR: couldn't parse the imported disk identifier from qm importdisk's output above." >&2
echo "The VM shell (${vmid}) and imported disk both exist -- finish attaching it by hand:" >&2 echo "The VM shell (${vmid}) and imported disk both exist -- finish attaching it by hand:" >&2
echo " ssh ${ssh_target} -- ${sudo_display}qm set ${vmid} --scsi0 ${storage}:<disk-id-from-output-above>" >&2 echo " ssh ${ssh_target} -- qm set ${vmid} --scsi0 ${storage}:<disk-id-from-output-above>" >&2
echo " ssh ${ssh_target} -- ${sudo_display}qm set ${vmid} --boot order=scsi0" >&2 echo " ssh ${ssh_target} -- qm set ${vmid} --boot order=scsi0" >&2
exit 1 exit 1
fi fi
remote "${sudo_prefix} qm set ${vmid} --scsi0 ${disk_id}" remote "qm set ${vmid} --scsi0 ${disk_id}"
# The disk data is now in ZFS; remove the source raw file (only for images
# we built on the node -- --image uploads are the operator's to manage).
if [[ -z "$image" ]]; then
ssh "$ssh_target" "${sudo_prefix} rm -f '${_import_path}'" 2>/dev/null || \
echo "Warning: couldn't remove ${_import_path} from ${node} -- you can delete it manually" >&2
fi
fi fi
remote "${sudo_prefix} qm set ${vmid} --boot order=scsi0" remote "qm set ${vmid} --boot order=scsi0"
remote "${sudo_prefix} qm start ${vmid}" remote "qm start ${vmid}"
fi fi
echo echo
-253
View File
@@ -1,253 +0,0 @@
#!/usr/bin/env bash
# recover-hosts.sh — Fix sops/SSH-key/GitHub-token issues on deployed NixOS hosts
# and trigger a Switch-nix rebuild on each.
#
# Run from the repo root on the workstation (nixos@nixos):
# bash scripts/recover-hosts.sh [<hostname> ...]
#
# With no args it discovers and checks every known hostname.
# With args it checks only those hostnames:
# bash scripts/recover-hosts.sh tor-relay
#
# Fixes applied automatically (then prompts before rebuilding):
# 1. SSH host key drift — live key no longer matches host-keys/<target>_ssh_host_ed25519_key
# Fix: scp the registered key back and restore it (needs sudo once per host).
# To push new keys proactively (before drift, e.g. right after
# sync-host-keys.sh --regenerate-all-keys), use instead:
# scripts/secrets/push-host-keys.sh --all
# 2. Stale/invalid GitHub access token — the rendered nix-github-token.conf has
# a token GitHub rejects (401), blocking any rebuild that fetches disko or
# other public GitHub flake inputs.
# Fix: empty the rendered file so nix makes unauthenticated requests instead.
# Public repos (disko, nixpkgs, etc.) work fine without auth. sops-nix
# re-renders the correct new token automatically after the first successful
# rebuild.
#
# Both fixes need one interactive sudo session per host. The script opens a
# single ssh -t per broken host so you enter the password once and all steps
# run in sequence.
set -euo pipefail
cd "$(dirname "$0")/.."
source scripts/env.sh 2>/dev/null || true
SSH_OPTS=(-o StrictHostKeyChecking=no -o BatchMode=yes -o ConnectTimeout=5)
SSH_USER=nixos
# Known flake-target → ssh hostname map for all currently-defined hosts.
# Add new hosts here as they are deployed.
declare -A TARGET_HOST=(
[lxc-docker]=docker
[lxc-nix-cache]=nix-cache
[lxc-pxe-boot]=pxe-boot
[lxc-tor-relay]=tor-relay
[lxc-minimal]=nix-minimal
[proxmox-server]=server
[baremetal-gui]=nixos
)
# ── helpers ───────────────────────────────────────────────────────────────────
info() { echo " [✓] $*"; }
warn() { echo " [!] $*"; }
step() { echo "==> $*"; }
ssh_host_age() {
ssh-keyscan -t ed25519 "$1" 2>/dev/null \
| nix shell nixpkgs#ssh-to-age --command ssh-to-age 2>/dev/null \
| head -1 || true
}
registered_age() {
local keyfile="host-keys/${1}_ssh_host_ed25519_key.pub"
[ -f "$keyfile" ] || return 0
nix shell nixpkgs#ssh-to-age --command ssh-to-age < "$keyfile" 2>/dev/null \
| head -1 || true
}
github_token_valid() {
local host=$1
local raw token code
raw=$(ssh "${SSH_OPTS[@]}" "$SSH_USER@$host" \
"cat /run/secrets/rendered/nix-github-token.conf 2>/dev/null || true")
token=$(echo "$raw" | grep -oP '(?<=github\.com=)\S+' || true)
if [ -z "$token" ]; then
return 0 # no token = unauthenticated, works for public repos
fi
code=$(curl -s -o /dev/null -w "%{http_code}" \
-H "Authorization: token $token" \
"https://api.github.com/repos/nix-community/disko" 2>/dev/null || echo 000)
[ "$code" = "200" ]
}
# ── discover hosts ────────────────────────────────────────────────────────────
if [ $# -gt 0 ]; then
HOSTNAMES=("$@")
else
HOSTNAMES=()
seen=()
for target in "${!TARGET_HOST[@]}"; do
h="${TARGET_HOST[$target]}"
# deduplicate (e.g. proxmox-server and lxc-server both map to "server")
if [[ ! " ${seen[*]:-} " =~ " $h " ]]; then
seen+=("$h")
if ssh "${SSH_OPTS[@]}" "$SSH_USER@$h" "true" 2>/dev/null; then
HOSTNAMES+=("$h")
fi
fi
done
fi
if [ ${#HOSTNAMES[@]} -eq 0 ]; then
echo "No reachable hosts found. Pass hostnames explicitly or check SSH."
exit 1
fi
echo ""
echo "Hosts to check: ${HOSTNAMES[*]}"
echo ""
# ── check phase ───────────────────────────────────────────────────────────────
NEEDS_FIX=()
for host in "${HOSTNAMES[@]}"; do
step "$host"
if ! ssh "${SSH_OPTS[@]}" "$SSH_USER@$host" "true" 2>/dev/null; then
warn "SSH unreachable — clearing stale known_hosts entry"
ssh-keygen -R "$host" 2>/dev/null || true
continue
fi
flake_target=$(ssh "${SSH_OPTS[@]}" "$SSH_USER@$host" \
"cat /etc/flake-target 2>/dev/null || true")
echo " flake-target: ${flake_target:-unknown}"
host_broken=false
# SSH host key
if [ -n "$flake_target" ] && [ -f "host-keys/${flake_target}_ssh_host_ed25519_key.pub" ]; then
live=$(ssh_host_age "$host")
want=$(registered_age "$flake_target")
if [ "$live" = "$want" ]; then
info "SSH host key OK"
else
warn "SSH host key MISMATCH (live ≠ host-keys/) -- use push-host-keys.sh proactively next time"
echo " live: $live"
echo " registered: $want"
host_broken=true
fi
else
echo " [~] No host-keys/ entry for ${flake_target:-unknown} — skipping key check"
fi
# GitHub token
if github_token_valid "$host"; then
info "GitHub token OK"
else
warn "GitHub token invalid (rebuild will fail with 401)"
host_broken=true
fi
# sops-nix result
sops_result=$(ssh "${SSH_OPTS[@]}" "$SSH_USER@$host" \
"systemctl show sops-nix --property=Result --value 2>/dev/null || echo unknown")
if [ "$sops_result" = "success" ]; then
info "sops-nix: success"
else
warn "sops-nix: $sops_result"
fi
$host_broken && NEEDS_FIX+=("$host")
echo ""
done
# ── fix phase ─────────────────────────────────────────────────────────────────
if [ ${#NEEDS_FIX[@]} -eq 0 ]; then
echo "All hosts healthy — nothing to fix."
exit 0
fi
echo "Hosts needing fixes: ${NEEDS_FIX[*]}"
echo ""
echo "Each fix requires one sudo session per host. You will be prompted for"
echo "the nixos sudo password once per host; all steps run in that session."
echo ""
read -r -p "Proceed with fixes + Switch-nix on each broken host? [y/N] " confirm
[[ "$confirm" =~ ^[Yy]$ ]] || { echo "Aborted."; exit 0; }
echo ""
for host in "${NEEDS_FIX[@]}"; do
step "Fixing $host"
flake_target=$(ssh "${SSH_OPTS[@]}" "$SSH_USER@$host" \
"cat /etc/flake-target 2>/dev/null || true")
fix_script=""
# Fix 1: restore SSH host key
live=$(ssh_host_age "$host")
want=$(registered_age "${flake_target:-}")
if [ -n "$want" ] && [ "$live" != "$want" ]; then
echo " Uploading registered SSH host key (private + public)..."
scp -o StrictHostKeyChecking=no \
"host-keys/${flake_target}_ssh_host_ed25519_key" \
"$SSH_USER@$host:/tmp/recover_ed25519_key"
scp -o StrictHostKeyChecking=no \
"host-keys/${flake_target}_ssh_host_ed25519_key.pub" \
"$SSH_USER@$host:/tmp/recover_ed25519_key.pub"
fix_script+='
echo "[fix] Restoring SSH host key..."
install -m 0600 /tmp/recover_ed25519_key /etc/ssh/ssh_host_ed25519_key
install -m 0644 /tmp/recover_ed25519_key.pub /etc/ssh/ssh_host_ed25519_key.pub
rm -f /tmp/recover_ed25519_key /tmp/recover_ed25519_key.pub
echo " Done."
'
ssh-keygen -R "$host" 2>/dev/null || true
fi
# Fix 2: clear invalid GitHub token
if ! github_token_valid "$host"; then
fix_script+='
echo "[fix] Clearing stale GitHub token (nix will use unauthenticated access)..."
echo "" > /run/secrets/rendered/nix-github-token.conf
systemctl restart nix-daemon 2>/dev/null || true
echo " Done."
'
fi
# Fix 3: rebuild
fix_script+='
echo "[fix] Running nixos-rebuild switch..."
nixos-rebuild switch \
--no-write-lock-file \
--refresh \
--flake "git+https://gitea.lan.ddnsgeek.com/beatzaplenty/nixos.git#$(cat /etc/flake-target)"
echo "[fix] Rebuild complete."
'
echo " Opening SSH session (enter sudo password when prompted)..."
if ssh -t -o StrictHostKeyChecking=no "$SSH_USER@$host" \
"sudo bash -s" <<< "$fix_script"; then
echo ""
info "$host fixed and rebuilt"
else
rc=$?
echo ""
warn "$host: rebuild exited with code $rc (may still have succeeded — check sops-nix below)"
fi
# Verify: re-check sops-nix result post-rebuild
sops_result_after=$(ssh "${SSH_OPTS[@]}" "$SSH_USER@$host" \
"systemctl show sops-nix --property=Result --value 2>/dev/null || echo unknown" 2>/dev/null || echo "ssh-failed")
if [ "$sops_result_after" = "success" ]; then
info "$host sops-nix: success post-rebuild"
else
warn "$host sops-nix: $sops_result_after post-rebuild (may need another pass)"
fi
echo ""
done
echo "Recovery complete."
+2 -3
View File
@@ -41,9 +41,8 @@ mkdir -p "$keydir"
keyfile="${keydir}/${hostname}_ssh_host_ed25519_key" keyfile="${keydir}/${hostname}_ssh_host_ed25519_key"
if [[ -f "$keyfile" ]]; then if [[ -f "$keyfile" ]]; then
echo "Key already exists: ${keyfile}" echo "ERROR: $keyfile already exists. Remove it first if you want to regenerate." >&2
echo "Reusing the existing key. Remove it first if you want to regenerate." exit 1
exit 0
fi fi
nix_extra_opts nix_extra_opts
-318
View File
@@ -1,318 +0,0 @@
#!/usr/bin/env bash
# Pushes newly-generated SSH host keys from host-keys/ to already-running
# NixOS hosts, so they can decrypt sops secrets after a nixos-rebuild
# following scripts/secrets/sync-host-keys.sh --regenerate-all-keys.
#
# Before pushing any key, verifies that .sops.yaml and secrets/*.yaml are
# committed and pushed to the remote -- hosts rebuild from the remote Gitea
# flake, so recipient changes must land there before any rebuild, not just
# before the key push.
#
# push-host-keys.sh --all [--dry-run] [--skip-git-check]
# push-host-keys.sh <target> [--dry-run] [--skip-git-check]
#
# --all Push to every reachable managed host. Default when no
# target is given.
# <target> Push to one flake target only (e.g. lxc-server).
# --dry-run Print what would be done; write nothing.
# --skip-git-check Skip the commit/push check. Use only when the remote
# already has the current .sops.yaml/secrets/*.yaml.
#
# SSH: connects as SSH_USER@<hostname> (default: nixos, the user with the
# admin authorized key), then installs files via sudo -S (reads the sudo
# password from stdin). The password is prompted once at startup and reused
# for every host -- no PTY or terminal required on the remote side.
# Hosts are reached at their bare hostname (relies on LAN DNS/mDNS).
set -euo pipefail
repo_root="$(cd "$(dirname "$0")/../.." && pwd)"
keydir="${repo_root}/host-keys"
# shellcheck source=../env.sh
source "${repo_root}/scripts/env.sh"
# shellcheck source=../lib/nix-eval.sh
source "${repo_root}/scripts/lib/nix-eval.sh"
: "${SSH_USER:=nixos}"
SSH_OPTS=(-o StrictHostKeyChecking=no -o BatchMode=yes -o ConnectTimeout=5)
dry_run=0
skip_git_check=0
sudo_password=""
usage() {
cat <<EOF
Usage: $0 [--all | <target>] [--dry-run] [--skip-git-check]
--all Push to every reachable managed host. Default when no
target is given.
<target> Push to one flake target only (e.g. lxc-server).
--dry-run Print what would be done; write nothing.
--skip-git-check Skip the check that .sops.yaml/secrets/*.yaml are
committed and pushed to the remote repo.
Environment:
SSH_USER SSH username (default: nixos).
SUDO_PASS Sudo password (skips the interactive prompt; useful
when calling from another script).
EOF
}
# Prompt for the sudo password once; store it for all _do_push calls.
# Accepts SUDO_PASS from the environment to allow non-interactive callers.
prompt_sudo_password() {
[[ "$dry_run" -eq 1 ]] && return
if [[ -n "${SUDO_PASS:-}" ]]; then
sudo_password="$SUDO_PASS"
return
fi
# read exits non-zero when stdin is not a terminal (e.g. CI, background
# agents). Catch that and give a clear message rather than a silent exit.
if ! read -r -s -p "sudo password for ${SSH_USER} on remote hosts: " sudo_password; then
echo >&2
echo "ERROR: stdin is not a terminal -- cannot prompt for sudo password." >&2
echo " Set SUDO_PASS=<password> in the environment and re-run." >&2
exit 1
fi
echo >&2
}
locally_managed_hosts() {
for f in "${keydir}"/*_ssh_host_ed25519_key.pub; do
[[ -e "$f" ]] || continue
basename "$f" _ssh_host_ed25519_key.pub
done
}
# --- git state check/fix --------------------------------------------------
# Hosts rebuild from the remote Gitea flake:
# nixos-rebuild switch --flake "git+https://<gitea>/nixos.git#<target>"
# so .sops.yaml (updated recipients) and secrets/*.yaml (re-encrypted DEKs)
# must be committed and pushed before any rebuild can succeed. This check
# catches the common case where --regenerate-all-keys was just run but the
# resulting diff hasn't been committed/pushed yet.
ensure_remote_current() {
[[ "$skip_git_check" -eq 1 ]] && return
cd "$repo_root"
local dirty_unstaged dirty_staged
dirty_unstaged="$(git diff --name-only -- .sops.yaml secrets/ 2>/dev/null || true)"
dirty_staged="$(git diff --cached --name-only -- .sops.yaml secrets/ 2>/dev/null || true)"
if [[ -n "$dirty_unstaged" || -n "$dirty_staged" ]]; then
echo "Uncommitted changes in sops-managed files:"
[[ -n "$dirty_unstaged" ]] && sed 's/^/ (unstaged) /' <<<"$dirty_unstaged"
[[ -n "$dirty_staged" ]] && sed 's/^/ (staged) /' <<<"$dirty_staged"
echo
if [[ "$dry_run" -eq 1 ]]; then
echo "[dry-run] would prompt to commit .sops.yaml/secrets/ before continuing."
else
read -rp "Commit .sops.yaml + secrets/ now? [y/N]: " ans
if [[ "$ans" =~ ^[Yy]$ ]]; then
git add -- .sops.yaml secrets/
git commit -m "secrets: update recipients and re-encrypt for host key changes"
echo "Committed."
else
echo "Continuing with uncommitted changes -- the remote won't have the"
echo "updated recipients until you commit and push."
fi
fi
echo
fi
# Check if we're ahead of the remote tracking branch
local ahead
ahead="$(git rev-list --count '@{upstream}..HEAD' 2>/dev/null || echo "")"
if [[ -z "$ahead" ]]; then
echo "NOTE: no remote tracking branch found -- skipping push check."
echo " Ensure the remote has the current .sops.yaml/secrets/ before"
echo " triggering nixos-rebuild on any host."
echo
return
fi
if [[ "$ahead" -gt 0 ]]; then
echo "Local branch is ${ahead} commit(s) ahead of remote."
if [[ "$dry_run" -eq 1 ]]; then
echo "[dry-run] would prompt to push before continuing."
else
read -rp "Push to remote now? [y/N]: " ans
if [[ "$ans" =~ ^[Yy]$ ]]; then
git push
echo "Pushed."
else
echo "Continuing without pushing -- remember to push before running"
echo "nixos-rebuild on any of these hosts."
fi
fi
echo
fi
}
# --- key installation (shared) -------------------------------------------
_do_push() {
local hostname="$1" target="$2"
local keyfile="${keydir}/${target}_ssh_host_ed25519_key"
local pubfile="${keyfile}.pub"
if [[ "$dry_run" -eq 1 ]]; then
echo " [dry-run] would scp host-keys/${target}_ssh_host_ed25519_key{,.pub} to /tmp/"
echo " [dry-run] would: sudo -S install -m 0600/0644 to /etc/ssh/ and rm /tmp copies"
return
fi
# Upload to /tmp (writable as nixos, no privilege needed)
scp -o StrictHostKeyChecking=no \
"$keyfile" "${SSH_USER}@${hostname}:/tmp/push_ed25519_key"
scp -o StrictHostKeyChecking=no \
"$pubfile" "${SSH_USER}@${hostname}:/tmp/push_ed25519_key.pub"
# Install via sudo -S: the password is piped via herestring so no PTY is
# needed on either side. -p '' suppresses sudo's own prompt string.
ssh -o StrictHostKeyChecking=no "${SSH_USER}@${hostname}" \
"sudo -S -p '' bash -c '
install -m 0600 /tmp/push_ed25519_key /etc/ssh/ssh_host_ed25519_key
install -m 0644 /tmp/push_ed25519_key.pub /etc/ssh/ssh_host_ed25519_key.pub
rm -f /tmp/push_ed25519_key /tmp/push_ed25519_key.pub
echo \" [ok] host key installed\"
'" <<< "$sudo_password"
# Drop the stale known_hosts entry for this host (public key just changed)
ssh-keygen -R "$hostname" 2>/dev/null || true
echo " Done. Run nixos-rebuild switch on ${hostname} to activate."
}
# --- single named target --------------------------------------------------
push_target() {
local target="$1"
local keyfile="${keydir}/${target}_ssh_host_ed25519_key"
if [[ ! -f "$keyfile" ]]; then
echo "ERROR: host-keys/${target}_ssh_host_ed25519_key not found." >&2
echo " This target may not be locally managed (e.g. &${target} was" >&2
echo " registered from the host's real SSH key, not generated here)." >&2
exit 1
fi
local hostname
hostname="$(flake_target_hostname "$repo_root" "$target")"
if [[ -z "$hostname" ]]; then
echo "ERROR: cannot resolve hostname for '${target}' from the flake." >&2
exit 1
fi
echo "==> ${target} (→ ${hostname})"
if ! ssh "${SSH_OPTS[@]}" "${SSH_USER}@${hostname}" true 2>/dev/null; then
echo " SKIP: ${SSH_USER}@${hostname} unreachable."
return
fi
# Sanity-check that /etc/flake-target on the host agrees
local live_target
live_target="$(ssh "${SSH_OPTS[@]}" "${SSH_USER}@${hostname}" \
"cat /etc/flake-target 2>/dev/null || true")"
if [[ -n "$live_target" && "$live_target" != "$target" ]]; then
echo " WARN: host reports /etc/flake-target='${live_target}', not '${target}'."
echo " Pushing the key you specified (${target}) anyway."
fi
_do_push "$hostname" "$target"
}
# --- all managed hosts ----------------------------------------------------
# For each unique hostname derived from managed targets, SSHes in and reads
# /etc/flake-target to determine which key to push -- handles the case where
# multiple targets share a hostname (e.g. lxc-server and proxmox-server both
# resolve to "server"; only one is actually running).
push_all() {
mapfile -t managed < <(locally_managed_hosts)
if [[ "${#managed[@]}" -eq 0 ]]; then
echo "No managed keys in host-keys/ -- nothing to push."
return
fi
echo "Pushing to all reachable managed hosts..."
echo
declare -A seen_hostnames=()
local t hostname
for t in "${managed[@]}"; do
hostname="$(flake_target_hostname "$repo_root" "$t" 2>/dev/null || true)"
[[ -z "$hostname" ]] && continue
[[ -n "${seen_hostnames[$hostname]+x}" ]] && continue
seen_hostnames["$hostname"]=1
echo "==> checking ${hostname}"
if ! ssh "${SSH_OPTS[@]}" "${SSH_USER}@${hostname}" true 2>/dev/null; then
echo " SKIP: ${SSH_USER}@${hostname} unreachable."
continue
fi
# Ask the host which flake target it actually is
local live_target
live_target="$(ssh "${SSH_OPTS[@]}" "${SSH_USER}@${hostname}" \
"cat /etc/flake-target 2>/dev/null || true")"
if [[ -z "$live_target" ]]; then
echo " SKIP: no /etc/flake-target on host -- can't determine which key to push."
continue
fi
local live_keyfile="${keydir}/${live_target}_ssh_host_ed25519_key"
if [[ ! -f "$live_keyfile" ]]; then
echo " SKIP: host is '${live_target}' but no host-keys/${live_target}_... (hand-registered key, not managed here)."
continue
fi
echo " target: ${live_target}"
_do_push "$hostname" "$live_target"
done
}
# --- main -----------------------------------------------------------------
mode="all"
target_arg=""
extra_args=()
for arg in "$@"; do
case "$arg" in
--dry-run) dry_run=1 ;;
--skip-git-check) skip_git_check=1 ;;
--all) mode="all" ;;
-h|--help) usage; exit 0 ;;
--*) echo "Unknown option: $arg" >&2; usage >&2; exit 1 ;;
*) extra_args+=("$arg") ;;
esac
done
if [[ "${#extra_args[@]}" -gt 1 ]]; then
echo "ERROR: specify at most one target (or --all)." >&2
usage >&2; exit 1
elif [[ "${#extra_args[@]}" -eq 1 ]]; then
mode="single"
target_arg="${extra_args[0]}"
fi
[[ "$dry_run" -eq 1 ]] && { echo "[dry-run] no changes will be made"; echo; }
nix_extra_opts
ensure_remote_current
prompt_sudo_password
if [[ "$mode" == "single" ]]; then
push_target "$target_arg"
else
push_all
fi
echo
if [[ "$dry_run" -eq 1 ]]; then
echo "[dry-run] Nothing was changed. Re-run without --dry-run to apply."
else
echo "Key push complete. For each updated host, run nixos-rebuild switch to"
echo "apply the config and let sops-nix decrypt secrets with the new key."
fi
+62 -87
View File
@@ -11,16 +11,17 @@
# sync-host-keys.sh --regenerate-all-keys Remove and freshly regenerate # sync-host-keys.sh --regenerate-all-keys Remove and freshly regenerate
# every locally-managed key. # every locally-managed key.
# #
# "Generate/register" is idempotent and additive only: an existing clan # "Generate/register" is idempotent and additive only: an existing
# var is never overwritten, and .sops.yaml only ever gains an anchor/alias # host-keys/ file is never touched, and .sops.yaml only ever gains an
# it doesn't already have -- safe to re-run any time, e.g. right after # anchor/alias it doesn't already have -- safe to re-run any time, e.g.
# adding a new host to flake.nix. # right after adding a new host to flake.nix.
# #
# --remove and --regenerate-all-keys only ever operate on anchors that # --remove and --regenerate-all-keys only ever operate on anchors that have
# have a corresponding clan var (vars/per-machine/<name>/openssh/) or # a corresponding host-keys/<name>_ssh_host_ed25519_key file. Anchors
# host-keys/ file. Anchors without either (&admin) are never listed, # without one (&admin, and any anchor for an already-deployed host whose
# removed, or regenerated -- this tooling only ever touches keys it itself # real /etc/ssh key was registered by hand, e.g. &docker/&server/&nix-cache
# manages. # today) are never listed, removed, or regenerated -- this tooling only
# ever touches keys it itself manages.
set -euo pipefail set -euo pipefail
repo_root="$(cd "$(dirname "$0")/../.." && pwd)" repo_root="$(cd "$(dirname "$0")/../.." && pwd)"
@@ -38,8 +39,6 @@ source "${repo_root}/scripts/lib/ssh-host-keys.sh"
source "${repo_root}/scripts/lib/sops-age.sh" source "${repo_root}/scripts/lib/sops-age.sh"
# shellcheck source=../lib/confirm.sh # shellcheck source=../lib/confirm.sh
source "${repo_root}/scripts/lib/confirm.sh" source "${repo_root}/scripts/lib/confirm.sh"
# shellcheck source=../lib/clan-vars.sh
source "${repo_root}/scripts/lib/clan-vars.sh"
mkdir -p "$keydir" mkdir -p "$keydir"
@@ -55,14 +54,13 @@ Usage: $0 --all [--dry-run]
<flake-target> Same, for just one target (e.g. lxc-server). <flake-target> Same, for just one target (e.g. lxc-server).
Reports if it already has one. Reports if it already has one.
--remove Interactively pick one locally-managed key to --remove Interactively pick one locally-managed key to
remove from .sops.yaml and vars/per-machine/ remove from .sops.yaml and host-keys/.
(or host-keys/ for legacy keys).
--regenerate-all-keys Remove every locally-managed key and generate --regenerate-all-keys Remove every locally-managed key and generate
fresh clan-var replacements for every current fresh replacements for every current flake
flake target. Destructive -- requires typed target. Destructive -- requires typed
confirmation. confirmation.
--dry-run Combine with any of the above: print what would --dry-run Combine with any of the above: print what would
change (clan vars, .sops.yaml anchors and change (host-keys/ files, .sops.yaml anchors and
key_groups, which secrets/*.yaml would be key_groups, which secrets/*.yaml would be
re-encrypted) without touching anything. No keys re-encrypted) without touching anything. No keys
generated, no files written, no sops calls, generated, no files written, no sops calls,
@@ -85,10 +83,6 @@ ensure_admin_decrypt_key() {
fi fi
local key_file="$DEFAULT_SOPS_AGE_KEY_FILE" local key_file="$DEFAULT_SOPS_AGE_KEY_FILE"
# Expand a leading ~ that survived variable substitution without tilde
# expansion (happens when SOPS_AGE_KEY_FILE or XDG_CONFIG_HOME is set with
# a literal ~ in the caller's environment).
key_file="${key_file/#~\//$HOME/}"
if [[ -s "$key_file" ]]; then if [[ -s "$key_file" ]]; then
echo "Found existing sops age key at ${key_file}." echo "Found existing sops age key at ${key_file}."
@@ -97,23 +91,36 @@ ensure_admin_decrypt_key() {
if [[ "$dry_run" -eq 1 ]]; then if [[ "$dry_run" -eq 1 ]]; then
echo "[dry-run] No sops age decryption key found (checked \$SOPS_AGE_KEY, \$SOPS_AGE_KEY_FILE, ${key_file})." echo "[dry-run] No sops age decryption key found (checked \$SOPS_AGE_KEY, \$SOPS_AGE_KEY_FILE, ${key_file})."
echo "[dry-run] Continuing dry run without one -- any 'would re-encrypt' output below" echo "[dry-run] Would generate a new one here -- continuing the dry run without one; any"
echo "[dry-run] couldn't actually run for real until a key is present." echo "[dry-run] 'would re-encrypt' output below couldn't actually run for real yet."
return return
fi fi
cat >&2 <<EOF echo "No sops age decryption key found (checked \$SOPS_AGE_KEY, \$SOPS_AGE_KEY_FILE, ${key_file})."
No sops age decryption key found (checked \$SOPS_AGE_KEY, \$SOPS_AGE_KEY_FILE, ${key_file}). echo "Generating a new one at ${key_file}..."
mkdir -p "$(dirname "$key_file")"
nix-shell "${NIX_OPTS[@]}" -p age --run "age-keygen -o '${key_file}'" 2>&1 | grep -v "^Public key:" || true
local new_pub
new_pub="$(age_pubkey_from_identity_file "$key_file")"
Place your admin age private key at ${key_file}, or set SOPS_AGE_KEY (inline cat <<EOF
key) or SOPS_AGE_KEY_FILE (path to a different key file) and re-run.
If the key is truly missing (not just mislocated), this is a manual recovery A brand-new age key was just generated -- it cannot decrypt anything that
situation -- generating a brand-new admin key won't help, since it cannot already exists in secrets/*.yaml, since nothing was ever encrypted for it.
decrypt anything already encrypted for the old one. Each secrets/*.yaml is That trust can't be bootstrapped automatically (nobody can decrypt a file
also encrypted for its respective host key(s), so a running deployed host can for a recipient that didn't exist when it was last encrypted).
still decrypt what it needs -- but the admin key is required for re-encryption
(e.g. adding new recipients via sops updatekeys). To actually use this key:
1. Have someone who currently CAN decrypt replace the &admin entry in
.sops.yaml with this public key:
${new_pub}
2. They re-encrypt every secrets/*.yaml:
sops updatekeys --yes secrets/common.yaml
sops updatekeys --yes secrets/nix-cache.yaml
sops updatekeys --yes secrets/server.yaml
3. Re-run this script.
Exiting without making any other changes.
EOF EOF
exit 1 exit 1
} }
@@ -126,17 +133,10 @@ discover_targets() {
} }
locally_managed_hosts() { locally_managed_hosts() {
{ for f in "$keydir"/*_ssh_host_ed25519_key.pub; do
for f in "$keydir"/*_ssh_host_ed25519_key.pub; do [[ -e "$f" ]] || continue
[[ -e "$f" ]] || continue basename "$f" _ssh_host_ed25519_key.pub
basename "$f" _ssh_host_ed25519_key.pub done
done
local d
for d in "${repo_root}/vars/per-machine"/*/openssh/ssh_host_ed25519_key/secret; do
[[ -f "$d" ]] || continue
basename "$(dirname "$(dirname "$(dirname "$d")")")"
done
} | sort -u
} }
add_keys_json="[]" add_keys_json="[]"
@@ -146,15 +146,13 @@ dry_run=0
queue_host_sync() { queue_host_sync() {
local host="$1" local host="$1"
local keyfile="${keydir}/${host}_ssh_host_ed25519_key" local keyfile="${keydir}/${host}_ssh_host_ed25519_key"
local has_local_key=0 has_clan_key=0 has_anchor=0 local has_local_key=0 has_anchor=0
[[ -f "$keyfile" ]] && has_local_key=1 [[ -f "$keyfile" ]] && has_local_key=1
clan_ssh_key_exists "$host" "$repo_root" && has_clan_key=1
grep -qE "^ - &${host} age1" "$sops_yaml" && has_anchor=1 grep -qE "^ - &${host} age1" "$sops_yaml" && has_anchor=1
if [[ "$has_local_key" -eq 0 && "$has_clan_key" -eq 0 && "$has_anchor" -eq 1 ]]; then if [[ "$has_local_key" -eq 0 && "$has_anchor" -eq 1 ]]; then
echo "SKIP ${host}: .sops.yaml already has an &${host} anchor, but" echo "SKIP ${host}: .sops.yaml already has an &${host} anchor, but"
echo " neither host-keys/${host}_ssh_host_ed25519_key nor" echo " host-keys/${host}_ssh_host_ed25519_key is missing locally."
echo " vars/per-machine/${host}/openssh/ exist locally."
echo " Not generating a replacement -- it wouldn't match whatever's" echo " Not generating a replacement -- it wouldn't match whatever's"
echo " already registered (and possibly deployed). Remove the" echo " already registered (and possibly deployed). Remove the"
echo " &${host} line from .sops.yaml first if you really want a" echo " &${host} line from .sops.yaml first if you really want a"
@@ -162,26 +160,21 @@ queue_host_sync() {
return 1 return 1
fi fi
if [[ "$has_local_key" -eq 0 && "$has_clan_key" -eq 0 ]]; then if [[ "$has_local_key" -eq 0 ]]; then
if [[ "$dry_run" -eq 1 ]]; then if [[ "$dry_run" -eq 1 ]]; then
echo "[dry-run] ${host}: would generate host key via clan vars" echo "[dry-run] ${host}: would generate host key"
else else
echo "==> ${host}: generating host key via clan vars" echo "==> ${host}: generating host key"
clan_generate_ssh_key "$host" "$repo_root" generate_host_ed25519_key "$host" "$keyfile"
has_clan_key=1
fi fi
elif [[ "$has_clan_key" -eq 1 ]]; then
echo "==> ${host}: clan-managed SSH host key already present"
else else
echo "==> ${host}: host key already present (host-keys/)" echo "==> ${host}: host key already present"
fi fi
if [[ "$has_anchor" -eq 0 ]]; then if [[ "$has_anchor" -eq 0 ]]; then
local age_pub local age_pub
if [[ "$dry_run" -eq 1 ]]; then if [[ "$dry_run" -eq 1 ]]; then
age_pub="dry-run-placeholder-not-a-real-key" age_pub="dry-run-placeholder-not-a-real-key"
elif [[ "$has_clan_key" -eq 1 ]]; then
age_pub="$(ssh_pubkey_to_age "$(clan_ssh_pubkey_path "$host" "$repo_root")")"
else else
age_pub="$(ssh_pubkey_to_age "${keyfile}.pub")" age_pub="$(ssh_pubkey_to_age "${keyfile}.pub")"
fi fi
@@ -306,7 +299,7 @@ cmd_remove() {
local hosts local hosts
mapfile -t hosts < <(locally_managed_hosts) mapfile -t hosts < <(locally_managed_hosts)
if [[ "${#hosts[@]}" -eq 0 ]]; then if [[ "${#hosts[@]}" -eq 0 ]]; then
echo "No locally-managed keys found (checked host-keys/ and vars/per-machine/) -- nothing to remove." echo "No locally-managed keys in host-keys/ -- nothing to remove."
return return
fi fi
@@ -315,9 +308,7 @@ cmd_remove() {
for host in "${hosts[@]}"; do for host in "${hosts[@]}"; do
local registered="not registered in .sops.yaml" local registered="not registered in .sops.yaml"
grep -qE "^ - &${host} age1" "$sops_yaml" && registered="registered in .sops.yaml" grep -qE "^ - &${host} age1" "$sops_yaml" && registered="registered in .sops.yaml"
local where="host-keys/" printf ' %d) %s (%s)\n' "$i" "$host" "$registered"
clan_ssh_key_exists "$host" "$repo_root" && where="clan-vars"
printf ' %d) %s [%s, %s]\n' "$i" "$host" "$where" "$registered"
i=$((i + 1)) i=$((i + 1))
done done
@@ -334,7 +325,7 @@ cmd_remove() {
local target="${hosts[$((choice - 1))]}" local target="${hosts[$((choice - 1))]}"
if [[ "$dry_run" -ne 1 ]]; then if [[ "$dry_run" -ne 1 ]]; then
read -rp "Really remove '${target}'? Its key files will be deleted and it will lose access to every secrets file it can currently decrypt. (y/N): " confirm read -rp "Really remove '${target}'? Its host-keys/ files will be deleted and it will lose access to every secrets file it can currently decrypt. (y/N): " confirm
if [[ ! "$confirm" =~ ^[Yy]$ ]]; then if [[ ! "$confirm" =~ ^[Yy]$ ]]; then
echo "Cancelled." echo "Cancelled."
return return
@@ -347,13 +338,11 @@ cmd_remove() {
apply_edit_plan "$plan" apply_edit_plan "$plan"
if [[ "$dry_run" -eq 1 ]]; then if [[ "$dry_run" -eq 1 ]]; then
echo "[dry-run] would delete host-keys/${target}_ssh_host_ed25519_key(.pub) if present." echo "[dry-run] would delete host-keys/${target}_ssh_host_ed25519_key(.pub)."
echo "[dry-run] would delete vars/per-machine/${target}/openssh/ if present."
echo "[dry-run] Nothing was changed. Re-run without --dry-run to apply this." echo "[dry-run] Nothing was changed. Re-run without --dry-run to apply this."
else else
rm -f "${keydir}/${target}_ssh_host_ed25519_key" "${keydir}/${target}_ssh_host_ed25519_key.pub" rm -f "${keydir}/${target}_ssh_host_ed25519_key" "${keydir}/${target}_ssh_host_ed25519_key.pub"
rm -rf "${repo_root}/vars/per-machine/${target}/openssh" echo "Removed host-keys/${target}_ssh_host_ed25519_key(.pub)."
echo "Removed key for ${target} (host-keys/ and/or vars/per-machine/ as applicable)."
echo echo
echo "Review the diff, then commit and push." echo "Review the diff, then commit and push."
fi fi
@@ -363,18 +352,15 @@ cmd_regenerate_all() {
local hosts local hosts
mapfile -t hosts < <(locally_managed_hosts) mapfile -t hosts < <(locally_managed_hosts)
if [[ "${#hosts[@]}" -eq 0 ]]; then if [[ "${#hosts[@]}" -eq 0 ]]; then
echo "No locally-managed keys found (checked host-keys/ and vars/per-machine/) -- nothing to regenerate." echo "No locally-managed keys in host-keys/ -- nothing to regenerate."
return return
fi fi
echo "This will remove and freshly regenerate ALL locally-managed keys:" echo "This will remove and freshly regenerate ALL locally-managed keys:"
printf ' %s\n' "${hosts[@]}" printf ' %s\n' "${hosts[@]}"
echo echo
echo "After regenerating, each host needs its new key before it can decrypt secrets:" echo "Every host above will need its new key baked into a rebuilt install"
echo " • Already running: push the key before rebuilding:" echo "image/tarball before it can decrypt secrets again."
echo " scripts/secrets/push-host-keys.sh --all"
echo " • Not yet deployed: rebuild the install image with the new keys baked in"
echo " (see docs/auto-installer.md)."
if [[ "$dry_run" -ne 1 ]]; then if [[ "$dry_run" -ne 1 ]]; then
if ! confirm_typed "REGENERATE" "Type REGENERATE to confirm: "; then if ! confirm_typed "REGENERATE" "Type REGENERATE to confirm: "; then
@@ -392,8 +378,8 @@ cmd_regenerate_all() {
apply_edit_plan "$plan" apply_edit_plan "$plan"
if [[ "$dry_run" -eq 1 ]]; then if [[ "$dry_run" -eq 1 ]]; then
echo "[dry-run] would delete ${#hosts[@]} key pair(s) from host-keys/ and/or vars/per-machine/." echo "[dry-run] would delete ${#hosts[@]} host-keys/ file pair(s)."
echo "[dry-run] would then generate fresh clan vars replacements for the same hosts" echo "[dry-run] would then generate fresh replacements for the same hosts"
echo "[dry-run] (not simulated further here -- run without --dry-run, or" echo "[dry-run] (not simulated further here -- run without --dry-run, or"
echo "[dry-run] preview a specific target with: $0 <target> --dry-run)." echo "[dry-run] preview a specific target with: $0 <target> --dry-run)."
echo echo
@@ -405,23 +391,12 @@ cmd_regenerate_all() {
local host local host
for host in "${hosts[@]}"; do for host in "${hosts[@]}"; do
rm -f "${keydir}/${host}_ssh_host_ed25519_key" "${keydir}/${host}_ssh_host_ed25519_key.pub" rm -f "${keydir}/${host}_ssh_host_ed25519_key" "${keydir}/${host}_ssh_host_ed25519_key.pub"
rm -rf "${repo_root}/vars/per-machine/${host}/openssh"
done done
echo "Removed ${#hosts[@]} key pair(s)." echo "Removed ${#hosts[@]} host-keys/ file pair(s)."
echo echo
echo "Regenerating fresh keys for every current flake target..." echo "Regenerating fresh keys for every current flake target..."
cmd_all cmd_all
echo
echo "Next steps:"
echo " 1. Commit and push .sops.yaml + secrets/ so the remote flake is current."
echo " 2. Push the new host key to each already-running managed host:"
echo " scripts/secrets/push-host-keys.sh --all"
echo " (this also prompts to commit/push if step 1 wasn't done yet)"
echo " 3. Run nixos-rebuild switch on each updated host."
echo " 4. For hosts not yet deployed, rebuild the install image (see"
echo " docs/auto-installer.md)."
} }
main() { main() {
+81 -182
View File
@@ -1,227 +1,126 @@
root-hashedPassword: ENC[AES256_GCM,data:Kp0nOZI7vDoLhJHiOJBwJn0rQZ5yhnwapGnAcA+qh8vlDETtFs/iQdetF/2ZxmANf62SviTNd+Ag0q5JIF1996x7onZGXqxgSMCuVzZLBdUlsO5IR0BslWWz47khYGTe4WkUg4NB1itBfQ==,iv:5Sra5vJ79V8hxQT3g9qJ+dOj2W2sumIhqpitqnHjJdk=,tag:3Igu0+8GeUZHqS3fKUVwog==,type:str] root-hashedPassword: ENC[AES256_GCM,data:Kp0nOZI7vDoLhJHiOJBwJn0rQZ5yhnwapGnAcA+qh8vlDETtFs/iQdetF/2ZxmANf62SviTNd+Ag0q5JIF1996x7onZGXqxgSMCuVzZLBdUlsO5IR0BslWWz47khYGTe4WkUg4NB1itBfQ==,iv:5Sra5vJ79V8hxQT3g9qJ+dOj2W2sumIhqpitqnHjJdk=,tag:3Igu0+8GeUZHqS3fKUVwog==,type:str]
nixos-hashedPassword: ENC[AES256_GCM,data:pT7tVRN6X4a+DNUgB7fIUUE3CbnetkjxmoSL1PxSU+ktsFU+fB0mEvJjA1uujsGH5Rcztg7YM815+M0Z67ILmHaXbza5DtFacrqhi4/b277xly0SHRX4yOvBwQh6mJG1jn/0O/wvUUIYdw==,iv:bp2nfhC8nFbk6o5iWDAugvbzu7J/a1xayFnBEtkhNpE=,tag:HqWgkIpSrSM/K9OK2WO+VQ==,type:str] nixos-hashedPassword: ENC[AES256_GCM,data:pT7tVRN6X4a+DNUgB7fIUUE3CbnetkjxmoSL1PxSU+ktsFU+fB0mEvJjA1uujsGH5Rcztg7YM815+M0Z67ILmHaXbza5DtFacrqhi4/b277xly0SHRX4yOvBwQh6mJG1jn/0O/wvUUIYdw==,iv:bp2nfhC8nFbk6o5iWDAugvbzu7J/a1xayFnBEtkhNpE=,tag:HqWgkIpSrSM/K9OK2WO+VQ==,type:str]
nix-github-token: ENC[AES256_GCM,data:k1vYz7SqVhzpWa6jTL6NUD8lKOCpHCgTm+HT4IcnbzbSTUZP/bJUYw==,iv:UqAULZnr/4+VcioUDfTwvOSuwM8K9JgGhiApvYQPyoc=,tag:1LKHXhWAO/AHPDIZFBb04A==,type:str] nix-github-token: ENC[AES256_GCM,data:OfNRGJg16Ede6EilWUetCs9za+xk5/Lsa3SpVajsqz8PMdA1xQNeCWdX7ZAMdijHClpBhU6ETFGsXvt41O9aORS951uijeGSW7/NH35/bnPISrKdYeBx/+xEiqwH,iv:QGU3v7xOy89uzRTCb1U9ICyJ8XYIpXrUsDt12aL3g2Y=,tag:Bde2wcWNv8H4WLxSEUAodg==,type:str]
beszel-token: ENC[AES256_GCM,data:OWmSRkZjb11y0Y8GdobqiE9GFwzdHOvvxCbYx69qUghGYARN,iv:i/JhGH0O7ThxPkL0SLAjfN0Fq8prm7tybI5kF2NRNpw=,tag:dBcqxOSHTnD4xngpOog55Q==,type:str]
nix-gitea-token: ENC[AES256_GCM,data:HRQ8ymx/D8pLcL/pYIhcSTv3tlBCKOU5nhXPp/W6rNI+t+yxfOoRPQ==,iv:0Av3lrxQew2bDFf67nX/UM+cD0/8CtbQ2fZSaKWHAzM=,tag:0K+atsB/YmYd/qCN3hMdBw==,type:str]
sops: sops:
age: age:
- enc: | - enc: |
-----BEGIN AGE ENCRYPTED FILE----- -----BEGIN AGE ENCRYPTED FILE-----
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSAyR3lnUnZwUnlJTkFaeVdz YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBQUFhWVFVlVlBnNE5FTnMz
T3Q2eW52cWI2eTlhSFpHNDRHTEVNVGZJYm1NCldaMmlHaCt1K3lWRHhEZExHK3lo VkxkTmxpRXlzZ3pSNTVZWFUrSllsYWo0alRnCkJSc25TYktSTFFJdkQydHcxOUlj
azZFck1URVg5ejBaRVdCWjdFMTc4dGsKLS0tIG9mNkxsZXI1N1hCRWk1NTZKcUUr ajhQU1ZIb1lodEpHTnVhQjJ6WEthaDQKLS0tIDJCY1E2UVBaU3BoMzhXUXlIdnMv
SlJoWGdWbXhqZEJHM3IzZmZSQ25QbzAKPzBIA/IJiZr5NpOhB6IPkUSDGQzPwpTU djZTcE1rcWNTOXFPMmFDYTVoRGo4ZTQKYy8g6pqP3VpTKDIBPbnC8NzCdDvOCKnL
vgFLMze8OSEviaGXKLt/ZwTXHsr5As7V9yGvJKJHhS/hzuSLRjs8rQ== 14kSrKmKlzefTrbkVyriz2Jdl2s0F374yfQQFreZ3m4AffSACCxziQ==
-----END AGE ENCRYPTED FILE----- -----END AGE ENCRYPTED FILE-----
recipient: age1njap586hc0q43kr03g6c8eqhdsmk8zcafkl3f83xwlc2gqhlmfgs4tmwad recipient: age1njap586hc0q43kr03g6c8eqhdsmk8zcafkl3f83xwlc2gqhlmfgs4tmwad
- enc: | - enc: |
-----BEGIN AGE ENCRYPTED FILE----- -----BEGIN AGE ENCRYPTED FILE-----
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBpYmZHNWVCL0ErNEF3OTlG YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBVQnpVWC9wcnIreG9GeE9U
TTJDT1hZMXJNNUdLZ3ByUDNndVpLWVVGUm1RCkZSRmJmd3NyeHc0V0tpZEdYakF1 UWhuRytkc2Flc0hyQm5yMjZnelNwaXhlWWc4CjBkWnd4cHNRRXQ0UXFkZGp4QlR1
WXhyWUJ0V1hTYzhrSUo1ZUNzR2J3OVUKLS0tIEthYXVvNVRibDVSQ2N0NWt3TmNP eW5NNnE1WFhnb054M1pac2ZidFg4Y3MKLS0tIFdTNmk2V1l2WC9rUk8yd0ZnOEJS
cTlteEtRYmxnUmdIUlBTREZzQXBBSVEKxQq10KDseuoVPe0cLBbk1+weuq0y6di+ VkNnejVGVUZPZkorQkltVEplN2FmdTAKRY7DPP5HeFQntn2f/fXLjU6M1V6iug86
wJjaMShqVQBUI+MFWiQokBhPt8gZS7cs33LkWf3BegNALLDd6HXoOg== BD09PI+T2DbIBQPotRZisw8IzHu9gY/O3+h0TccyIsXjI9wy/XPCAQ==
-----END AGE ENCRYPTED FILE----- -----END AGE ENCRYPTED FILE-----
recipient: age19m0m7vdfg86yqy8l5mmle5jdd0unrn3f55t232w8h5ey42cqw34sfpt32n recipient: age19gfn2yedg76dmztm4hncr7vf3r3c9j0qpt4rap7y7gersjk4m3ks2lhd0e
- enc: | - enc: |
-----BEGIN AGE ENCRYPTED FILE----- -----BEGIN AGE ENCRYPTED FILE-----
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSArOE9oU210Vkk1Z1VKMS8x YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBkczFSTVhxWHlIWjhRcWlV
YTJGbFlSYmx6eWgvMXExZ243SndXWkZ0ekY4ClJVUTFSVkRERjY1UVZMZjlFeE1P V2JPQXd5Wnk5R3NwWC81T3Z0MW4vYnd5S1ZZClV1NlU1Tzd6UkxPQ2M4MmhLV01G
Vm40WHY4UjMwZWxVa05lblBMUW9iNEEKLS0tIGVlbDVxakR4MlBEaFlrM0tNOWRX d3VIb0RhR1RiNTZqNjlQcmg2YjdPeGsKLS0tIDQ1RTFTWGN4MnEvWkRUR3VnN204
elIvMnp5NmhnYVBYOFA0aUdtOE5DbEUKy+soKNLlRe0SC8kcnwrpKqvSrTGE114/ WVdFOXdmNC9FVFhBSGNEUUgyYWpYYzAKfdpeaFL/RrIbqpD9hNj8L7UxpmiBjE2I
FaX2829gQWm0bYI0M4ixeTc5ME2O2Ct2tvYlzfZQPnAuub+jVx2k8g== go/dR2E1LLXsDPtnSuJb2EZYoFvSsjsIQQQDt+YwRv0fplRtssKdxQ==
-----END AGE ENCRYPTED FILE----- -----END AGE ENCRYPTED FILE-----
recipient: age1rrxqea6q6pn39sw8y5te63h2py8jgjl9v0jyper86w3ggtn67upqg3ah39 recipient: age1ll6hj5ggruetgjwjfnplpn5xtq35uhlcdflksx3xmnjm6s3uad9sz70jkf
- enc: | - enc: |
-----BEGIN AGE ENCRYPTED FILE----- -----BEGIN AGE ENCRYPTED FILE-----
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSAzR0R1VEc2eTRIOWNxaVpa YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSB6TjhmOTJ0bUpMQWowb0hB
RXVaNkxjaDVDTmxMVWJEU0dWNkxDeE56c2swCjdkZGVQM3V0SnhGMXFxSEFEeEpP T3Era1loU1pMdmkxdnAvRkViekpqZWZjaWgwCkFRQXhFUy9PRVBma2JMUDhqY2F1
OFNwTWQwYTY0S0trMy9FMjNjbWFxajAKLS0tIHhjUW1lZ1EycUh2Sm1yanhzS3dP VVFDRFNVbWpNaEczY1JVQUMyck9XdEkKLS0tIGpxc0tGdVFKK3FteVJKM1Fxa2ky
WFlMcndzOThUVVVDZlJJeGFBT2JLTEUKHsJ6cwSPcO0IB1CQe2RqKeid8Q92BTNF a21WLy9qV05hUURCTVBvcVh3cE45Z3cKXCYfXSjhApBoLbHDu2OOd57Y1zN54yy+
RfURqE7Curj1yaFB45mzv2ThBgTKN5FE6y5BWgBnF6+szdMMXRw4uQ== WDQvz8PpMxhc1nU5Kw/cI+WmL1KvN0qQZfOx/7D4W+dy/ZDWX27TpA==
-----END AGE ENCRYPTED FILE----- -----END AGE ENCRYPTED FILE-----
recipient: age1adur9g330gua4l6ndk8cqjg35qc8yxwgme6wrl2hpylcc7vxm38q05ejuy recipient: age120le4a5l8dh3lyfgvmj3d9ksmej6ajs5mer5y7r0vfg3x9fn69dqf8xgzu
- enc: | - enc: |
-----BEGIN AGE ENCRYPTED FILE----- -----BEGIN AGE ENCRYPTED FILE-----
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBlQkVPTHBYWkdMM3Jyd3NW YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBZK0FYQnBHMHZ6dlpMYTlC
bkNGa3VtYXNBanZPc1puSGdwUEN5ejIwRkFJCjlNNXFlTEV1eEZNVTFxWnNJZG5O WFdOWDFkRVBuY1pmdTFiUndLV3JXcndZa3pNCnJsd0tHN0FveWV6UUNQSEdpdWw5
UG5hci9XVmMyeXF1ZHF2Rm5DUVdPcFEKLS0tIDBEZUtZNXR2R093OVNNYzRnbXlj dWZITkxWelNIRlpKS1pnN0ZmVlQvZjAKLS0tIEUwMXdtNFdkUWdIRjlxc0owdTRr
S3R2UzFoTmhZT3Y0K1d4STFjU2RRamsKU9LcaOLLjmcarmdir9Hnt/qaNvlxvSsE c1o1TmptWWd1ZGxzcWJJNzJ0K25PTTAKoos5rnkyQBCm+ZuhCCaMJwqJBo1fpnsl
RdXIdOKuaQqJyJ1VEpuDCfuZgtIdkr7OG1360giXFUIEUDliM9OPiA== G74wu5vbTBG4VjVhI5KqyiuiTRU4jPcGxysECqe7AyZUBGp7ndewgw==
-----END AGE ENCRYPTED FILE----- -----END AGE ENCRYPTED FILE-----
recipient: age17e89ty6p0fw24daanen57wg8uald9s025t3wwxsw269svwpmgvrshfvfvt recipient: age1jy444f9d9stygj4p3w9kh54cqcfr654tvr75tdvee5cxsgtdtc9q3v60ep
- enc: | - enc: |
-----BEGIN AGE ENCRYPTED FILE----- -----BEGIN AGE ENCRYPTED FILE-----
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBWUFJETUk2TGZTdVJqUTcz YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBTVkVXaCtGTXVkRXRZMFZ1
MUZvUWhmWGR4SGVxOW8xcy9Pd1A2ZnhPcXpzCnhvL1lPTmZJclJ3OHVqbGtGZFMw ZUdXM1hRcDhaSURqcGh2eksxTUNCckk5SVJjCnhkSDRKdDFSckxUWXd4SmVxWG5p
VysxdnY5ZzgwUEFiVUxpYStUb3NsTEUKLS0tIHRpS0dpb1ZNMXBiRUhFVVZiSHQ2 bU5OV0hzaWR6VDFwcDY2WlY4WnN3cFEKLS0tIElOVzRCcXR4U0dhajJySUhaZGps
UTRoTXFjNVFOeE84WXdsT25UQUprbE0K5U8S5xojEgUn8pAgY6X+Njllv/jqm/Qp MW9rQk1JVDFWRnFxVzhCUkRIS09EamsK1rVidD48PqwlEWQyjF7iQWU7aBdPqQHy
tqeTbMw+w/afAhRxY80x/mTJdCAxUh4guLTKO6eojHYcFvwT6eZA4w== z5LaSi3LvJX3rNE/+q0E8/gbZyjGpbEn3AUI5mBF64GY3IZkRxZSXQ==
-----END AGE ENCRYPTED FILE----- -----END AGE ENCRYPTED FILE-----
recipient: age1hrx8qj02fj2ea6d4g9vqhyj9hl7fppkjqfdx2l37py3h6pdkr95s8n8rvs recipient: age120whqj96g26lsgy4udvgsn8dc9lumh8jeu3a564fx79rjr5lxffqmrljuu
- enc: | - enc: |
-----BEGIN AGE ENCRYPTED FILE----- -----BEGIN AGE ENCRYPTED FILE-----
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBpY1FUZ3RKU080RVNPZFZ6 YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSAvbnJCWU9UMXJXMG92b3hm
cDExYmhLdlU4K2xWU213MkZtUWY4elVxdlF3Ci9LK0cxV2hKbXZEMEd5dzVCWGVU WWNBTFpQamVWQThITmt5QUVwR3h2OHI2SlVRCjcxOEJTVkFjN0NhamFZQ1plK29w
SjErTmpYbDhmcDlSSENhNnRrN1QrcTgKLS0tIFRlbVlKOW51R3ZiZ2pLd0JXdks0 dk5XYkYxOXQ0YkVzcVc3VnhCQWlsV0UKLS0tIHZ5cWtFZUhDKzZkOE1BK2Y5TStR
b0FsZ3VETmREVEhqTGI4Vm5KQnZaWjAKXq+8u2Qk84Vt+eDUxzE6sDk4DDm78P7H SFlDRjE4ZHpiVEJOQk5TUGNEN1B4amMKUCJ8CL8QpmRpFs83HD9TUn7NrPguuP8S
KVnrZfhAmwP3X7dSuBhW+dK8in8D3jaqRK50d/eHUUWX0NIqniUMKg== JQH/bzPorXTXJuyOKuKAZq1hK8BmiMUFksaZ03yN6YaFVIOeelEEMg==
-----END AGE ENCRYPTED FILE----- -----END AGE ENCRYPTED FILE-----
recipient: age1e7l8dusgmgfzd2cxrrzwepzjxt69hzqj4epee0cs27u6yg4kxcuqm34ncx recipient: age1xjst4frdh0th6q8m7p7u9g5af7ty5jqeum0p6z8a52a9q7st7ewqw8yl9j
- enc: | - enc: |
-----BEGIN AGE ENCRYPTED FILE----- -----BEGIN AGE ENCRYPTED FILE-----
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBZU2R4VVVkWTdqS1dxbDhv YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBxWHV0S2dxeHZEMDNRSkhO
K21KeWs2NVo4UTBvNEF6SDRYdE9WUldYUmpRClloR1lBRU9UYkpRSzYxY0NsR2c0 ZFJHaiszd0tzYzlzd3gvbXNSQmlMVlJUNGcwCnVrM29MdFZCR1NBYnpkQ1k5VFZQ
WHlXT0MycWJPdHhHaThWUHltdGZYQXcKLS0tIGJuaTEwSnpmMnFGVml4UUs1ZCtT b3Z2Q3ZGekVQZkZKWGlka3NDOHJ0R1EKLS0tIDlXTmNzUk0wVXo0UWhkd0ZvK3FI
U1V3U2NNN0l4L2JPRHFDOWpWWE9HaVkK/i5m6YFiAR6xtms/pbcDNhKaZreqIpjT UzJxU3RkdWs4aTZYVVkrS056bTN1ek0KgKJNz8GvynX5pK33aW9x3v6yr2Ox0LCT
8tvnqHz2HDSuCMAjAnZfluvuP1USHvjJQGZpBfreZ/XGhW0oqa7D7w== GGrt+ddbKLcwpBpYjfWkFhffO330EKui73S+c/qMf8N9j6wzalOTpQ==
-----END AGE ENCRYPTED FILE----- -----END AGE ENCRYPTED FILE-----
recipient: age1jcx3yajjhghn8qh8za3yeu8nxykzlg3p4nrv03vnfvzl0mzayg2qmg940e recipient: age10at8862478urh0eeuwh8hzln6ck78jgwtztgxatwqlzwagg77y5snm4xzg
- enc: | - enc: |
-----BEGIN AGE ENCRYPTED FILE----- -----BEGIN AGE ENCRYPTED FILE-----
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBKNklENW8yZ3VlMytlVHJs YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSA3ZWJlbDl1TEkvWEpBMC9Y
dUhDcDI0amdZWlhmN3UzbFJyR2x4RGVOdFNVCmovVzFlYnh6YWxYTnYxVzBNejVT ejY1MnFMUExBeDhITndxNy9YUi9hU0tXM1JJCjcwNEppcDdxYzlCSVMrMExWa3A4
MGFhdjZzU2hnMW42d1RucENGeFdXa28KLS0tIGhIZy8rZ0pFM3o0cDBKTXQzaFNO cmFPSjV3LzkyMXZCUDU2QmtHRmpHRmsKLS0tIFFTdXBOaDJJTERseXlGbmdrQzhD
bG9XNzFGdVNyczRhWjRqQXlxNHFvL1UK8NMj76782tmIdJJ4qIzLicFytNhj6ZMk TWtnRFdIRXpsNkY0U1BiczNsdUk1V1kKGpndKmT8kj/oIxQuxQALfzscw+CsVmnj
HIIOJGCrBnqgCtcKTiCrTGRhGbqGzhkId1oJkZkhMFRoU7kdSvwD8Q== cyPC3bF+tG6LcqqoKLjPSJfcIgzhnX7cAr/wwESavemLn8L/zQMe4w==
-----END AGE ENCRYPTED FILE----- -----END AGE ENCRYPTED FILE-----
recipient: age1f7usptjx9rv4rxauasve200gxtdt9jkqhhdqstlf20wvlm7u75rsjfw50m recipient: age1ezk9x53zt8kcnscdm80jcyf0xq97vndv7jsn3rl8cc0cwm2jmpmq372dzs
- enc: | - enc: |
-----BEGIN AGE ENCRYPTED FILE----- -----BEGIN AGE ENCRYPTED FILE-----
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSAzbFU5YWREdFpsK3NScXNU YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBpQVVyY3UrSHFWSmpDdmRE
NUFMMml2VEVOQU5pWkIrUjkzM2tZOG9vcTNRCmR6Vzc5OWRQaGdvU1Naam51dXhD dlh5akFqVFdDVFNMSHE1eVJnZzR6YzFHSVFRCko4UE9EdXNxZzF2MW5PTTN6dEdU
ME1yNFJObktNNTA2eXFsNExTL3JNeHMKLS0tIE13L0xYMCt5TE5ac3FEbkxVWmVv ZHM1MGowcVB2Y1ZlOTVHdnNtY3diM2cKLS0tIDBYSmh5dVVPaTM3d0ErcC8wMDNB
V3NCSk9LUCtPZFZCbVNmSjR5QkFKTmMK5qFJXtZCKLjOCg1r+sVQpMKl75GNcrbI eUpHWnZlYnJsbHZuS3pwbG15UGtwN2MKVPQA1MpjIfYAsNacoAbpvZNuAIkvx7ER
Dum/K/3HU03wv5reG51UDsQ1tMrsFDsaFh2fjR+LxLKSGlG7b3Rz4w== CvWBKEHUVm6m8905BXzv8MdGTAk0EyCIP3aMmYqTIYfv2k9pP0T08A==
-----END AGE ENCRYPTED FILE----- -----END AGE ENCRYPTED FILE-----
recipient: age17jqc66x9yeshfgd9v78mj483r4zzarqdtuxtrkxe4x5mw679gphshd94th recipient: age1fxxzpnfse8nd9wz78ht3m0plrmraacf4cpga0pe8fm2tdnqcgy8q7qsyvp
- enc: | - enc: |
-----BEGIN AGE ENCRYPTED FILE----- -----BEGIN AGE ENCRYPTED FILE-----
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBEWFdCNGpPYU5pSXkzd3lG YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSA0MTQ4MzN0bDJNL1l1bmpX
TGU1WkhuUEhmaDdybzFUbVVaVmVqU0FyQXhBCnRmRngrbXlsTVVMSDV6dkhBS1lD ekxDUXRNa0JHWWltZFNGTVltTFdSSE82SVZvCmdUWUdja3JIajMzY09IMUE5elox
MmNUd0NjTE5ybFEweFhkWXdGbDNwTmcKLS0tIGU5Sk5hM3Y2bUVsS3pTaDZiaTBi MDdEakFJTmtkRWF2R1BGNkQ4U3grNWsKLS0tIE9hZUhkVGI1ZEpzdDhRU21EZm91
S21vU1ZVbjZ1SHE0WStab1VHQUNLRUUKBDW9hwI90Yn+B2mB7LUNTVxFGbwEFSw3 VnJNb1kyQ05MM0RJa1lLUEtjWWxkSTAKHVAKcGcWl6LncJALRBU9RKP7ot6C6GSE
LGGqgZdvnN5p6NMigBbJz+kSwOk1gQ/yo86HtRj8Ejllp7P7jRsfmw== 1iZtj1SNX6wzEWrhOEnV37aQ8bKZj6u+Y/q6/vJ4qiBs78y/drdIzA==
-----END AGE ENCRYPTED FILE----- -----END AGE ENCRYPTED FILE-----
recipient: age1px0h5l9zp2dww0m8fncrc82kfdmzplsfv2ltat7sna28xpg09pqqcl3s2k recipient: age190htw7prp4vln076dxjx3gxxaq06h0zl0te7cqgpx79vl3lhkaes8suy05
- enc: | - enc: |
-----BEGIN AGE ENCRYPTED FILE----- -----BEGIN AGE ENCRYPTED FILE-----
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBaMWdPVkV4WnNHNmU1dTZZ YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBNUEVWY2NpVEU4OTZPZTZR
ekVoVTJWY3laQ1ROM1NLUnlZY2FKTnJwNkFFCm8wVWtoNFpxMS91RkxzQWRhOWRs UzdKcktpUGJOcnJ0aEhkQlhvUWdUYkV2SkdzCm51eGJVeHJMcVRRRld0dFRCYUxr
V3k2STdvN1JRdjIrSHVFaFpSYmlmT00KLS0tIExnSmw2anNtdzVGMjRYdzdpcFRq TTN2WEhOVjRqV0FtQXowZWNTbkJneEUKLS0tIFpUazZpTUNWZUZBSFE0VDZZbkJu
alFrUTJVckpTVEVIcUxTK3pPbWNpUzAKSe3Z9V5u8+om2s+HqUcx6qXIaTmQUgOT SFVlUVhySnNqUENYOG9qUm5ZMDc1ZW8Kv0lY5dhnCEheM0sttfr4p7IL+EVog16T
zIrU8T0tLOHzdS1SAdh1Yb50yevN+P2+5TORdLjrtigpyzeYc2U/zQ== OapUdbuXL2l7t7URzHnvfG/nbOtJIjH8a0XFsWyJChtNXpF2d/vf2g==
-----END AGE ENCRYPTED FILE----- -----END AGE ENCRYPTED FILE-----
recipient: age1ufg390ydrmma849t9xfkxxl5xvdkk6mngnlzhmy7mvuaje8sgcmsmnq6l7 recipient: age1ukpqxzl44mnjpy5r96sfuc5sqzm47u4k8ujjh5qdgy6jvl9uqgpspymqfk
- enc: | - enc: |
-----BEGIN AGE ENCRYPTED FILE----- -----BEGIN AGE ENCRYPTED FILE-----
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBtWDI1Z2hVRG9ieHkyZnlP YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBvTzc2aExGTHFrc0ZtL3pq
SVQ3VHNsSlZxMjdIMUsxQWdIclNQQk5OUWhzCnZSMzFZUGkvYlNSRTJiT1ZMU0h3 YlZLbzd3MVZHZTB0VUQvSXZ4NGVsZk42c2hBCk1vbTk0Tnl3b01vbVZaMkJ5aldE
L0RIWkt4R1F5bUlWNVlQQklKc0tGa3cKLS0tIEJpTXNzZkFUOWp1ZnFocjJEVExa WmZBMGFjb2pjQXpYcnBxWmp0UUsrdXcKLS0tIG16SG9JbFdkbmVidCsxUnpBR3V3
dGlqRDRLd1BDYzYwSWZXaG1ueXhrSzAKdJ8yA+igZKyetNuQdoN2Woi2bl2I3Xj7 RzNOY3hIRWk4UXh6N3NrcjNSU3ZwWTgKaExY4U2s8E6ojljJ+4TU+YJhcLXyuVA1
dDBVAn8bvxx5ZG1eyDXXcHSh76xUp5ZPRlLOFCvRK9oOLWd8FntBmw== ROB70jQCjFvQOeo6thjQohSSUoPKhxSl1/nr4ZiGBO3/VskzihckKg==
-----END AGE ENCRYPTED FILE----- -----END AGE ENCRYPTED FILE-----
recipient: age16j42pdc5dr6wnj7xayhkqdj2rny9u68fcqejs50hqq42scssh4gsnrrnlt recipient: age15kh7akxlx7zn00tey79rq2g8lgs4j5y77rcnyfxrxap8ckfu0a9sqvtdhh
- enc: | lastmodified: "2026-07-19T02:30:40Z"
-----BEGIN AGE ENCRYPTED FILE----- mac: ENC[AES256_GCM,data:UiL3VMDF6rq4Nr87KspcDx434q3tfNXeb5pwH2O+4ssNQ6xzcYDdzXBnhAY3zLBsqPMKrvHBd4Ot/gEMcq3FMIVe7Q6p9yWKpep66KZ/yWEhAlwIVhD79Oj8VS+1CHKjf25zpRdhZorp04oeFQQd9VfjJB4EE/Q1aVbwTGlpIic=,iv:i/0conaFgFia+wzNTdUL6tlSTw35HTK3Ap1Sr5RGHf8=,tag:ULbz5FllShA/JjlSRdxA0g==,type:str]
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBkb3RrRDNqTGZGYlR1eTls
Y2FYQWxncmJBQ1ZiNWFpMTBMUmtlSnJzemhVCk5NTWZRcTJMWkIrTnVsbVAzcndB
MVp2NHJSazJjMVN5UXJHOWpnc0dLK1EKLS0tIGJlelUvbEZZZkFKd3BpdWx4VVNH
cHJSVmJ4ZWlVTlA3VXpWeE5DNkphcWsKTGpdWU/cE8vC/43lwmnwJDh4IPqHQoVV
yjnUdZnPDGyQHwwrydVgun2fdaAKH1zxAut8TZlH9pxD4ir0B69G+g==
-----END AGE ENCRYPTED FILE-----
recipient: age1k7d2du5mejsmv5rzavm4xwgpthqvcfsehduquv28nzs53zppa3kqngfxq2
- enc: |
-----BEGIN AGE ENCRYPTED FILE-----
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBaZ0tVT09Sd1hXVHZPd0tD
YmZKaStnUXg3cERMTjRmOVh2RmxFbG1FZ0hJCjNMWWJmaUFqTURoL2NvRnJIdzNW
OWFmSFAxWVRrdit6S2lpaFdERGNrOWcKLS0tIEJlVVFBSE5PK3k0dTRTRlZYMGZt
elcxVlltUnJVd256NHQ4TithUmFnSkkKkbFSRRktuO+tNxlfIEuJ27C7LWalHJMb
aCnjE9301XMceGB8SstPtGi5SJRSHD4WrJleeJbKXYMronfvWnYDSg==
-----END AGE ENCRYPTED FILE-----
recipient: age16kqfmvz4e23hmdlqresnyw69ej604s320mmd49h4hm3fhqchtgyqrws0k2
- enc: |
-----BEGIN AGE ENCRYPTED FILE-----
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBEa3E2ek9BcHZRRUR1cUpK
UEFERzc1ZEtrYjc4bEdEeDRnSDNkWHV6elhFCjljdnJLa2syR0UzYTRveGhPYVk1
cGNnWkNMNUVwbXFrOUZ1aE5UelFyMGsKLS0tIFNyQ29ieUpReW96OGhJaE4yU2Ni
Y3VWNUFYTU5LdHlmd0krK0NyQk9mMVEKfh7I/9+V9+0DLkzTf6n4sBKs+oZlMlTx
ar203b95cR/jjsUekF0NoZjj6MW1IZSV7BkaDoizVwrAxh7WS0vwqQ==
-----END AGE ENCRYPTED FILE-----
recipient: age1arhf2q45zw6wf2uevju4savp575x3m2tfvved5zzq3ay92ynua9s3cm92c
- enc: |
-----BEGIN AGE ENCRYPTED FILE-----
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBtSUcxeE5pM3NZVHlhcGNt
d1EwODM3Wm5BOGNybkNMdnM0SnJmTG5ScEhNCk5yZHhieVN4QnNPMkxzbUZ3SS9G
aG5WVENIbVBYRGI5ZzQ3QndUTG12NjgKLS0tIFF6ZFp5YWJ6UzlqMWRsb3pTS3l0
ODRZZG9pRkNxL1Mxa1dwUG9vdkJZZTgKYrjkSkjNQCU2wMIuZMvJnd/regVplzMi
3kPGRBMWzO1t4CSFtf2PMqv5AHw754I+vVNs5CpTXBuxVUcQDcGJCA==
-----END AGE ENCRYPTED FILE-----
recipient: age19mn8zrxl8zpps9yvrh4euquvygpp4fp8queg7xc6qhtnl4ng8c9qx02qwn
- enc: |
-----BEGIN AGE ENCRYPTED FILE-----
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBGNHY0Wm9oNk1BWFBiYTRm
OXBxNU9ETDNyTEhyMnBxZFRnZmFjVGVHN0RjCkdPNWpvRkVMTCtWdDhlR3Q0U3Nw
QUh6KzBVMU1ONXh2TTYrd0FaWlhCU28KLS0tIEhsaGxvcGFtaVF5T1NIc1NUOFBt
d2NGeW9YNVVxdUpYbFRoVnhUM2VTUGsK4adI9pgC2PipcAY4zXMRf9hPv5kilTvc
BsUNz0Qr4YdRfVfrPlPBzMCPOTDofTp6qBv+Gc8FE4hvwxtbjIcJ3A==
-----END AGE ENCRYPTED FILE-----
recipient: age1jlltcv5jcnm40z5k0q6hv053k2rqpqvemtuecdwn527uw8uqz4es3x7m68
- enc: |
-----BEGIN AGE ENCRYPTED FILE-----
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBBT25YN3BtMVZrVnhDazhO
OXpzOEd0Wk5KUnZ4a2JjRGI3U0RsMExXN1FZClM3UFoyeUIzY3FVNnlkTU1NSCtn
dXhNb3c4ZFlMTU5VTFc3blBUV0c5MkkKLS0tIHUzUzliL1NEUSt3cWx2d0tjRHcv
a1Nmbm1MZEptQjJuZ2ROWGUzN01zbDQKy3dPKE5oYwsTwE2vnhUi3auqJ/KBOBPX
vSKOROUuS/uEfsdo8NoVxuZ2RaJHetrbHAYPHFHOMgpe/X72pCcQNQ==
-----END AGE ENCRYPTED FILE-----
recipient: age1ug787sgt6st6k82fgkrug2lzltw4qsukrrqqs3w27ewwqj8rg4hsxcmylz
- enc: |
-----BEGIN AGE ENCRYPTED FILE-----
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSA2Q24xSkJaVDVUZlBWcW11
cWFOdjhXZEtuNkJtaUhQTEN3QkVUQnRRRjBZCnY4V1lCOTBNM3Mrc09naWNTQndn
VXU2ZWt5Zi9CbjBFa3YrejFHM041VFEKLS0tIHd5L0hRN00wdmdsSUhkdFpTcUNS
YUNiUGhVR2M2clFrOUkxNW5idkR2cnMKXO02RJh4ew22tf5GWH4lNdLlhf4bWyef
+hW9R4TGCJOnIO1xSGpBX0wJsM3oiW4qy/1tgfQw2ejVih5qUr7JRg==
-----END AGE ENCRYPTED FILE-----
recipient: age1zhfyuzlq40reuqlr34gf77852nhs3t6mqfzrqmas8z6sxk7tcfhsungrm0
- enc: |
-----BEGIN AGE ENCRYPTED FILE-----
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSA1SnhhMEpwOS9vUWVuT0Iz
bWFCUzE4djFyUzN0alRvOG94bGJldUhlb1ZnCkxHd1pUU3B4b0ZhRklnczUyRzBR
djF1bS9scjBlQXJEbmtiUFgzTndKQVEKLS0tIDRpUnhWYTM2U3lLcWZyVFkwcVNL
NFpldXQraURraWZNdUNqZFNwVFY0WlEKhV8IZYbBXKGb0x+2E5pJkoMavH2ox4qp
dK/XRlBpEG0SCUkKZ5sDegzz/HyqbxRF26jC3IOL1uR8A5nsnwC2Cw==
-----END AGE ENCRYPTED FILE-----
recipient: age1k73g8x47hs93wcv7qh92n3htz8pl295g49hyvlrf3570mts0hgys5g04d6
- enc: |
-----BEGIN AGE ENCRYPTED FILE-----
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBEWmIvVG1GWEZmbnpFU252
bFhOajA2dnJXNlFGMmF0TDRuR1YrYmZqc0hnCk85WGFUMGZXdElWU2tvdFV3eE1v
czEyTHBkZXFnSi9MVWFQVHJabEpUb1UKLS0tIC9TNGZiQ0xYR1dQWHl6NTAxbzlD
YmNhQkFVZTZidWFtbURqYnY2eGMxZ2sKaUPhe5mQ4QSyxuQMqLNI1jqfkCUBnBWc
HR5ck3H80/H0pMwzP5uaBnN/hD5L4CT2FcZBgUygbyNz4G3VgbjaMA==
-----END AGE ENCRYPTED FILE-----
recipient: age1fefy6dk8zn5c3edwmrs9vwx79quftnt784m628t9e34q3ft3cehqz8u72r
- enc: |
-----BEGIN AGE ENCRYPTED FILE-----
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBlU3BybGcyckNaT2dQR0ZO
dDVXa2NtNGFTS1RqZ0ViLzVJSk5PbWdYU3lVCkRZNTJ4MytYYVpON0JBQTh4SWRY
ay9xdkViektHNGRuSFNSeVoyMFdrNE0KLS0tIGJ1QW5RUjhWR2NlclR2dnh4SUQ5
SlJEWHhBTFI5Rm1rZVFjTGZlUWhZMGcKReNE3l+U35iutlQ07AZ+3fOFF4YdVbdx
bR/Sz3NqpqmZqBEmgjUjjjQI4h4Xturv4tT8/JUzEmVnEyDsqCbhHQ==
-----END AGE ENCRYPTED FILE-----
recipient: age16y0vfts5v2k20e2rj23qa5lc5gm96gm64uwsqsr0y688xl0ne46qcfr0sm
- enc: |
-----BEGIN AGE ENCRYPTED FILE-----
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBTMjhleFpqRGRBMGIrVmU0
aXcrclh1TmtIV1JZSlNWRXRRc2RCYThyUVMwCmZOdXptWnhvajYxeERlOEJFVTU4
WTY0R2NZSXplMDNNSzZ6eCtEeTN1NjgKLS0tIFIvYmY3VHM4T1RTZzljR2RnV0ND
Z0ZhY2JpL0psb0RJcEJNN3FHUHJVQlEKJdoXKdvm6fGA+C21lsxY4Zq+VVPt/6k4
vH0IirsPd9Cg33+tBzrySwkF+GVNtYLtekup2L61pFQ6/Y/necfQ2A==
-----END AGE ENCRYPTED FILE-----
recipient: age1tm3zj5lp3elw2j832az4nj9xxmhqd66ag3cqcv3vskvmd7qdmqmsdpdcn0
lastmodified: "2026-07-30T07:49:07Z"
mac: ENC[AES256_GCM,data:D5Y+zyLvs3gz8BWS3+lrra/tc1TsDIk7FykxSPiz0Er1jUB5LOe4IKircytMpwGqdLcVhUKgXYdRA+C0KAXaI9KpQp5qKfRDvX2jX24jveVNFPgbvxA2LidYasVMzA7ayTU0I+oirPcEtj+5VH/ahSjDz9YtarTDeHMpXvBZMLU=,iv:KcIGEs+61yS8Ul734fqsrC95iA82wIinf72DE2bLk/A=,tag:RzL48FqVcMro2W6VAgIxzQ==,type:str]
unencrypted_suffix: _unencrypted unencrypted_suffix: _unencrypted
version: 3.13.3 version: 3.13.1
-26
View File
@@ -1,26 +0,0 @@
{
"data": "ENC[AES256_GCM,data://9IEHIVCfHjyMNao5sCu2zNlZ/CaW+JyxVpGpD/vab2qvURunCUY7eMfSOyvOx/2WPXnWWlkoVJPCR1ec/yUg09EaSMfxrvqlu4UJI3Sxvu9xNuDszMwMMD/sCulJDiMWFNLp8qaYt7UGzexp4+GlGiqWDxk3ZEu/iLmSApzBrpciTF0lfehT4qblDovo9QXG2KDWFhCt2SwEKmHJ61Yl3pVAQnPLyTWaNhwWD/mYL76mIiDVKq7DJlvi9MBxzi3aYh/ttuHgRCvnCL0C9oUvOyT2cljwra0LXuOrum7FhPIXXboA7WTEbTHJm1LB5yLfxDrnWCrGxQzFQAwNixVHIcjuvjjvA/LifTvbj5FYM8x7an+DwXPfPhruFrDew1paAWsEGNNYXSmAlju2QLI/OwLHFFLuDyLnBKQ6RLtZbCEAOUKf2h4iZj7nH86eb/aGU=,iv:rj76+MJBCpiYyx2Ogut5UxJj3Gn+bygvu2mtuY5hFLA=,tag:yQmTvWVu8ZLug/7fo6RnwA==,type:str]",
"sops": {
"age": [
{
"enc": "-----BEGIN AGE ENCRYPTED FILE-----\nYWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBqcURBS2tIMGl1aWxLMHFQ\nS3ZGVTE0cnYzTmpnQ0daL1F3QnlpemRCMmtjClJiZzIrc2syN29sRXVoaXlPRUNF\nakw4RndmbEduTFg5ZVRKUTVwRWpGRjgKLS0tIHZ5ZFlyODlGWTJoNGpXR3RhY0ZH\nOGpPdDZQVmt6UWZXbHkxQTBoeW1JencKbsfH1V1lUj8mmHyLNj36VaRDgaBojcDU\ndoQWmSEXxjticJqdadbVKb3UABpvzAZxASCy81sa3wH0gT+7zLaNyQ==\n-----END AGE ENCRYPTED FILE-----\n",
"recipient": "age1njap586hc0q43kr03g6c8eqhdsmk8zcafkl3f83xwlc2gqhlmfgs4tmwad"
},
{
"enc": "-----BEGIN AGE ENCRYPTED FILE-----\nYWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBrbFdaMmRjcHgyV3Y0eWZL\neG1uMEtpQVUxQmNTU0Z3aC95Zi9uQlgvMEhRCmJiVlRTR2NocGlnbVU1UmlLVVA2\neERiMXpiQlVPNVhRU09XVGxrY0Ixa2sKLS0tIFRuOHdoelJxOXNRRHBJNnlhSE9j\nUUZHZmQzOVFwRmhFV3pvdnpRMTMraGcKRHBuSUpbHaEzH2tuSBE5MsLJDCuH3vUx\nO0jnDldCWkCw7Wvr/tQAkaDI8axZcYVDUkCEk+xqAdxDozLCPhEO6g==\n-----END AGE ENCRYPTED FILE-----\n",
"recipient": "age17e89ty6p0fw24daanen57wg8uald9s025t3wwxsw269svwpmgvrshfvfvt"
},
{
"enc": "-----BEGIN AGE ENCRYPTED FILE-----\nYWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBZMW5taXh2QkJkT0JjaVpq\nQlBjVFpFYUhXTHhOT3prbFcvUFB0YnRGUzMwCjhHT01lY3dPWDE2dDZWZnJza1hN\nS1AvZWIvdjZoYmRjWlliK1hiOEdrT00KLS0tIDljWWY0NlQveS9VR2hOSUNyTUtH\nK2gvbEVibmZSdktPemdEQ2p5U0Y4d2MKKitoTi2vbxJ41IoWMlj4vO91Ahpj0hHP\nrKCPyx7ws/IUMmKGyvDLpZ3pYqHD9jl5pLfB05Hh4Emhv4lA1qtwwQ==\n-----END AGE ENCRYPTED FILE-----\n",
"recipient": "age17jqc66x9yeshfgd9v78mj483r4zzarqdtuxtrkxe4x5mw679gphshd94th"
},
{
"enc": "-----BEGIN AGE ENCRYPTED FILE-----\nYWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSA3K2dVV1BHbDk2Z3FhNmdV\nZEN2Um94K0pma1k4b1V5ZUJxQ1hWU1grZjJNCi9iRVVqOVpEMmtBUi80NThlYldY\nTUtZOVErT2VSWk43NndpVzBlL1lQaTgKLS0tIEpCM3Q4NjM3N1lpUE56N04rTXN0\nZkZ6TkV1TU9OV3dpc25RNWRpVnprM00KItUzKBdShakOfX8Sr+k906nsvYPl8QLb\nge//1GA+ukGsaS9rcChOY89vFdm61JDmj1jXSJ0CN4wLMW9/eblZRw==\n-----END AGE ENCRYPTED FILE-----\n",
"recipient": "age1arhf2q45zw6wf2uevju4savp575x3m2tfvved5zzq3ay92ynua9s3cm92c"
}
],
"lastmodified": "2026-07-28T01:44:28Z",
"mac": "ENC[AES256_GCM,data:efFXIqbauOURR7lrVpK7kqRIPgYjgWfenBYoX7mUrQ/thFc+ApcAx58Fi1zg4biwIs7NarJAgDqAxZi+yKb2Sll8H/wlsEjWDU4iQlLJdIQw7wey9eDW8pgBLd+6E7FXrgn1Vh49dS5GXlC6clAYk1lCiB4NvfzPDy5Ktpl+Chs=,iv:+zCqj5I1MLJfRRiIrqgocYB49PZnlV45PUTH4gYlfrQ=,tag:WxY1sF4kFOTEzbSFcfJFbQ==,type:str]",
"version": "3.13.2"
}
}
-52
View File
@@ -1,52 +0,0 @@
wifi-password: ENC[AES256_GCM,data:SZQPtU6PYHbf9o83wq3KTupx,iv:FxO68Pn/+N58r/OPLfkAMYPFpP8TYxszMniFd/01E38=,tag:jwxaY6zDEcO5r9OWSfvUyw==,type:str]
sops:
age:
- enc: |
-----BEGIN AGE ENCRYPTED FILE-----
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSAzTFFlZHdGUzk0b1Zva1U5
V1UwREEwS05icWxYNEdKTE8rQ3lJMjM5YXdzCkw4OE4xWVUzVWZMaXg5OFo1UG8z
WkNwK20yb09rV2VVSENwNWUvTmhJNk0KLS0tIGVPWUhFS2RDcTNjY2JLaWxvcTZt
Z2RscURQdDVCMUdiVng2YWRMSlEvVVEKmyd3re6AaKn4gBjoT0x3e/zJznvJFYKn
ugKu3EsUX+gbailPmY1ss9+MVtpJFGZa2FiM0x1wSMKm6UJH0aPhVA==
-----END AGE ENCRYPTED FILE-----
recipient: age1njap586hc0q43kr03g6c8eqhdsmk8zcafkl3f83xwlc2gqhlmfgs4tmwad
- enc: |
-----BEGIN AGE ENCRYPTED FILE-----
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBtK2tud3pWSmd2aC95ZkV4
RkVvZXNrK09ZVmZsYWhqZ3JlaXBCd1NRWGhzCk5lelhSN2N6N2VhWEVXUjhLTjZH
V2VlMm5XdUtuK2d0MjIyRi9xeTh2ZXcKLS0tIDh5Rk1UT0dWblBkUXYzU3YzVGkw
OFk0bkJ5RXZpWE4rK05QUHlLQktYSmsK/HsVEIhBEIo3qqVWdUJEWnHZiKB3uHVH
R+nJGuXa2B/oUoxEwMP2YBHwwjLLiJCTYy+aQtiPdTrVq0YJ0HmC7w==
-----END AGE ENCRYPTED FILE-----
recipient: age1rrxqea6q6pn39sw8y5te63h2py8jgjl9v0jyper86w3ggtn67upqg3ah39
- enc: |
-----BEGIN AGE ENCRYPTED FILE-----
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBpVVVxWE1nUDBLTVFUaGJJ
Vm9EUldwRGcvbXhuT3JPd3N0WXo2S0gwRnh3CktabkdaSndaZUNSeUJGRzFKcjlH
b0x0SFdEQ0VqNWdQcEkxb0drVVJNcVUKLS0tICtyVGRqUmFtRHV1SlpXSVd3N2VN
elQrVFdhWTJ1R2pJbjdYa2w4NmRmc2sKJHqLYdNQJcna71KNhGF80iS1hIYG1U1w
I2kihepJsrmYr76ld9k+u1ZfnuIuJ1ozYsZothE+dr4pV0k8s6wkpg==
-----END AGE ENCRYPTED FILE-----
recipient: age1adur9g330gua4l6ndk8cqjg35qc8yxwgme6wrl2hpylcc7vxm38q05ejuy
- enc: |
-----BEGIN AGE ENCRYPTED FILE-----
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBSWXJxUEk5Mkp5eTZXamFR
SXZ4L1JTSGNaQjFiVXFGaEdvWkV2ZzBlVzJJCmd4NmUzZEdCbi9vdWdkVGZRL2Qv
cnVXa05xd2gzaXh1SnlMZmtueGpZMHMKLS0tIFowdnhFVGFheURQU1V2M0ZuM1Y0
NTJFWXBEYUxmOU9ROTkxWGhYUmlqMHMK3pexvc16BLKjh2meqtNm3M1zyLQ3eEsz
7C5WkdcSpCkW1lPDGtW7pEdAIL15StD4x7ut4MkSk0BjG1S+RpDbzA==
-----END AGE ENCRYPTED FILE-----
recipient: age1hrx8qj02fj2ea6d4g9vqhyj9hl7fppkjqfdx2l37py3h6pdkr95s8n8rvs
- enc: |
-----BEGIN AGE ENCRYPTED FILE-----
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBkR0o0SFh1L2xjZ1RjSC9K
ZnFyb1lJbnlMbENYN0VjSVQyU1F6RDBrL2hjCnRwTUp6Y1hJWldVeFdQZSttZ1Vw
L0dCS0ZROFArb0ppVzB5WmV4bWI5alEKLS0tIDh4VDV2TFhhaUp4L09jYm52UCsr
NGh5a3VMY2ZMZVBQbmRHeWsrQnZVWDQKR1UeSZ/EzZEXMqyjB1I2SHELv8Ha/tmI
kJKs2WT1RtDhAiTrbty3f4oVrXWSYKZr40kNiP/RLbUcH1s65ys/tQ==
-----END AGE ENCRYPTED FILE-----
recipient: age19mn8zrxl8zpps9yvrh4euquvygpp4fp8queg7xc6qhtnl4ng8c9qx02qwn
lastmodified: "2026-07-22T01:15:21Z"
mac: ENC[AES256_GCM,data:dC/oIqMUHkOh3AocOwP7Gc6XGH3L+nTqJfhFNts1DNbRXsopNIxVBtIz2pEhwnWSQrqPisDLmPHFBwRpGVn01u8w8IU1FKbAKC0J2nJXF8ozpInbjzDOmehqPWZG7yaKoq8cwAnp5XOk+IVO4l6tPxLxkExU5fT2ALuMq+sgOko=,iv:jaVyArpf6zMCFa6J9X1aQMGrmFq+W2CPZdWO6vVW68c=,tag:S+qF8/FkgHc4uW0e4ICmSQ==,type:str]
unencrypted_suffix: _unencrypted
version: 3.13.2
-22
View File
@@ -1,22 +0,0 @@
{
"data": "ENC[AES256_GCM,data:JwjmYg0qzoer+8/jv7KOfK/BxmpObHOn04GNsUFQd6hSFBWbQlRbODfluL54hnWQ4gsS9MHgwWh8nrLciBhye05kRQCqADB7Crm0CZYubNb65WzM/devh14jmerc3MYIp1M3VmDisML3x5IqhJkFMSqKM/VaNAC3f0otcVR8TTgZw/fJyvPlfqpPTjikGmo5xg++Yk7Cy8A1Dj70kYk10+EQjQ78jf4k/agLoaS+YvMmKXFk4EljtHg8Fe5H6Rn9nfMSSKGTZeAH3Riix2e+fIH+nWEZiPHC0UlPjJrB3SeBqqGRSUUJshVKHFIdKI1Tq2Ea+GDWrAyIRI0qO8e7WQ==,iv:mr39NxbZuYUDHDw0e+fJDshbt9R1hLgVqkUxUVX/+l0=,tag:u31xZRPfvH5j3Sw5U8/nqg==,type:str]",
"sops": {
"age": [
{
"enc": "-----BEGIN AGE ENCRYPTED FILE-----\nYWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBtVW5ENXlkUW83elc3ZnhE\nOEZ4MWIvQWZ4aXhuUmZGYXZNQnhJemsvOWlVCjIwNmFjS3pJOWEwRnhBOGNRamlM\ncXZid1RNYVBXOFZpMlpnU3ZnVE1scnMKLS0tIGFhWFZFd2JPTFNrV1VHSDVKVHRK\nbmhoeFh6YlUySU1ZRVhHbjZnM1IxMFkKo7aHkz2pEeV64m+OEkBZ2V1e+PUzoChu\nUc/Mnh37dNXSSJtg2KnocHdyzDGi1yQbA72xKTxx6QjYJWrAC/jurg==\n-----END AGE ENCRYPTED FILE-----\n",
"recipient": "age1njap586hc0q43kr03g6c8eqhdsmk8zcafkl3f83xwlc2gqhlmfgs4tmwad"
},
{
"enc": "-----BEGIN AGE ENCRYPTED FILE-----\nYWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBReWRLb09reXUxS0YyMUVV\nQlR2bEdYY08rK0RoR2pIelpLVEtYL0pqNUVNCklRaDNoSjBWT20xbk1UdmNqWHNt\nR3JoSlVwODVMUVBzMERUNktHMVViZm8KLS0tIHc0SXRIS1IreDF1VCthb0FBR0Na\ncGFRWTROUHF3clFtTmY1azEwNUpPNzQKsgCND/BZcMTgBTAcnHunQcT6LG1jNrtO\n+W7Yx7bFtFBajWnRYiNpUZPibQJlv5SE9os47WDH1gs86xftgV+uyA==\n-----END AGE ENCRYPTED FILE-----\n",
"recipient": "age1k73g8x47hs93wcv7qh92n3htz8pl295g49hyvlrf3570mts0hgys5g04d6"
},
{
"enc": "-----BEGIN AGE ENCRYPTED FILE-----\nYWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBGbllNTlhxNVVkd01GSUNi\nR0lxQWZoRXI5NWxoTExRUnBSb2FPR1BONnpzCkxCN3pNOXVrTW5GR2J0WFhZUE9r\ncm1QTThINDVVVlZEalJiS1RXcHJRS28KLS0tIHV6aGh5QjB0M1VPWDVMT1k3cTBh\nNTNGY25NV0tWSUl3UHlJSExFQVdpVDQK3ARj8xhFRYU6oqYxNQ5+Ryza86ALNUwX\n7yK/8ATnquYC8/ZIYUbgTPxzsocmXX9lVT0+ktIALqjtG8PgNTlOow==\n-----END AGE ENCRYPTED FILE-----\n",
"recipient": "age1fefy6dk8zn5c3edwmrs9vwx79quftnt784m628t9e34q3ft3cehqz8u72r"
}
],
"lastmodified": "2026-07-28T16:04:51Z",
"mac": "ENC[AES256_GCM,data:9X0dkAEGJiug16LDy/8//QOkHITTzt4zwTKiIU13tRgSRXRpt/7b27+3eDauscXDMamg5wbYo1YF0+VwLl21Ip4icnnwrZBHd6HWleP30HTgef8rmo21SstshnclZz6RH8SEGVDO/vjrMDChaZ3A2WD+nbcomw47DpK3cG3kBIk=,iv:G7CScSD5Z1Z9WMEMydyeGK/RD4W43xA0PlcpmTCxLOc=,tag:DZmby/xkxpinoJztZZRWwA==,type:str]",
"version": "3.13.3"
}
}
-18
View File
@@ -1,18 +0,0 @@
{
"data": "ENC[AES256_GCM,data:81JCCVaOeEYNyqTT3vXkFDDV1oSAlOrElGmvN+1Jy+U+dF6EaCSeFYT0U2i2BvUu9VAYEiY6NRAXwWUgZXeQXsI65eziB6d/8NKTr7UWb0eQxvhHbthgzxoCfdCvofZH49DdGuCEl8kU6hppSeM+wnDIIYvEgWljO4JQJr4/qnQRegMqJ33564+ZvhiYqrbfojcBYpTjVBzh+4aitqZEAYoFx4xhFAtdBfmAA1W6FwDvZLrX09bbdFT5jOqnSSkPunvTHiyWtyAfLgaNzG656F1U+3eRDv4eZQYOaLXXo9H4BaBMC5jyJFV1rDgl5s3diCuQggSyDbWUnwSS3Prcixj3dkrF8h8heNk/LDV5lGBtGEV3BQsEAlCPl01PIe3O5McfsRaR8IqwD7wQQ8ULyA56fHHAYJR1xe9bNGgdLryhqLDBFWi06Ztidi1MQXFLnNFJxoIhvANIAmt//TBWimkHnXidkg==,iv:WNChOcSG+QiZOMUi8jpb2gmOCLonM3zK7vcZCrXt4Yg=,tag:zq6Zvf3xu4kofaCechUpcg==,type:str]",
"sops": {
"age": [
{
"enc": "-----BEGIN AGE ENCRYPTED FILE-----\nYWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSAxMzVHK0tOazNoazU0RWZF\nT1h2N01NNktHRjVMQ0pEOWR4RDIxNVlQTWlZCkV2a2pQS0o0VFZWN1lyaklLWllp\nZXJvZEt6Zng1Q09xQTFkNnVtTDNNVDQKLS0tIFhnNW1sY09ZcWpVcGdmM1hrK0o3\nZ3VscjBXMVFEa0xtbVlZM0lMN1dHS2sKcYtCKhw6D1ax3Isf5Vk93cDteUEjx79j\n1fheqgOjytY3W9t8U2NOEpUP1hT7zWYswnaq+agm8nfEjf1fbwR9MQ==\n-----END AGE ENCRYPTED FILE-----\n",
"recipient": "age1njap586hc0q43kr03g6c8eqhdsmk8zcafkl3f83xwlc2gqhlmfgs4tmwad"
},
{
"enc": "-----BEGIN AGE ENCRYPTED FILE-----\nYWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSA4K3VSL1NrSm9rMm5WNTk1\nVEpudHhJUkJUeWxieEtKSUdCL2c1YkNHWVdNCk9EeGNoWHdLOVNUR2dRY2YrMVlJ\nblliN08yQW1xdXN5VTRjbm9vZUxZaDgKLS0tICtLaldjemJHSkZzY2R1MHlySEsy\ncXJ6UGdZMi9hTnNjRHRTVW5UMlNMQkkKWhLTte9gTpppGCmA0lf/FEYu5b/ZaGYT\njn9j9qj3Lets0gNj9qBVaHUkCoJ2TAGORdkhnLEdY/Tm/J1w3XfjNQ==\n-----END AGE ENCRYPTED FILE-----\n",
"recipient": "age16y0vfts5v2k20e2rj23qa5lc5gm96gm64uwsqsr0y688xl0ne46qcfr0sm"
}
],
"lastmodified": "2026-07-30T09:34:38Z",
"mac": "ENC[AES256_GCM,data:oC5OPZyZ4kEh0A8Mmwoi5oZ3+fBNvEDN3A+o3n099RdPREn6mcH5QC+XD9BXvWVxp9BzQEwRBmExlyZBEoRuwrB6DQy5Fn8TneEjQoCvZrudpbXIB66/EvORk1utpUsbAov3wn9A8tXUDAUJaRsY93/Pe1mTc8KzIOGv/QTMR9w=,iv:+p87aL7dKsbyAfCWxc0pecVB36yUUa+ltfzQZc4oovk=,tag:tNRRC056bqMToaHz1UrqQQ==,type:str]",
"version": "3.13.3"
}
}
-18
View File
@@ -1,18 +0,0 @@
{
"data": "ENC[AES256_GCM,data:6Is5g/gqFdQ9aTE6dpe80g7kSgDbhRr3yIwffv6tT5CLqsQLA6i+V5XPAqqhH0zbpXnT7k5Rc2tA2880MDa0eVoI+x7qDnhadfLKVgErWZNwQ4Nz9ukovf+lDfoKXn9AWNFxbiBckpCoxNOms3zMv6kqf1gbEwV7eRIoiq/i2xypke4nOJSmRH22dqFUldfQOOUL1D0H6RFQpNK+5O/Okb7URxIzdUgmq6/9zIsZTOlX5j1V7kzpx8U4P+YPyoPoRSLTOxpK/5gh+Gx1BQbXVTf8z5lm42tJxuQAt5kXqu0Tu8aXZzUUcMKf1oGVbo4TwRaQN+9tFgkhegVMSS/l2don3X3vWPimBcYHTfXfvvCh3krH3hoSmej8es2jCG8AesObDEmCyMNu9hdqhGGdCWspakrXVEuefCJ2Tq8fJfOn5XQhh4AX1wAf04Us0FJO4NjByNrgW8Bo1344wocAStxBxuD3rw==,iv:kkk5muigMC6iTIiTHKwXFreCuJGl7CWmwp1W7ILmq7U=,tag:W49PPiOISdFjk54wroD8ZQ==,type:str]",
"sops": {
"age": [
{
"enc": "-----BEGIN AGE ENCRYPTED FILE-----\nYWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBQN0NWRllWaEtPdTdlOHFz\nUXJBbG1IRXlPbnNzMEJhT0FuazJLYVZHWUdFCnJXdUtvV2RvUU05eDBaaVdxSlBw\nUVh1azM0T2dFZlpMMFhsU2tWbm4vSVUKLS0tIGNlUXMrbUlDYWpiRTQzV2svcXJH\nd1RIcmpGbkYvSlZwYkwxanBZajVTMUEKt4HCEEPjsSDKvp6XuSSrQjVXFQLQHwk0\n1Fi/HrkhhdkIO1f4DyTOOznvK5bc+Z4JKT5lrOkKQln92uAho2ECiw==\n-----END AGE ENCRYPTED FILE-----\n",
"recipient": "age1njap586hc0q43kr03g6c8eqhdsmk8zcafkl3f83xwlc2gqhlmfgs4tmwad"
},
{
"enc": "-----BEGIN AGE ENCRYPTED FILE-----\nYWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBock0vOGVLMC9HZ0NYckxr\nZWRwK0NhbkVaV1FVVndDNmRoNWhGVXlUOEdjCk50NFVtdmdxbEFuZE92YU1IUDlN\nbVJ6QnN1TEk5c1VmWDQwNXJReFdCSXMKLS0tIEl6TW1iV1NvTk5OeUkyOTRtT0ZZ\nR3NsV1lwa00rdHdTRUkwbzdrc1VaVkkK2ioVnzacNrQD6cpNOomKz9WfRRq+B3oK\nnabokq5rEOoNsSFNip5TBeDjo/34kOTXZCKXpFYwmBSoHCrkdFHvYg==\n-----END AGE ENCRYPTED FILE-----\n",
"recipient": "age1tm3zj5lp3elw2j832az4nj9xxmhqd66ag3cqcv3vskvmd7qdmqmsdpdcn0"
}
],
"lastmodified": "2026-07-30T09:34:40Z",
"mac": "ENC[AES256_GCM,data:QOvDbiPiQBPYBGg3BzJNZzLBv//UccDfOnoOVqnpdskOpPw88uQOcpK0mcF/unW8o+B0AJe4wzr8/XhlhtlLSRi6buES2uZr8pjgCqrgFMCUX8WP+4rxiSq9DkZtmLO1XBtBtjJy37PgFr8bwUw1p8fNdhnOEJBYR75MDC8JG38=,iv:YCCfUyncprAkIbvxWwMGGZZywNHLTrr5bYjBBdQp2OI=,tag:mHyHw11ok4jm79iYn6L74Q==,type:str]",
"version": "3.13.3"
}
}
-19
View File
@@ -1,19 +0,0 @@
{
"data": "ENC[AES256_GCM,data:VpxnzrdX5FY2gfDcZpqg1f2wqL6mAawcrdE5A9A3fQOKr4BWXmCT/hjQEtt8qG+yVOTApdG+vF8SKDj1AKZ25KsZ253WS7W5C4j8oiT/xHo0CtvVani4fZqN60lB/9381h7aUZ+lcBMPaQcNn35zoQqYthuCGrC9PH1qQgD2VVHr0v8J6kceXKp0KTJldtDsQxky6ROC9Zyc44wwkrFPgPss4/yGaRWWqiojDvK5fK8238hmR8HyfD6Tf5KhgbqCsSGS2g6yt1muUhmegaZwtpi/KpoqBmf1XFaObuHGRPKvblq4Fa0x/sSjpPogBZyNuvRKIwRhla0FkKOH1bX4gC9bzOlgiUUfVcRJT6Nr9jocYuhIv9wvIT4TujGbgHyorYaDSQLFPJoJEUa3fgpGq+AC6p1DJJE3b/wqUSngNbdOtAPE/imaEuprOfnbS7cRx+Ti/Yu+ZYZuAlDtLkSVdXi1hQ4fLQ==,iv:rq/BzePXa/w/Gqewz8JfM/NdU+x4ShiH1OuQT7wRyCQ=,tag:IcO3ddoj4M57CkndHEmNkQ==,type:str]",
"sops": {
"age": [
{
"enc": "-----BEGIN AGE ENCRYPTED FILE-----\nYWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBRdWFEUlRJMlZId2Yvc29F\nNnkwREkwQkllY25vZnRjV3Nob3BvNE1Zd21NCm56UVg5YUFGbEc2ak5MakxDbCs2\nTFVsRFhjVlRwMXdUbGpMKzQ1N2hBUUEKLS0tIEhmdm9hZEJqUEhRN0dsSk80bDY2\nY1JpajNOVFpEejhibjdKZlNHVjE1bGcKycyDfEBZ1WYc4EfAK2y5x/nKRqq7mnb4\nDlpR1Som4bSt7+B+OCZa48mC0Zv05HbP8PnoZdCe83swGNljvRONcQ==\n-----END AGE ENCRYPTED FILE-----\n",
"recipient": "age1njap586hc0q43kr03g6c8eqhdsmk8zcafkl3f83xwlc2gqhlmfgs4tmwad"
},
{
"enc": "-----BEGIN AGE ENCRYPTED FILE-----\nYWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSB0SGFVbXYycTE3amR3ZVhi\nVFJHVVhWNWVQQlJKZUJvZ2p4ditqdnYrWnd3ClZaaUVLUXFKaWlDdDQ4QW5SQUZY\nbG05c0RGNnZUenI1QTBlemVjT0szRncKLS0tIHBWTmNpd2FjWHc3MmR3cWt4SCtH\nWS8xZnVON1lGZVc4YzZFRmRBamdSVlUKdlQTlwiExTUVsiY05MXFG2/IQt1bZwKU\nlFDnuy6YCQzawBLQZuSUAp7WSGedupexlZKQKNUzPf+dis26XEYZoA==\n-----END AGE ENCRYPTED FILE-----\n",
"recipient": "age1k73g8x47hs93wcv7qh92n3htz8pl295g49hyvlrf3570mts0hgys5g04d6"
}
],
"lastmodified": "2026-07-28T11:53:06Z",
"mac": "ENC[AES256_GCM,data:PMsPSffQkxRmqT6HwFezIoT4WKJ9WDlAcflWynsFLtzbqVTREN0YGOJgMwLd0yBR2sbGBWogG5xAuKjO16aiWUVpCFPRvP97um351679ZoXIlrOL+GlFn4NKwoLH9/eCp8BteB9mv5huhEEN631lqCr+jtYO6yS0ULk6QS4+7/c=,iv:oYWV4J1JUuGryvqntVXkj/HOzRDRr9E2C1RKNvr2bT4=,tag:e4p+I40y1LrgdlK9tLRicA==,type:str]",
"unencrypted_suffix": "_unencrypted",
"version": "3.13.3"
}
}
-19
View File
@@ -1,19 +0,0 @@
{
"data": "ENC[AES256_GCM,data:Ea5AGwloFMyRgmQhJaDNTW19mpDvx0+dSc3ibENEXEniEIBV4Wv135mAwE6jQPbN8wF6f5ki3B5mDXHOMuhDQEqUXg6jcs4uEm2nwz7wAKYXEX4T8p0gUj7FPISiSQrw8O3jwzt4qF7AVcPsNhMtZLgrx9ssMtVmGct5do+E61Y7dFe0XbpJLynfzeiMKYvaBngTbHwQjv1mn9AMpX98apsp3tyzg8avLDR+WXRcMusds2mdov0jCoKuWPLG/srhMwjKOp1Z1Tu+aZCr9lHTT76mytpGIWGSbrP5l3UTp3oPLmy87Kwi/FJStZYdYikMnvmSo5qovePiYsLKVD74dS+d3oFGARyW4TILd9OQdXEjzdDRak5HcPv2/HhR638kvukVMdEdH0hpeJcqcV0h+BoLyIHe1e037iF1ZkCjuXE+A8LtlzRR0wqP202uzBIyg21ca4X5bRiHkkAOX4P3riaEcMgKfw==,iv:IXq4sHHsjnK6maVd3RRVZzhnxx1hOnxfsvYX8IKLn+4=,tag:jRZinAVLH8nGVf3gE4aSKA==,type:str]",
"sops": {
"age": [
{
"enc": "-----BEGIN AGE ENCRYPTED FILE-----\nYWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSByOTkrVVd0TXlBRzdkTktO\nalU3WlVOb2ltc1k5RG5sRE9EN2M2WUlnQVRnClFGZFYvSHdpVlhtNXExSXZMZnU3\ncnl6aHRzU3k2V3R3dU9RVDA3VWp3VkkKLS0tIHcwREM1dDREK0h0UFp0bU5JTW9Q\ncWdyanNMVjY0SnZUWUl3Y2R2SXNyc1UKI1FPIE66to19oQK2TwM5B4snGOVMLFfx\nl6JMEilDVSgy2RR/tSiovmo4NwOsv0hmmF1t34547rbZb+dB5KrMEQ==\n-----END AGE ENCRYPTED FILE-----\n",
"recipient": "age1njap586hc0q43kr03g6c8eqhdsmk8zcafkl3f83xwlc2gqhlmfgs4tmwad"
},
{
"enc": "-----BEGIN AGE ENCRYPTED FILE-----\nYWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBnbWp2UWphSnFibnA4N0ZV\nRHlNUnBNYUNzMzNZakFVc3JXT1d0R3Yyakg0CkowMFlOcEhPcHE5dXd4ZUd3cDAx\nbjQ2T1NqejNRMnFvbTdNRGFGdW1pM2sKLS0tIDhGb1JBQ0RFQ3NxbmpkdXg2WVF6\nS25uZW1hMVo4Q1JGdGNQWWRyWE84b0EK0FmFZB6sN7uZynhzYU932x461zSZIUWU\n8oyDEKNoUuPqpWK8RktgjWlKHtB0oLXC+SLAOJEhtnhWF+GSqloGfw==\n-----END AGE ENCRYPTED FILE-----\n",
"recipient": "age1fefy6dk8zn5c3edwmrs9vwx79quftnt784m628t9e34q3ft3cehqz8u72r"
}
],
"lastmodified": "2026-07-28T11:53:32Z",
"mac": "ENC[AES256_GCM,data:EhWEEIamG62xHI6SqO4tzdK6gcEXxUU/UGcF25m+X11kUVXzHd6yLcfiM2DvqcwGLmwmOOf3SpFLABlQUzTka4QRzTMFIAJi7x1rzQV4J0YIGITHk9rgvy0V50TI0loONBb2du+vDt8IlTBNvF9WMbNCki+fHPAjIemyoRBdubo=,iv:wfsx0H4fD3L7WeGzz57SzZtajBr/xTRDljSpz22DD7Y=,tag:naDyGH0WTKkqgd954Ya8jA==,type:str]",
"unencrypted_suffix": "_unencrypted",
"version": "3.13.3"
}
}
-26
View File
@@ -1,26 +0,0 @@
{
"data": "ENC[AES256_GCM,data:z2bazlADvtWkGcY4an7NgZgIDSDq8KmYpSq8tcBleW/33HlElMNhdmje8m8hd7WxgrVvCQDdlAfodBLj2mjbgSwBptlPX2zbwcmJdm0g0Ope+YyQZucIio7NnBaYa+OK8meWqmxO2WEeOk+hnNfNPbt8MkzF/FMMbP8xv9wSTKP9CDdmO7s8rdg8G1JxVfn9dyaSDUTuA2JGNtbh+osd4CtNCV/bc84MjXDZTgTfT0W/uv8VH3iPNT7BhQbU9FIY4dsFRYBf758VtJ+3mTBiQkR38IevJBvhFB7Wiqn118LSBgjdTKJnBwmGo6lMdo0LRYHEgUnb2ejhWMNie19x+LEAErDki0RBmjc2BbnOF908kUlW2fUPo5fz/hlfqlhDLS1uOxESKhfnKt7s5qzWiMX/nohpknVfyMbmu9j/M571z9sgFnJAylpW3NLp53lX8N9mINwRGXsyL/ijbNE=,iv:M/MNquAGN+lIIyVVvFRgKR6DKAhLclc/OY1HV5Xwoog=,tag:eEuURMErSqsoh+l9ZaHY9Q==,type:str]",
"sops": {
"age": [
{
"enc": "-----BEGIN AGE ENCRYPTED FILE-----\nYWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBqZlUxRG0zR3IzMlpDL3Vl\nNmRWQXUvc1NDek5wbWtKVEFTcTlxL3ZoWldNCmJVMkFtbDkxN1FGQ2VLTlhBUW5Q\nZ3VzL0UydUQ2YnlUMmtqa1Z0d0FSVFUKLS0tIGJFelc4Rm84T3ZHSGZDdnNUbUFM\nRXZHV3N3aVlSOXhUZk04UFRMWU5TbFUKhPU80PVYuDFUCxu1CA8+W8bqkr+Ne2fh\n+nBUPJbGxfN9TyD9tUC77AMbcL1R2L5x+SSAh0bEgSWE24/RjsuVKw==\n-----END AGE ENCRYPTED FILE-----\n",
"recipient": "age1njap586hc0q43kr03g6c8eqhdsmk8zcafkl3f83xwlc2gqhlmfgs4tmwad"
},
{
"enc": "-----BEGIN AGE ENCRYPTED FILE-----\nYWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSB3M1o3S2FibHV6N2tIOFEx\nRytJRDFIUU05azNDbWlpaDdFQXFRWlBUNnlFCnBwNVB0SEVwL3NWNS9aeG4xSUI1\nV3NOcUFxUUhaODYyWXdFMEhWeFpvancKLS0tIHY4NHFQRGJwUURQbWZ2cGlSdm9s\nOFJodGZkekk3UnBRRUwxb2YrQ1ZPcU0K0tCapb1hfVHFSNpmESXexYa5k9OE9Tha\n51QpU4mZHKGrnWPK/kyj7rHiz95TLsmwUdA8Q2hOiYNSxEaVBjYDQw==\n-----END AGE ENCRYPTED FILE-----\n",
"recipient": "age1jcx3yajjhghn8qh8za3yeu8nxykzlg3p4nrv03vnfvzl0mzayg2qmg940e"
},
{
"enc": "-----BEGIN AGE ENCRYPTED FILE-----\nYWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBkTmNrbUFoVFVGcytzQkk1\nMUxDdmJGL0xDcUNZaDFRQnUwNXoyMU90SEVvCi9EaTlVOGlMeUZtQzRJS1J2TTJY\neWdkMGNFMWZzQXVKOU5KcXVvalk1ekkKLS0tIDN2MlRQcG5aQnN3bUMrSDdvc09X\nM29hQ0pJRzk1amRHaS9mRlhrb2FMZG8KEWkSSP+MqGRU75qo7ctOCL7qHhyCM3l4\nL+ga1XOKxRsQkKQPozCi5Bm+k28W9hIDaC34Rw9difxICM9Z1rAbOA==\n-----END AGE ENCRYPTED FILE-----\n",
"recipient": "age1ufg390ydrmma849t9xfkxxl5xvdkk6mngnlzhmy7mvuaje8sgcmsmnq6l7"
},
{
"enc": "-----BEGIN AGE ENCRYPTED FILE-----\nYWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSA1TUtxVjJ2STFpU3RNdWpJ\nVjcwUHVORVN6L1pKVC9TeWt4cTNoR2pPWVRBCkR4QVJ1bHFUa2Z5aWw1bG0vQWVy\ncjNGRDl5REZzbzdZWTFMRkU1eHFTc1kKLS0tIHNLVDFiRkRKTHhrSEx2S0J1R3cx\nSkdEQWJya1ptTTRqUjJZT3FFaFhkd0kKgecHrnzW9+Eb+b7c0z1yR+Y0Czsr9Kjh\nx1UUOfleHntJUCs8oYcOogKknEnhBJJoJ5oe/gWruRSwle1fs3cBqQ==\n-----END AGE ENCRYPTED FILE-----\n",
"recipient": "age1jlltcv5jcnm40z5k0q6hv053k2rqpqvemtuecdwn527uw8uqz4es3x7m68"
}
],
"lastmodified": "2026-07-27T21:22:58Z",
"mac": "ENC[AES256_GCM,data:VG2ygV4X6yMxGlJgq3sN4GAzgVqiD06noyMxB2yb/FCmiFVYZAH/9LqF//G7MTInfIjFrTDyoHAny+m/ztUrJkHXI1Fzg0U8R/zYyQ6wj/GguL0gyhA70uriQhvHsRHefHPa7Km61Blo9cME6CBJWejtRvh1IMw9itAioVrSIZE=,iv:zjxNG6IYxkcNqLk08NADoDiIbS6at86y136SBzadW1U=,tag:sj4ZRR7ixU8h7aOUJmTCYg==,type:str]",
"version": "3.13.2"
}
}
+21 -29
View File
@@ -1,43 +1,35 @@
beszel-token: ENC[AES256_GCM,data:meuzUP/6wCssJDVTgbC0XwiLZPMGyDl55HEIiON9xOXCD9k6,iv:TDqWcp+8Mxd8wN09r5otQRQXq3XTeQphaTWxvvuLTAs=,tag:cRPZQGlwB/dTguBAheWPQg==,type:str]
cache-priv-key: ENC[AES256_GCM,data:6vQKIf7eS0WNL2Eptoi4VWr18SRMZfN/H/aFUUtXdMYQY5LLyBp2EHRKqZcGFuh1nZhUdAxUztq/CVXx+QFxKW+ElHxCxUSp0QqI1fdSkBkKZb8hlit5SoX9JtLzZGg0HBNM3nJu,iv:0J+xmrPJhInHhFR/c41ACjuTfaIoMkQFSfbL2KkgFa8=,tag:f4s9Szs5oprVVRSyXaX48A==,type:str] cache-priv-key: ENC[AES256_GCM,data:6vQKIf7eS0WNL2Eptoi4VWr18SRMZfN/H/aFUUtXdMYQY5LLyBp2EHRKqZcGFuh1nZhUdAxUztq/CVXx+QFxKW+ElHxCxUSp0QqI1fdSkBkKZb8hlit5SoX9JtLzZGg0HBNM3nJu,iv:0J+xmrPJhInHhFR/c41ACjuTfaIoMkQFSfbL2KkgFa8=,tag:f4s9Szs5oprVVRSyXaX48A==,type:str]
sops: sops:
age: age:
- enc: | - enc: |
-----BEGIN AGE ENCRYPTED FILE----- -----BEGIN AGE ENCRYPTED FILE-----
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBwMklZTFJuR3NYbkFvV0l0 YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBPWE1HTUhiSUp5ZEUwWEpI
eUhMWU4vMHpnN1NKVThuVVdiOFpkZW9rTjBVCjJabWkvOFpOSm1hdEdlZTYxc3BP bGpkZlBIMUo5ZlYrQ09SN3Q1a0ZkQ0ZnOEhVCnJPNEZQenVWWGZiODlzQzNEc1Zq
WkRURDFEQzMvRFlka1VnaU9zRjhiSEkKLS0tIHZXRG9GaE5iVjg3M3I0SzhtN0JP c3l4OWZJTElJc2Y2UE15OGtEUzhyY1EKLS0tIHNJUStyWnlQWjZBbEZjQ3UwdUpz
UlJVUzRzN3NEZWxXZHJ6RW1oYWwxS2MKonnhq7YDg4v93PZtoaLANDy8mdRCenjo ZndoUDR6bisrNGJCUHk3TGI4bTZaMFUK87fFsm9ne9s+PK2pcwtrDjqyGBss2r2E
FjRUzozMLpgWBll4DwRWikejsrRofRBwlcsiIdrBr90f8Lr9pHdQ/Q== 8lhqoeiKZ2j96z8kP/7ChzovwTCmqdcmAQuyNQD+ZAFijseipSvfbQ==
-----END AGE ENCRYPTED FILE----- -----END AGE ENCRYPTED FILE-----
recipient: age1njap586hc0q43kr03g6c8eqhdsmk8zcafkl3f83xwlc2gqhlmfgs4tmwad recipient: age1njap586hc0q43kr03g6c8eqhdsmk8zcafkl3f83xwlc2gqhlmfgs4tmwad
- enc: | - enc: |
-----BEGIN AGE ENCRYPTED FILE----- -----BEGIN AGE ENCRYPTED FILE-----
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBsWDJQYzM5Z1JRT3FlVkJy YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBUYi9SRFFGV3Z6cFd2Znk5
TkVaSXdzanYwbnpOM1d2RFh0c2NaSUFmdW1ZCit4SVFxSS9HNXhVM2gveERGVS81 b2FLbWtzTllJMDBUaGk0NTViOTNBa2hQclVZClZHKzNhbGVjQUJhWkFWdTFBMG5a
NVFQaUV2RGR5Q3kyUTQ5eEpDVXJVMWcKLS0tIHdHSy9WM0o3Q0o1THhXZW11K2Vp cUFJdUdyVG5HQXJRRnJId3hqRTN2cXMKLS0tIEFMRjh3WE1ON0U2TTNTZ3hxMTR4
NTNtSGM2UDFuWDBEdS9tbTY1ZWt6UlUK8z5qoi0kGn0ES3m9khummuU51rkR0Sb9 ZGRlemlIbDZKeExmVHROc3Eyak5DdzQKaLwIVDi6BN4cxpVxJoqTYvJETPOp4thc
TWT92+BWvPdNrAsDFjv0fgpUKyTMzN72EzHZKAJCIM3crUG9I0tX2g== l9uVMvIGuEsEZgDsvShw1dYLljd+uGy/A+dXbcxIUCP/mmPkwmd1Pw==
-----END AGE ENCRYPTED FILE----- -----END AGE ENCRYPTED FILE-----
recipient: age1jcx3yajjhghn8qh8za3yeu8nxykzlg3p4nrv03vnfvzl0mzayg2qmg940e recipient: age120le4a5l8dh3lyfgvmj3d9ksmej6ajs5mer5y7r0vfg3x9fn69dqf8xgzu
- enc: | - enc: |
-----BEGIN AGE ENCRYPTED FILE----- -----BEGIN AGE ENCRYPTED FILE-----
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBpY3o4SEpwZEdFQmVnOTVV YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSArOWovSW9DeFpxL0VDUDQ3
d2NqL0VudHM4VjdDK1N4dWdlS0lGR1V6a0c0CmJUTHlsbEMwaEU0Kzdaa21lRFRm SHUwTzJVZUtPV01ZRkdCUXZGL2lTRCtCNFNnCjBSNExqRW5mTEN5SFVucHJHSzZt
RTRnTUlGUG9GQVB4U2pzTHdRY09udkEKLS0tIGlFQW9Wd0N5WFg2WUpCQzhWUm5v cDlNc3BjY3M1c1k1Z2tkVEg4R1pacGsKLS0tIEFWbHNKZW0vbVh1Y2VhQW93OUwx
aG1VVWV5ajBmc2o4ckgyQWpWaFVxVmcKisAw40bGQBRH+u6uNygfYpfb7iEgfEHj MWV0eW9sOXdQd0l2ZjlWOEVVc1dwcTgK2s4p9xoNkawH2OkGsl80bNIo3ad5vn4W
E+g9n2WeVH8kUzD2O1o6VAu4m/SVuI4+IQ77j9GEWmlI/wgp8wKXgQ== Z2w+jwppSoUmbQnD3WFbLmSSxmuobmU8HILwElv6SZu+KE3aspF6XA==
-----END AGE ENCRYPTED FILE----- -----END AGE ENCRYPTED FILE-----
recipient: age1ufg390ydrmma849t9xfkxxl5xvdkk6mngnlzhmy7mvuaje8sgcmsmnq6l7 recipient: age1xjst4frdh0th6q8m7p7u9g5af7ty5jqeum0p6z8a52a9q7st7ewqw8yl9j
- enc: | lastmodified: "2026-07-19T23:30:21Z"
-----BEGIN AGE ENCRYPTED FILE----- mac: ENC[AES256_GCM,data:kLGE2xawQT7mx+sfw68hmGk5nCEGiEjZrqTEl9B1dtQmTrMwmoVr/1RISi4LfJrwxy31mDgff4lcIL4wIJuM373uk3X8j4RNyYQNTfKEkORT6r8NHeepNs267O77pKGd7OmcM4MT/BqOnB8ELS7Wlf2ect7CAlvUUVyc8icxgZE=,iv:EYLDsHYHZ1XOQXafOTqHHWpk/OBNq/R6IJnOBYV33E4=,tag:thxrCPC5oGvDjhK7Dz87YA==,type:str]
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSB1elpNZXVJbmJWSU85aUV6
NG9YTWVBeWxiRHUveStTQnl4NC8rT2VNdHpNCldxZVJYNVhUR2V3Vk41VnJxenVT
WS9rRWlKcGpVdVJPakkwUTY2a2xQTlkKLS0tIFhFVTRucFNuS0pMK0FNRk1ndnFO
SlVvakczaktUa2VLY3RLYUdVRzFyamcKIhctg0mbYL7OE08dRwj5wMu2x+O8/BMu
IqA477+noQ/Rjrszb2hEvxID7keogcDUWMWQzQvdMc22+3mvAr9qIg==
-----END AGE ENCRYPTED FILE-----
recipient: age1jlltcv5jcnm40z5k0q6hv053k2rqpqvemtuecdwn527uw8uqz4es3x7m68
lastmodified: "2026-07-29T01:59:11Z"
mac: ENC[AES256_GCM,data:/nbcfause6G6F8IvMoyPZtkWS1XRLAivhwTFu6y5P0Mm0eCcO6M7/rgioN9dngKzPXrCUl3Dx/EvhrrWKe2/Saq9WEOFgvS2V05pTRbQgYjuugVzW2paPq1fgmoDYNjHz2yFYWAovbIFtjVxMR9tmcASMTe6r/FocvMXbFCJh7Y=,iv:n3KoaPPtDhWK8mxJNJRk7WPVUUNR59nZSA8JyysWNDc=,tag:tYhkNXsj2ohVjHkvxife2w==,type:str]
unencrypted_suffix: _unencrypted unencrypted_suffix: _unencrypted
version: 3.13.3 version: 3.13.1
-30
View File
@@ -1,30 +0,0 @@
{
"data": "ENC[AES256_GCM,data:apiijrtrqd77CTizITg0R35BfCi8PBnufpxIyC+hLYqwoBzP//3z/yjFyHPLG98m/c/qywoi3Kn+zsaTT7MjP++9OMhhX94YKlSHV1/cHB76OkwsNc+ClqWxl6vpaFX29Qvh3gFX9c/NR3xvYQutYwrIrQ9NR+t/M52IMC8hvtR1LQy0ak3VIuXJlSnG2r4kF2Ym1iP7phjuq39Gd245Axzw8OB7yGvOjNxSdTPxW/qL0fMlzNcMrjr9hw15WlqnZfWPOsB1+gZjHXpGfPD5BCbAAMoTRJd75vhKKXP/ERhIffewuuH2x/QHfSFvXVB3QyhBQMxd2b8QEEE5cjvcExOST3tkj6QARkzoUpRT7AE3jhl3XZ0uA2qu9SwyrSvbr0tBRKxCdK0g2E2/hqwcK/Tck5GB1eKb4aN+UkqxOblNDH+B1RfDoyNAuN+KEg==,iv:0p+ScrKpP4kQvO52gBAlwAis6oAzZ0EHFnU74hYPrn4=,tag:ON7qOjztF52xsJWAou7ogg==,type:str]",
"sops": {
"age": [
{
"enc": "-----BEGIN AGE ENCRYPTED FILE-----\nYWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBlYXB3cVAzb2xEZ2pGa1RJ\nbS9ZVTc4Ums5eUZJUjEvd1g1aGVyNUNramowCnptZXFOZVB3MFRFcUtzSXBEZk1B\neEtKcDdLS0h0b1h3VjRjRXRvV3V5V3MKLS0tIHBDemkyUnV6ZXhTeE5VOVVOMlky\nWWMzVGVzZlAxMjZYUGpQUCs5QmxiYkkKcuBshCgWX4TwfVlQ5lHikzvwWdLEXWD1\n/uSiy0J6yMSiu8u6cg2SxeFrlKJ3j47dDlT6WHCxS0PfeEA0bJb3LA==\n-----END AGE ENCRYPTED FILE-----\n",
"recipient": "age1njap586hc0q43kr03g6c8eqhdsmk8zcafkl3f83xwlc2gqhlmfgs4tmwad"
},
{
"enc": "-----BEGIN AGE ENCRYPTED FILE-----\nYWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBGa2VTc2NWdkFRNUJxelVR\nWFk4RWxoelYzNHo1UFVhU2ZkLzEySlRWN2xNCmNmcmJod2crL3NMRlVsSmpmVkU2\nMjlXMktjc3piUVNhUXlTdnVGTWJkUTQKLS0tIGQrMUxrNDlNTkRCSUtFWkxRdXgw\nRlV4ZmtYSGhPQU84eWtiQXVqTmxUK3cKk5fn72UZPH68t5ZappfAhZJwzpLkfKmT\ny9TbUPIr4Pbrexau6YiH43QIbDQFdwYPfkBjGkd57zCg8AVo1+MBRw==\n-----END AGE ENCRYPTED FILE-----\n",
"recipient": "age1adur9g330gua4l6ndk8cqjg35qc8yxwgme6wrl2hpylcc7vxm38q05ejuy"
},
{
"enc": "-----BEGIN AGE ENCRYPTED FILE-----\nYWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSArNWkwQXFWT1RpaDNvbzYy\nQWc1aHhmNHFEUXVsQjZqb0EzM0wrV0dwN1hnCjFsUFJiT3REK05uSGRWTEw2SFE4\nY1FleE1XVjhBbndiMmZxTWNTYmhYeVEKLS0tIEJiZzJvS3BsYzB3cHIxa2k5N1Ro\nenFFZDVaODNnVGdBZTBOYWJwRjQzc1kKlXJgee8wTSN4Beq4P0t9cYbk0BWHCseQ\nyaWpiPT9aZBEGLFmuEd3zKABc8lrilX/ySTmOG49vRg6CPmr7cT0Wg==\n-----END AGE ENCRYPTED FILE-----\n",
"recipient": "age1rrxqea6q6pn39sw8y5te63h2py8jgjl9v0jyper86w3ggtn67upqg3ah39"
},
{
"enc": "-----BEGIN AGE ENCRYPTED FILE-----\nYWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBEa0JYenFMNmNzWnVmcXdz\ndG90ZUZ0WWlIU0FCZG9OaWpBM3ZDWnFhZFhNCjhuV1FTOTJ2WVJGa2RuNVV2MjR0\naVNXa3diaWxWUlJtdkNOQXZ2R2NsQkUKLS0tIDcyYXh3N3B2QmNiK3dzemFFMGV1\nNTZpTk5yNGV5YVo3cGswK0NLWFQxQlEKIe0N5OxooWXzt1cUViBmjihmGEe3G6/f\nkz2/IscnG78ZvNgYKjdoG1jlsyje/3zI4C8aWXLq2DnIyxUyAhPgsQ==\n-----END AGE ENCRYPTED FILE-----\n",
"recipient": "age19mn8zrxl8zpps9yvrh4euquvygpp4fp8queg7xc6qhtnl4ng8c9qx02qwn"
},
{
"enc": "-----BEGIN AGE ENCRYPTED FILE-----\nYWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSB2b0NVVm9QNm9waUUzcXBi\nWnZWU0JETjZRZmprZVRUL1h6ZHB4cXkrdm5rCm9GZ0VnTXB3S1BYSmlGWFJVcDhJ\naUl3RjR0ak9BRmQvVk1GRnQxNmtYM00KLS0tIFlTU1p2OHhWUGlOWngwbE56NEhF\nRW5QSkVVUWZpdDZXWEIxZ1BkbzVwclEK2P25nBgf8255vaKW/+T97aNTecRgNjLu\nedIUiPdXbFATCe3v/YRo6sqzFwIsvM6Bl9yHh/SXo6Ftc7eWZZd8zQ==\n-----END AGE ENCRYPTED FILE-----\n",
"recipient": "age1hrx8qj02fj2ea6d4g9vqhyj9hl7fppkjqfdx2l37py3h6pdkr95s8n8rvs"
}
],
"lastmodified": "2026-07-28T01:44:24Z",
"mac": "ENC[AES256_GCM,data:J0D8bEs5mHLraLS6TvYuCgfiNU1xKM2Yfb5Y0f/q/4wM4LzXufNzv3+SWDHumTe328U8UnNXLqjNHEKL0bZi0coxpU5hVM+BvPcmqD72vscETzbQ2hnU05sfW+XjfZhcN8/ke0bpLt7nP0crD5hsZv3esV1E2UWvzjEiYtWzFHY=,iv:lcmXYG2H469UKBYDndWKMO+GP0mSGLztenm+kBaUdYI=,tag:ujiXbLM9CUsuoFwQQWI84Q==,type:str]",
"version": "3.13.2"
}
}
-22
View File
@@ -1,22 +0,0 @@
{
"data": "ENC[AES256_GCM,data:Q++XWxg9tvY7ugT8+8FWCC5jgOfQ1+LYLsnqN0/WGxTf5XT2VmoXvszcE/ow2BUOBG3qBTXS3OHYFPwmj1GgaBHB8rpdXX6+LveSBE2gmx1VR+NUDTxy+0/DBq411MK9n/J9eIYcybdFIE11biFSAob9EgfxBF5roCDXIPDPyVCSe6LyhNvqYPnGQsfHCbejSewLTcRQEiguP9BX96CpMPIpmaB9fHN25t5RWCgMI7MtacrRxRsyKg44+2FZstXdZrp2Wv9u86BxqdAFqtZE8qpPeGdrdzluZx9jhnw0wZPzHdKg5wS7/UrLCb0UxIQiDxDMDuuBuLGkKXfAhb6SGZU41wWAWYk548iGRUaG+79BO4HhRZNObOFvfsjpMXEfcH/Vbv/wHVY4OpJUPe/ZWK4wpL9YrY4nT0t62HH7MirOy1uhwLE50h2+6dHKj6ur5G7thkSqgzVWXQ==,iv:zaRBwS+gfXLhH30havn6Q2+oPWuLV3qBfbOj00kewlQ=,tag:Vr5iEQ2u+9YNahryhgzwSw==,type:str]",
"sops": {
"age": [
{
"enc": "-----BEGIN AGE ENCRYPTED FILE-----\nYWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSArTGJnaVhleUtsWnlITE1s\nNmRCaW1QWTFNSG5LRFdmb1lzRlV5NWl3YTBNCmF6eUN6RkJBZzc0MmJJa0dKTW01\ncVh4K1VLR2lURUpKQXpxNGpQNnpSUlUKLS0tIFJONnJFWkNCR3pqalRsUW9POVBj\nQ0pFQ3ltKzBETTVXTW5sV1ppWTFJc1kKzxUboNZO+Nwn2eTWy11VP9w1pRswCHaJ\nE2dYU0oUOClVzc0oSuIJxraG6TPj1N4WGC24gS+UmpkmSuCiOeZBsw==\n-----END AGE ENCRYPTED FILE-----\n",
"recipient": "age1njap586hc0q43kr03g6c8eqhdsmk8zcafkl3f83xwlc2gqhlmfgs4tmwad"
},
{
"enc": "-----BEGIN AGE ENCRYPTED FILE-----\nYWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBDSmxNR1I3dUFKT0xUVG9h\nWkRIakNVeWRQOEN3blNRVjZlWHF4K2NRa0hBClRiOXlmTTJ4T2JTUEw2c1l0R2N2\nMnRwdDA5bEZlQWJRTm9vUmNKclBSU1EKLS0tIFcyeDFjbTZyVEVDUjN1VzU1VHly\nNWNDMW9rTXY2bHNWYVR0SmtMckovUzQKhTWr6yFVW9am3okCiIswwqR5+/p9OLmB\nWCgPtwoFaBt1RjUXPK4/eS4LlucR2K6V/mNMn4xVsnkIl193U9632g==\n-----END AGE ENCRYPTED FILE-----\n",
"recipient": "age16j42pdc5dr6wnj7xayhkqdj2rny9u68fcqejs50hqq42scssh4gsnrrnlt"
},
{
"enc": "-----BEGIN AGE ENCRYPTED FILE-----\nYWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSA2ME9CbmVGcENRWksyL0pW\nOEJyUGhpbGMwWlBhVXBSeXQ2MW1EWnFuR0E4CnMwU0pjdk1YMzF5ZEhTVFlBaHZq\nam94UWVGbjhZSEx6VHBmem9JRWgwYzAKLS0tIHFJZWRyRjRHNzhXdDJSYWN3bDlR\nYVp3eGJWWkh3Y09ZWElyclZQN1ZSVFkKjR32//EcFAdMjVlNgky5zvVkwXwEN68D\nrkTuHKjiO5aV7yAQGPkdNw0UM0oRGF0u4YF3oOUcZfSvnKgDeoi2Zw==\n-----END AGE ENCRYPTED FILE-----\n",
"recipient": "age1ug787sgt6st6k82fgkrug2lzltw4qsukrrqqs3w27ewwqj8rg4hsxcmylz"
}
],
"lastmodified": "2026-07-28T00:27:46Z",
"mac": "ENC[AES256_GCM,data:AbJIHYcFpeanQsJ3x7RPL9Yjlg5BJgkepKax0fL9L/PpA03Antab93iUNG95Mp6k/duovp8Jm445lbuppDZq1dh9ij/deBa8GbzJ50wwEe9zMc3EwRKScpqZEhRPF7KlJsIjsHJyd8NkcI5ji49XkHb4Ae1//8zG5HpVgy+3b04=,iv:uxQCbMwMIfP5S1dbsvIx3F79YWEguxwox8T0YZvUBdc=,tag:3utjmBGDmPc8q4JjaXvCkA==,type:str]",
"version": "3.13.2"
}
}

Some files were not shown because too many files have changed in this diff Show More