Archived
Compare commits
476
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
2741451642 | ||
|
|
9f1fed26c2 | ||
|
|
3b6ac30946 | ||
|
|
cee33aa2d8 | ||
|
|
ac9099a1f2 | ||
|
|
c96752e5d0 | ||
|
|
43314b6aa3 | ||
|
|
db84a0bcc1 | ||
|
|
419b0f706b | ||
|
|
89e1c5dc9a | ||
|
|
77db51a88f | ||
|
|
ad0b4a14e6 | ||
|
|
c9dbb03e31 | ||
|
|
b6ed0c6007 | ||
|
|
78fdf3d94c | ||
|
|
f85c65870f | ||
|
|
97c9353fa8 | ||
|
|
cded77919d | ||
|
|
71381ad990 | ||
|
|
857c81f64a | ||
|
|
334ffbda09 | ||
|
|
620a344d78 | ||
|
|
24e5c9fe0c | ||
|
|
4ed2db906a | ||
|
|
944af9597d | ||
|
|
4f0b07031d | ||
|
|
8c86144694 | ||
|
|
3c28d48bc7 | ||
|
|
e9a4913069 | ||
|
|
bf4836efac | ||
|
|
5eea38d3ca | ||
|
|
6f38ad67e0 | ||
|
|
8d43b7039c | ||
|
|
8e58226d2b | ||
|
|
a5fb0404e2 | ||
|
|
4ba9b141fe | ||
|
|
f3c2965f78 | ||
|
|
b9d3b51028 | ||
|
|
539bdf9833 | ||
|
|
a5308a7ee5 | ||
|
|
dc6c37ed2e | ||
|
|
b5e61d62bd | ||
|
|
ced1657407 | ||
|
|
7b4794211d | ||
|
|
fdf41c659c | ||
|
|
0ef8259225 | ||
|
|
629c1457a9 | ||
|
|
7ba603c005 | ||
|
|
decf3ddff9 | ||
|
|
4490dfab7d | ||
|
|
fa163b613d | ||
|
|
15bc5bd369 | ||
|
|
5a4fbbf7ad | ||
|
|
e720bbb018 | ||
|
|
40cdf724b4 | ||
|
|
aaa193645b | ||
|
|
f22ff7db79 | ||
|
|
0da3c3070f | ||
|
|
e1551feefd | ||
|
|
b4bc30cb2c | ||
|
|
5c8d55bd78 | ||
|
|
999c9a3151 | ||
|
|
1e0fff1b26 | ||
|
|
24c6469f10 | ||
|
|
c9458ac8a6 | ||
|
|
99ba0ed52b | ||
|
|
1e4072029e | ||
|
|
46649fc7e0 | ||
|
|
4cbcc3beb9 | ||
|
|
35f696ccd5 | ||
|
|
e18b605706 | ||
|
|
ee2451d87c | ||
|
|
71a6c4738c | ||
|
|
793a2b5924 | ||
|
|
98b5429fb9 | ||
|
|
f658bdbabc | ||
|
|
d7e63cd80e | ||
|
|
b88ea49880 | ||
|
|
c7268ade8e | ||
|
|
367158548e | ||
|
|
585126beee | ||
|
|
ceb491b18d | ||
|
|
6d16eaff89 | ||
|
|
e085d4707c | ||
|
|
c8d4440787 | ||
|
|
e7531b276e | ||
|
|
49a10d7cc5 | ||
|
|
ead4f55805 | ||
|
|
cac5ec45cc | ||
|
|
ec90753a09 | ||
|
|
df1ddee735 | ||
|
|
8e8261be31 | ||
|
|
5e9541741f | ||
|
|
07543d7d56 | ||
|
|
a7c4a24fc3 | ||
|
|
a98955a8da | ||
|
|
7d60741e73 | ||
|
|
427be2b287 | ||
|
|
684351b89b | ||
|
|
327ff0b44d | ||
|
|
43b4cc6aa6 | ||
|
|
dd984019a1 | ||
|
|
cb8a51b2fd | ||
|
|
18ab0ff254 | ||
|
|
5c5f22f84a | ||
|
|
094eaa752b | ||
|
|
fe9fc7364b | ||
|
|
1ce4589830 | ||
|
|
7417cc1b0a | ||
|
|
ec44b7955b | ||
|
|
cf3d8ea9a5 | ||
|
|
756e45743c | ||
|
|
7e8755d141 | ||
|
|
8194707478 | ||
|
|
d740064a35 | ||
|
|
164d14eb87 | ||
|
|
720399b00d | ||
|
|
da4d808c6a | ||
|
|
79cde50e27 | ||
|
|
d31d9fa584 | ||
|
|
c76698efb7 | ||
|
|
601db79689 | ||
|
|
acebbdbe26 | ||
|
|
fc8f7baf3e | ||
|
|
3f9b968a41 | ||
|
|
1263c7540c | ||
|
|
a8d8b1465c | ||
|
|
b76d54e702 | ||
|
|
59854a0229 | ||
|
|
dc83333526 | ||
|
|
a960c662f0 | ||
|
|
6f602b2245 | ||
|
|
e62e9c9a6a | ||
|
|
5beed2d75c | ||
|
|
fe7fc55c04 | ||
|
|
6cdae391f4 | ||
|
|
95eb494370 | ||
|
|
9294d2fd25 | ||
|
|
0fe7ddf6e8 | ||
|
|
5c2d61d35a | ||
|
|
186e9187ce | ||
|
|
e6f15404f5 | ||
|
|
44a0acc18f | ||
|
|
1fc6e63178 | ||
|
|
3589fc31d7 | ||
|
|
89746718a9 | ||
|
|
232d0e7c40 | ||
|
|
0b60d3ff2d | ||
|
|
ff82e8885d | ||
|
|
9a7411d1dd | ||
|
|
2f829ba3e7 | ||
|
|
dedd69dc42 | ||
|
|
f7f670ca4c | ||
|
|
55ba283c82 | ||
|
|
2d46d25a67 | ||
|
|
f5d29be041 | ||
|
|
e10a1572c3 | ||
|
|
578ef70aa9 | ||
|
|
e9fcbbbbcb | ||
|
|
f46ae18672 | ||
|
|
fc277294f3 | ||
|
|
f3e5ea67a0 | ||
|
|
7a8aebf679 | ||
|
|
a8d95aad02 | ||
|
|
dac5fbd574 | ||
|
|
da0c651c60 | ||
|
|
97019205da | ||
|
|
5487490b8e | ||
|
|
543ea432f0 | ||
|
|
c7bbf88dce | ||
|
|
d57145b31e | ||
|
|
1cbe80c0ca | ||
|
|
01ee261cf8 | ||
|
|
f6f30c675f | ||
|
|
60b80cbd96 | ||
|
|
27a8c7fad9 | ||
|
|
d9cee0a674 | ||
|
|
6c1891812e | ||
|
|
59316c982e | ||
|
|
9eca3bd719 | ||
|
|
f2f0fcf756 | ||
|
|
b4fb9c25f2 | ||
|
|
8955840f0a | ||
|
|
a4b49c9909 | ||
|
|
c0936a10e7 | ||
|
|
f4bbd6331d | ||
|
|
b99ba87cf6 | ||
|
|
2128353f9f | ||
|
|
7b4ce0ab3d | ||
|
|
3123565011 | ||
|
|
36f5ebdf86 | ||
|
|
bba054db85 | ||
|
|
b63a529a1d | ||
|
|
ee96713a50 | ||
|
|
1ece0c75d9 | ||
|
|
548c6f5041 | ||
|
|
bae4c8171f | ||
|
|
3748c86049 | ||
|
|
9dd969cf4c | ||
|
|
7e4b2d33fb | ||
|
|
288d50fd33 | ||
|
|
f0e76f8aff | ||
|
|
e80d195284 | ||
|
|
c770feebc9 | ||
|
|
7fd6d558d5 | ||
|
|
58c40292e2 | ||
|
|
94842875d0 | ||
|
|
cfa36b97fc | ||
|
|
4444398cac | ||
|
|
f80378f92f | ||
|
|
cd4997f429 | ||
|
|
6ce4784376 | ||
|
|
5856d45575 | ||
|
|
e3498b1087 | ||
|
|
724d9a45af | ||
|
|
109429c7da | ||
|
|
40856b2e5e | ||
|
|
23634134f0 | ||
|
|
34c55f27ca | ||
|
|
dc36a47ac9 | ||
|
|
750121e9dd | ||
|
|
479444d26a | ||
|
|
8c19ee9d72 | ||
|
|
2514d3bc89 | ||
|
|
48d6d6f7a2 | ||
|
|
cf0d62696f | ||
|
|
b2a3d5fbdd | ||
|
|
86e55ff954 | ||
|
|
d3d6382360 | ||
|
|
b8d21d78d9 | ||
|
|
dcce023b14 | ||
|
|
4c5ade5605 | ||
|
|
b232daf5e1 | ||
|
|
b65736c0dc | ||
|
|
4f54a1f0cd | ||
|
|
2d85ecec8f | ||
|
|
34bb14d9f6 | ||
|
|
515da66db9 | ||
|
|
2e9d3da301 | ||
|
|
321048626e | ||
|
|
16d6baea5f | ||
|
|
d082c6a084 | ||
|
|
ee93322ae2 | ||
|
|
d1ce8d3e71 | ||
|
|
f7c32aff12 | ||
|
|
bf88a6ebb0 | ||
|
|
de508141a8 | ||
|
|
18cd9c2342 | ||
|
|
997918e2f7 | ||
|
|
9872b8ff1d | ||
|
|
e9b225d2d6 | ||
|
|
2860f750b4 | ||
|
|
781b1d324e | ||
|
|
3014a45936 | ||
|
|
cda2132d6a | ||
|
|
6c1cc821a0 | ||
|
|
4be064572d | ||
|
|
89d506180d | ||
|
|
6e1e992652 | ||
|
|
5467c2e140 | ||
|
|
123cd2b3d7 | ||
|
|
006dd8097a | ||
|
|
18cd6e884e | ||
|
|
3102d66337 | ||
|
|
dfa5452af5 | ||
|
|
852ba2240f | ||
|
|
096dff4fa0 | ||
|
|
b5f749daa9 | ||
|
|
adaf53d647 | ||
|
|
01679f1639 | ||
|
|
8f4c88347d | ||
|
|
e9832d87c4 | ||
|
|
08aac4261f | ||
|
|
2526b2dca7 | ||
|
|
2df53fd5d7 | ||
|
|
5e2ff76cf7 | ||
|
|
e8c4122460 | ||
|
|
5db41b1166 | ||
|
|
ea7794dc05 | ||
|
|
67752fb1e8 | ||
|
|
1a14b1d4d3 | ||
|
|
68aea4cdcc | ||
|
|
0853952269 | ||
|
|
055577ee91 | ||
|
|
a63e1c70c3 | ||
|
|
a9745b594b | ||
|
|
78bb784265 | ||
|
|
784e064fa2 | ||
|
|
4a5afa6c1c | ||
|
|
c844ccc4e3 | ||
|
|
9a0aea8f89 | ||
|
|
b013e28dcd | ||
|
|
5a56030f6e | ||
|
|
f8719437ba | ||
|
|
9e34b9cbb9 | ||
|
|
4dabd725f0 | ||
|
|
2743d664a5 | ||
|
|
cfa34f3565 | ||
|
|
e021b49412 | ||
|
|
0c29c6a93d | ||
|
|
c329988cdd | ||
|
|
7a2b5ecf71 | ||
|
|
d74efd9f66 | ||
|
|
63a8c627f5 | ||
|
|
e4b335be23 | ||
|
|
0bf99c56cc | ||
|
|
e368f68ad7 | ||
|
|
fa52c2849a | ||
|
|
dce3788499 | ||
|
|
e00be5d2da | ||
|
|
82eea7f088 | ||
|
|
955a443b36 | ||
|
|
4800aebf43 | ||
|
|
cb737642e5 | ||
|
|
6d5670c8d2 | ||
|
|
c911a605e9 | ||
|
|
6002c5c738 | ||
|
|
92c50df2f1 | ||
|
|
45e61844d5 | ||
|
|
e72df8fed5 | ||
|
|
5497a5b0ae | ||
|
|
e10e493ddd | ||
|
|
ae9acecbf3 | ||
|
|
a3be05538b | ||
|
|
289163c712 | ||
|
|
2123e4ad69 | ||
|
|
177950dd3d | ||
|
|
cbf1239be4 | ||
|
|
8a282ee32e | ||
|
|
25079a7f0a | ||
|
|
4952e5224d | ||
|
|
98409f4502 | ||
|
|
7779f3e137 | ||
|
|
1462829aa6 | ||
|
|
48ce2c4097 | ||
|
|
bcae177d8e | ||
|
|
d35aca3138 | ||
|
|
147cb3803a | ||
|
|
98445565d6 | ||
|
|
eef4b05254 | ||
|
|
619324589a | ||
|
|
400af07154 | ||
|
|
9bb626327f | ||
|
|
1f8bf8c852 | ||
|
|
5d7a6327b7 | ||
|
|
0b9f124713 | ||
|
|
9479d56e11 | ||
|
|
c53c1940d6 | ||
|
|
79e8f9f2ce | ||
|
|
5fe575d362 | ||
|
|
b46424343f | ||
|
|
33b1d5ec79 | ||
|
|
f565e9c2a1 | ||
|
|
a91634c460 | ||
|
|
42919ea15c | ||
|
|
60c155327d | ||
|
|
0f78e96b81 | ||
|
|
12a2354fad | ||
|
|
f237a6a3d2 | ||
|
|
a2ce01d6ce | ||
|
|
fb6ee27e10 | ||
|
|
85ff5e01e8 | ||
|
|
eb881d4cd8 | ||
|
|
96cc63671a | ||
|
|
104804dbf6 | ||
|
|
0a2298b0e2 | ||
|
|
e73ae6044e | ||
|
|
14621e7ad5 | ||
|
|
cb141f0a41 | ||
|
|
e92aab617f | ||
|
|
b4474cf1e1 | ||
|
|
453c7b5513 | ||
|
|
b3463e4b33 | ||
|
|
013b2c7009 | ||
|
|
12f9153957 | ||
|
|
1004538f00 | ||
|
|
87873300e1 | ||
|
|
e9e2312163 | ||
|
|
6b09a808ed | ||
|
|
9496efdd22 | ||
|
|
33c9506c7d | ||
|
|
abe3763cb3 | ||
|
|
744904b19f | ||
|
|
9cbaf1a070 | ||
|
|
42da626397 | ||
|
|
d7aba8554d | ||
|
|
e176ff723d | ||
|
|
61bbe5e6da | ||
|
|
ab719cc8eb | ||
|
|
91c977e5e7 | ||
|
|
7e9c0c2a6f | ||
|
|
fd773b65da | ||
|
|
d340aca403 | ||
|
|
bf8ee3ce48 | ||
|
|
98d4545e8f | ||
|
|
d8687d979c | ||
|
|
a2b557c034 | ||
|
|
1d44523181 | ||
|
|
222a3ced69 | ||
|
|
03137eef9a | ||
|
|
af0fe5bdfd | ||
|
|
23b910a011 | ||
|
|
19f076bba1 | ||
|
|
9a1d6842d7 | ||
|
|
f5ef3194d4 | ||
|
|
90e3397b42 | ||
|
|
be5812d5bb | ||
|
|
84f7e038cb | ||
|
|
91d8f8fab1 | ||
|
|
723212a81f | ||
|
|
a5990ccf7d | ||
|
|
75d09d57e3 | ||
|
|
6847a7a6f4 | ||
|
|
17dd00bee1 | ||
|
|
b3c81453e4 | ||
|
|
656dd975f0 | ||
|
|
2661f6d271 | ||
|
|
0532c3a282 | ||
|
|
ac8c9a20e3 | ||
|
|
0e66cdabc9 | ||
|
|
9c892ce1c2 | ||
|
|
bfeea90597 | ||
|
|
cafeb8853b | ||
|
|
2c2d464503 | ||
|
|
5ec7033439 | ||
|
|
9133afd444 | ||
|
|
eeec9ce302 | ||
|
|
a62c4fc023 | ||
|
|
7e51168d1b | ||
|
|
97ede62f6d | ||
|
|
ab5206b1c7 | ||
|
|
2041557ab3 | ||
|
|
2fd483697b | ||
|
|
89186b0dee | ||
|
|
8e3606cbd3 | ||
|
|
a18dfb0127 | ||
|
|
75f1342339 | ||
|
|
36ba99c9a1 | ||
|
|
0cd8f15b48 | ||
|
|
8613b93fa8 | ||
|
|
20f9475a7d | ||
|
|
6babb3eec5 | ||
|
|
33730e6ccf | ||
|
|
65f89806cb | ||
|
|
c3007097a6 | ||
|
|
9724babcea | ||
|
|
7055bcdb97 | ||
|
|
d973da487c | ||
|
|
c939454983 | ||
|
|
274d54a774 | ||
|
|
ad274d99fb | ||
|
|
bd8d93d890 | ||
|
|
53b9a64826 | ||
|
|
0ba837817e | ||
|
|
80f86b086b | ||
|
|
a351cbcf80 | ||
|
|
2aa625d566 | ||
|
|
288835db29 | ||
|
|
559c538a3d | ||
|
|
feee2f1679 | ||
|
|
120240f14a | ||
|
|
b0ccbb1162 | ||
|
|
95d4db5609 | ||
|
|
6f8c6c8ef1 | ||
|
|
627aad8c29 | ||
|
|
745f4d6fb4 | ||
|
|
c5f8bb4d1d | ||
|
|
e337063a95 | ||
|
|
d8d14db505 | ||
|
|
be05c63a67 | ||
|
|
ff695c1917 | ||
|
|
fa2a9a595e | ||
|
|
a90c4909d5 | ||
|
|
649be34dcf | ||
|
|
3c3c5ae821 | ||
|
|
eadb1e35ce |
Submodule
+1
Submodule .claude/worktrees/scripts-dedup added at e578443914
@@ -13,17 +13,17 @@ jobs:
|
|||||||
steps:
|
steps:
|
||||||
- name: Check out repository
|
- name: Check out repository
|
||||||
uses: actions/checkout@v4
|
uses: actions/checkout@v4
|
||||||
|
with:
|
||||||
|
fetch-depth: 0
|
||||||
|
|
||||||
- name: Install Nix
|
- name: Install Nix
|
||||||
uses: DeterminateSystems/nix-installer-action@v19
|
uses: DeterminateSystems/nix-installer-action@v19
|
||||||
|
|
||||||
- name: Evaluate all NixOS hosts
|
# Scoped to files changed since the PR base / previous push -- see
|
||||||
run: |
|
# scripts/codex-maintenance.sh. CI never passes --full-check: that
|
||||||
set -euo pipefail
|
# full sweep is for local/manual use, since it's slow enough to time
|
||||||
hosts="$(nix --extra-experimental-features 'nix-command flakes' eval --json \
|
# out this runner.
|
||||||
.#nixosConfigurations --apply builtins.attrNames | jq -r '.[]')"
|
- name: Run maintenance checks (secrets, fmt, lint, eval -- changed files only)
|
||||||
for host in $hosts; do
|
env:
|
||||||
echo "Evaluating ${host}"
|
MAINT_BASE_SHA: ${{ github.event.pull_request.base.sha || github.event.before }}
|
||||||
nix --extra-experimental-features 'nix-command flakes' eval \
|
run: bash scripts/codex-maintenance.sh
|
||||||
".#nixosConfigurations.${host}.config.system.build.toplevel.drvPath" --raw
|
|
||||||
done
|
|
||||||
|
|||||||
@@ -13,17 +13,17 @@ jobs:
|
|||||||
steps:
|
steps:
|
||||||
- name: Check out repository
|
- name: Check out repository
|
||||||
uses: actions/checkout@v4
|
uses: actions/checkout@v4
|
||||||
|
with:
|
||||||
|
fetch-depth: 0
|
||||||
|
|
||||||
- name: Install Nix
|
- name: Install Nix
|
||||||
uses: DeterminateSystems/nix-installer-action@v19
|
uses: DeterminateSystems/nix-installer-action@v19
|
||||||
|
|
||||||
- name: Evaluate all NixOS hosts
|
# Scoped to files changed since the PR base / previous push -- see
|
||||||
run: |
|
# scripts/codex-maintenance.sh. CI never passes --full-check: that
|
||||||
set -euo pipefail
|
# full sweep is for local/manual use, since it's slow enough to time
|
||||||
hosts="$(nix --extra-experimental-features 'nix-command flakes' eval --json \
|
# out this runner.
|
||||||
.#nixosConfigurations --apply builtins.attrNames | jq -r '.[]')"
|
- name: Run maintenance checks (secrets, fmt, lint, eval -- changed files only)
|
||||||
for host in $hosts; do
|
env:
|
||||||
echo "Evaluating ${host}"
|
MAINT_BASE_SHA: ${{ github.event.pull_request.base.sha || github.event.before }}
|
||||||
nix --extra-experimental-features 'nix-command flakes' eval \
|
run: bash scripts/codex-maintenance.sh
|
||||||
".#nixosConfigurations.${host}.config.system.build.toplevel.drvPath" --raw
|
|
||||||
done
|
|
||||||
|
|||||||
+15
-3
@@ -3,13 +3,25 @@
|
|||||||
result
|
result
|
||||||
result-*
|
result-*
|
||||||
|
|
||||||
|
# Disko's proxmox-* image-builder writes the finished .raw disk image
|
||||||
|
# directly into the current directory, not into a result-* symlink (see
|
||||||
|
# docs/proxmox-images.md, scripts/create-proxmox-resource.sh) — several GB
|
||||||
|
# each, never meant to be committed.
|
||||||
|
*.raw
|
||||||
|
|
||||||
# Ignore automatically generated direnv output
|
# Ignore automatically generated direnv output
|
||||||
.direnv
|
.direnv
|
||||||
|
|
||||||
auto-installer/flake.lock
|
# Python bytecode cache (scripts/lib/*.py)
|
||||||
auto-installer/result
|
__pycache__/
|
||||||
auto-installer/nixos-auto.iso
|
*.pyc
|
||||||
|
|
||||||
|
# Locally-generated SSH host keys staged for transfer to a new machine
|
||||||
|
# during install (see scripts/prepare-host-key.sh) — never commit these.
|
||||||
|
host-keys/
|
||||||
|
|
||||||
# Temporary Milestone 1 audit checklist (remove-sensetive-info-refactor.md)
|
# Temporary Milestone 1 audit checklist (remove-sensetive-info-refactor.md)
|
||||||
# - working notes only, never committed, deleted once every row is rotated.
|
# - working notes only, never committed, deleted once every row is rotated.
|
||||||
secrets-inventory.md
|
secrets-inventory.md
|
||||||
|
.claude/worktrees/
|
||||||
|
.claude/settings.local.json
|
||||||
+197
-12
@@ -1,8 +1,28 @@
|
|||||||
keys:
|
keys:
|
||||||
- &admin age10nd382a9klsn2mrs60emdtsxe43pht3a0m9p29phfrhy0wfyt3vsq9r667
|
- &admin age1njap586hc0q43kr03g6c8eqhdsmk8zcafkl3f83xwlc2gqhlmfgs4tmwad
|
||||||
- &docker age19gfn2yedg76dmztm4hncr7vf3r3c9j0qpt4rap7y7gersjk4m3ks2lhd0e
|
- &proxmox-minimal age19m0m7vdfg86yqy8l5mmle5jdd0unrn3f55t232w8h5ey42cqw34sfpt32n
|
||||||
- &server age1ll6hj5ggruetgjwjfnplpn5xtq35uhlcdflksx3xmnjm6s3uad9sz70jkf
|
- &lxc-gui age1rrxqea6q6pn39sw8y5te63h2py8jgjl9v0jyper86w3ggtn67upqg3ah39
|
||||||
- &nix-cache age120le4a5l8dh3lyfgvmj3d9ksmej6ajs5mer5y7r0vfg3x9fn69dqf8xgzu
|
- &baremetal-gui age1adur9g330gua4l6ndk8cqjg35qc8yxwgme6wrl2hpylcc7vxm38q05ejuy
|
||||||
|
- &linode-docker age17e89ty6p0fw24daanen57wg8uald9s025t3wwxsw269svwpmgvrshfvfvt
|
||||||
|
- &linode-gui age1hrx8qj02fj2ea6d4g9vqhyj9hl7fppkjqfdx2l37py3h6pdkr95s8n8rvs
|
||||||
|
- &linode-minimal age1e7l8dusgmgfzd2cxrrzwepzjxt69hzqj4epee0cs27u6yg4kxcuqm34ncx
|
||||||
|
- &linode-nix-cache age1jcx3yajjhghn8qh8za3yeu8nxykzlg3p4nrv03vnfvzl0mzayg2qmg940e
|
||||||
|
- &linode-tailscale-router age1f7usptjx9rv4rxauasve200gxtdt9jkqhhdqstlf20wvlm7u75rsjfw50m
|
||||||
|
- &lxc-docker age17jqc66x9yeshfgd9v78mj483r4zzarqdtuxtrkxe4x5mw679gphshd94th
|
||||||
|
- &lxc-minimal age1px0h5l9zp2dww0m8fncrc82kfdmzplsfv2ltat7sna28xpg09pqqcl3s2k
|
||||||
|
- &lxc-nix-cache age1ufg390ydrmma849t9xfkxxl5xvdkk6mngnlzhmy7mvuaje8sgcmsmnq6l7
|
||||||
|
- &lxc-pxe-boot age16j42pdc5dr6wnj7xayhkqdj2rny9u68fcqejs50hqq42scssh4gsnrrnlt
|
||||||
|
- &lxc-tailscale-router age1k7d2du5mejsmv5rzavm4xwgpthqvcfsehduquv28nzs53zppa3kqngfxq2
|
||||||
|
- &lxc-tor-relay age16kqfmvz4e23hmdlqresnyw69ej604s320mmd49h4hm3fhqchtgyqrws0k2
|
||||||
|
- &proxmox-docker age1arhf2q45zw6wf2uevju4savp575x3m2tfvved5zzq3ay92ynua9s3cm92c
|
||||||
|
- &proxmox-gui age19mn8zrxl8zpps9yvrh4euquvygpp4fp8queg7xc6qhtnl4ng8c9qx02qwn
|
||||||
|
- &proxmox-nix-cache age1jlltcv5jcnm40z5k0q6hv053k2rqpqvemtuecdwn527uw8uqz4es3x7m68
|
||||||
|
- &proxmox-pxe-boot age1ug787sgt6st6k82fgkrug2lzltw4qsukrrqqs3w27ewwqj8rg4hsxcmylz
|
||||||
|
- &proxmox-tailscale-router age1zhfyuzlq40reuqlr34gf77852nhs3t6mqfzrqmas8z6sxk7tcfhsungrm0
|
||||||
|
- &proxmox-ha-server-1 age1k73g8x47hs93wcv7qh92n3htz8pl295g49hyvlrf3570mts0hgys5g04d6
|
||||||
|
- &proxmox-ha-server-2 age1fefy6dk8zn5c3edwmrs9vwx79quftnt784m628t9e34q3ft3cehqz8u72r
|
||||||
|
- &proxmox-ha-docker-1 age16y0vfts5v2k20e2rj23qa5lc5gm96gm64uwsqsr0y688xl0ne46qcfr0sm
|
||||||
|
- &proxmox-ha-docker-2 age1tm3zj5lp3elw2j832az4nj9xxmhqd66ag3cqcv3vskvmd7qdmqmsdpdcn0
|
||||||
|
|
||||||
creation_rules:
|
creation_rules:
|
||||||
# Shared across every currently-deployed host: root/nixos password hash,
|
# Shared across every currently-deployed host: root/nixos password hash,
|
||||||
@@ -13,24 +33,189 @@ creation_rules:
|
|||||||
key_groups:
|
key_groups:
|
||||||
- age:
|
- age:
|
||||||
- *admin
|
- *admin
|
||||||
- *docker
|
- *proxmox-minimal
|
||||||
- *server
|
- *lxc-gui
|
||||||
- *nix-cache
|
- *baremetal-gui
|
||||||
|
- *linode-docker
|
||||||
|
- *linode-gui
|
||||||
|
- *linode-minimal
|
||||||
|
- *linode-nix-cache
|
||||||
|
- *linode-tailscale-router
|
||||||
|
- *lxc-docker
|
||||||
|
- *lxc-minimal
|
||||||
|
- *lxc-nix-cache
|
||||||
|
- *lxc-pxe-boot
|
||||||
|
- *lxc-tailscale-router
|
||||||
|
- *lxc-tor-relay
|
||||||
|
- *proxmox-docker
|
||||||
|
- *proxmox-gui
|
||||||
|
- *proxmox-nix-cache
|
||||||
|
- *proxmox-pxe-boot
|
||||||
|
- *proxmox-tailscale-router
|
||||||
|
- *proxmox-ha-server-1
|
||||||
|
- *proxmox-ha-server-2
|
||||||
|
- *proxmox-ha-docker-1
|
||||||
|
- *proxmox-ha-docker-2
|
||||||
|
|
||||||
- path_regex: secrets/nix-cache\.yaml$
|
- path_regex: secrets/nix-cache\.yaml$
|
||||||
key_groups:
|
key_groups:
|
||||||
- age:
|
- age:
|
||||||
- *admin
|
- *admin
|
||||||
- *nix-cache
|
- *linode-nix-cache
|
||||||
|
- *lxc-nix-cache
|
||||||
|
- *proxmox-nix-cache
|
||||||
|
|
||||||
- path_regex: secrets/server\.yaml$
|
- path_regex: secrets/tor-relay\.yaml$
|
||||||
key_groups:
|
key_groups:
|
||||||
- age:
|
- age:
|
||||||
- *admin
|
- *admin
|
||||||
- *server
|
- *lxc-tor-relay
|
||||||
|
|
||||||
- path_regex: secrets/docker\.yaml$
|
- path_regex: secrets/tailscale-router\.yaml$
|
||||||
key_groups:
|
key_groups:
|
||||||
- age:
|
- age:
|
||||||
- *admin
|
- *admin
|
||||||
- *docker
|
- *linode-tailscale-router
|
||||||
|
- *lxc-tailscale-router
|
||||||
|
- *proxmox-tailscale-router
|
||||||
|
|
||||||
|
# HA file server per-node secrets (beszel-token).
|
||||||
|
# proxmox-ha-server-1 / proxmox-ha-server-2 keys are added automatically
|
||||||
|
# by scripts/secrets/sync-host-keys.sh once the hosts are provisioned;
|
||||||
|
# until then only the admin key can decrypt these files.
|
||||||
|
- path_regex: secrets/ha-server-1\.yaml$
|
||||||
|
key_groups:
|
||||||
|
- age:
|
||||||
|
- *admin
|
||||||
|
- *proxmox-ha-server-1
|
||||||
|
# proxmox-ha-server-1 added by sync-host-keys.sh
|
||||||
|
|
||||||
|
- path_regex: secrets/ha-server-2\.yaml$
|
||||||
|
key_groups:
|
||||||
|
- age:
|
||||||
|
- *admin
|
||||||
|
- *proxmox-ha-server-2
|
||||||
|
# proxmox-ha-server-2 added by sync-host-keys.sh
|
||||||
|
|
||||||
|
# Shared HA cluster corosync authkey (binary sops file).
|
||||||
|
# Encrypted for both HA nodes so either can decrypt on boot.
|
||||||
|
# Both host keys added by sync-host-keys.sh; admin key allows initial creation.
|
||||||
|
- path_regex: secrets/ha-corosync-authkey$
|
||||||
|
key_groups:
|
||||||
|
- age:
|
||||||
|
- *admin
|
||||||
|
- *proxmox-ha-server-1
|
||||||
|
- *proxmox-ha-server-2
|
||||||
|
# proxmox-ha-server-1 added by sync-host-keys.sh
|
||||||
|
# proxmox-ha-server-2 added by sync-host-keys.sh
|
||||||
|
|
||||||
|
# gui-host-specific secrets (currently: wifi-password, see
|
||||||
|
# modules/networking/wifi.nix). Only *lxc-gui has a registered key today
|
||||||
|
# -- proxmox-gui/linode-gui/baremetal-gui haven't been provisioned via
|
||||||
|
# scripts/secrets/sync-host-keys.sh yet, so whichever variant is actually
|
||||||
|
# deployed next needs its recipient added here (and `sops updatekeys` rerun)
|
||||||
|
# before it can decrypt this.
|
||||||
|
- path_regex: secrets/gui\.yaml$
|
||||||
|
key_groups:
|
||||||
|
- age:
|
||||||
|
- *admin
|
||||||
|
- *lxc-gui
|
||||||
|
- *baremetal-gui
|
||||||
|
- *linode-gui
|
||||||
|
- *proxmox-gui
|
||||||
|
|
||||||
|
# IPA host keytabs (binary sops files).
|
||||||
|
# Each keytab is encrypted for all platform variants of that host so any
|
||||||
|
# deployed variant can decrypt it at boot. Run
|
||||||
|
# scripts/ipa/create-nixos-ipa-host-account.sh <hostname> to enroll a new
|
||||||
|
# host and produce the keytab; this section is updated by that script.
|
||||||
|
|
||||||
|
- path_regex: secrets/nix-cache\.keytab$
|
||||||
|
key_groups:
|
||||||
|
- age:
|
||||||
|
- *admin
|
||||||
|
- *linode-nix-cache
|
||||||
|
- *lxc-nix-cache
|
||||||
|
- *proxmox-nix-cache
|
||||||
|
|
||||||
|
- path_regex: secrets/tailscale-router\.keytab$
|
||||||
|
key_groups:
|
||||||
|
- age:
|
||||||
|
- *admin
|
||||||
|
- *linode-tailscale-router
|
||||||
|
- *lxc-tailscale-router
|
||||||
|
- *proxmox-tailscale-router
|
||||||
|
|
||||||
|
- path_regex: secrets/pxe-boot\.keytab$
|
||||||
|
key_groups:
|
||||||
|
- age:
|
||||||
|
- *admin
|
||||||
|
- *lxc-pxe-boot
|
||||||
|
- *proxmox-pxe-boot
|
||||||
|
|
||||||
|
# nixos = the workstation (hosts/nixos/host.nix). All gui platform variants
|
||||||
|
# share the hostname "nixos" and must be able to decrypt at boot.
|
||||||
|
- path_regex: secrets/nixos\.keytab$
|
||||||
|
key_groups:
|
||||||
|
- age:
|
||||||
|
- *admin
|
||||||
|
- *baremetal-gui
|
||||||
|
- *lxc-gui
|
||||||
|
- *proxmox-gui
|
||||||
|
- *linode-gui
|
||||||
|
|
||||||
|
- path_regex: secrets/docker\.keytab$
|
||||||
|
key_groups:
|
||||||
|
- age:
|
||||||
|
- *admin
|
||||||
|
- *linode-docker
|
||||||
|
- *lxc-docker
|
||||||
|
- *proxmox-docker
|
||||||
|
|
||||||
|
- path_regex: secrets/tor-relay\.keytab$
|
||||||
|
key_groups:
|
||||||
|
- age:
|
||||||
|
- *admin
|
||||||
|
- *lxc-tor-relay
|
||||||
|
|
||||||
|
- path_regex: secrets/nix-minimal\.keytab$
|
||||||
|
key_groups:
|
||||||
|
- age:
|
||||||
|
- *admin
|
||||||
|
- *lxc-minimal
|
||||||
|
- *proxmox-minimal
|
||||||
|
- *linode-minimal
|
||||||
|
|
||||||
|
# Host keytab for ha-server-1 FreeIPA enrollment (binary sops file).
|
||||||
|
# Generated by scripts/ipa/create-nixos-ipa-host-account.sh.
|
||||||
|
- path_regex: secrets/ha-server-1\.keytab$
|
||||||
|
key_groups:
|
||||||
|
- age:
|
||||||
|
- *admin
|
||||||
|
- *proxmox-ha-server-1
|
||||||
|
# proxmox-ha-server-1 added by sync-host-keys.sh
|
||||||
|
|
||||||
|
# Host keytab for ha-server-2 FreeIPA enrollment (binary sops file).
|
||||||
|
# Generated by scripts/ipa/create-nixos-ipa-host-account.sh.
|
||||||
|
- path_regex: secrets/ha-server-2\.keytab$
|
||||||
|
key_groups:
|
||||||
|
- age:
|
||||||
|
- *admin
|
||||||
|
- *proxmox-ha-server-2
|
||||||
|
# proxmox-ha-server-2 added by sync-host-keys.sh
|
||||||
|
|
||||||
|
# Host keytab for ha-docker-1 FreeIPA enrollment (binary sops file).
|
||||||
|
# Generated by scripts/ipa/create-nixos-ipa-host-account.sh.
|
||||||
|
- path_regex: secrets/ha-docker-1\.keytab$
|
||||||
|
key_groups:
|
||||||
|
- age:
|
||||||
|
- *admin
|
||||||
|
- *proxmox-ha-docker-1
|
||||||
|
|
||||||
|
# Host keytab for ha-docker-2 FreeIPA enrollment (binary sops file).
|
||||||
|
# Generated by scripts/ipa/create-nixos-ipa-host-account.sh.
|
||||||
|
- path_regex: secrets/ha-docker-2\.keytab$
|
||||||
|
key_groups:
|
||||||
|
- age:
|
||||||
|
- *admin
|
||||||
|
- *proxmox-ha-docker-2
|
||||||
|
|||||||
@@ -6,12 +6,14 @@ This repository contains flake-based NixOS configurations for Wayne's LAN
|
|||||||
servers and workstation.
|
servers and workstation.
|
||||||
|
|
||||||
The flake exposes NixOS configurations named `<platform>-<buildtype>`
|
The flake exposes NixOS configurations named `<platform>-<buildtype>`
|
||||||
(platforms: `linode`, `proxmox`, `lxc`; build types: `minimal`, `nix-cache`,
|
(platforms: `linode`, `proxmox`, `lxc`, `baremetal`; build types: `minimal`,
|
||||||
`server`, `docker`, `gui`, `pxe-boot`), generated from `modules/platforms/*`
|
`nix-cache`, `docker`, `gui`, `pxe-boot`, `tailscale-router`,
|
||||||
and `modules/build-types/*` by the `mkTarget` function in `flake.nix`. Not
|
`tor-relay`, `ha-server`), generated from `modules/platforms/*` and
|
||||||
every combination is built — `pxe-boot` has no `linode` variant. See
|
`modules/build-types/*` by the `mkTarget` function in `flake.nix`. Not every
|
||||||
`README.md` for the full current target list; treat `flake.nix` as the
|
combination is built — `pxe-boot` has no `linode` variant, `ha-server` only
|
||||||
source of truth since this list can drift.
|
exists on `proxmox`, and `tor-relay` only exists on `lxc`. See `README.md`
|
||||||
|
for the full current target list; treat `flake.nix` as the source of truth
|
||||||
|
since this list can drift.
|
||||||
|
|
||||||
Do not deploy, switch, reboot, repartition, format disks, or run destructive
|
Do not deploy, switch, reboot, repartition, format disks, or run destructive
|
||||||
install commands from this repository unless explicitly asked.
|
install commands from this repository unless explicitly asked.
|
||||||
@@ -35,9 +37,14 @@ Use these commands when validating changes:
|
|||||||
```bash
|
```bash
|
||||||
bash scripts/codex-setup.sh
|
bash scripts/codex-setup.sh
|
||||||
bash scripts/codex-maintenance.sh
|
bash scripts/codex-maintenance.sh
|
||||||
bash scripts/codex-maintenance.sh dry-run
|
|
||||||
```
|
```
|
||||||
|
|
||||||
|
With no flags, `codex-maintenance.sh` scopes fmt-check/statix/eval to files
|
||||||
|
changed against a base ref — this is what CI runs on every push/PR. For the
|
||||||
|
full sweep (every host, every package — slow; CI never runs this), use
|
||||||
|
`bash scripts/codex-maintenance.sh --full-check` (add `--dry-run` for build
|
||||||
|
planning on top of whichever scope is active).
|
||||||
|
|
||||||
Host evaluation is safe when limited to drvPath checks:
|
Host evaluation is safe when limited to drvPath checks:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
|
|||||||
@@ -17,11 +17,89 @@ machines when deployed.
|
|||||||
- Validation is limited to evaluation, linting, formatting checks, and
|
- Validation is limited to evaluation, linting, formatting checks, and
|
||||||
`nix build --dry-run --no-link`.
|
`nix build --dry-run --no-link`.
|
||||||
- Do not add secrets, tokens, private keys, or new password hashes to the repo.
|
- Do not add secrets, tokens, private keys, or new password hashes to the repo.
|
||||||
- This repo currently contains **committed password hashes** (e.g.
|
- This repo currently contains **committed password hashes** in
|
||||||
`prepare.sh`, `hosts/nixos/configuration.nix`) and SSH public keys (e.g.
|
`modules/installer/common.nix` (the auto-installer's own root/nixos login —
|
||||||
`modules/nix-cache/server.nix`). The hashes are known tech debt — do not use
|
a deliberate, documented choice, see `docs/auto-installer.md`, not
|
||||||
them as a template for new hosts, and flag any *new* secret-like string you
|
accidental tech debt) and **SSH public keys** in `variables.nix`
|
||||||
encounter instead of committing it.
|
(`vars.adminSshKey`, `vars.remoteBuilderAuthorizedKeys`, `vars.beszelHubKey`). Don't use the installer's hardcoded hash as a
|
||||||
|
template for a *real* host — every other host uses sops-nix
|
||||||
|
(`hashedPasswordFile`, see "Security Notes" in `README.md`). Flag any *new*
|
||||||
|
secret-like string you encounter instead of committing it.
|
||||||
|
- `host-keys/` is gitignored — used only by the auto-installer's own
|
||||||
|
environment for pre-seeding non-LXC host keys before first boot (see
|
||||||
|
`docs/auto-installer.md`). Never commit its contents; if `git status`
|
||||||
|
ever shows it as trackable, something is wrong. All deployed hosts use
|
||||||
|
clan vars (`vars/per-machine/<target>/openssh/`, committed and
|
||||||
|
sops-encrypted) for their SSH host keys — those ARE tracked by git and
|
||||||
|
belong in the repo.
|
||||||
|
|
||||||
|
### Two Proxmox nodes: `pve1.sweet.home` (production) and `pve-test.sweet.home` (sandbox)
|
||||||
|
|
||||||
|
There are two SSH-reachable Proxmox nodes on the LAN, both defined in
|
||||||
|
`scripts/env.sh` (`PVE1_HOST` / `PVE_TEST_HOST`), individually targetable
|
||||||
|
via `scripts/proxmox/create-proxmox-resource.sh --node <host>` or by
|
||||||
|
overriding `PROXMOX_HOST`. `PROXMOX_HOST` itself still defaults to
|
||||||
|
`PVE1_HOST` (production) — that default, and every other script behavior,
|
||||||
|
is unchanged from before `pve-test` existed; the only thing new is that
|
||||||
|
`pve-test` can now be reached at all. They are **not interchangeable** —
|
||||||
|
one is real production infrastructure, the other exists specifically so
|
||||||
|
there's somewhere safe to test. The restriction below is a policy for
|
||||||
|
Claude specifically, not a change to the tooling's own default or
|
||||||
|
anything the operator needs to opt into.
|
||||||
|
|
||||||
|
#### `pve1.sweet.home` (production — off-limits to Claude)
|
||||||
|
|
||||||
|
A real, live Proxmox node hosting production VMs/containers — not a
|
||||||
|
sandbox, and not Claude's to touch by default.
|
||||||
|
|
||||||
|
- **Off-limits at all times unless the operator has given explicit,
|
||||||
|
same-session instructions to act on this specific host.** That
|
||||||
|
authorization is scoped to the task it was given for — don't carry it
|
||||||
|
forward to unrelated later work in the same conversation, and never
|
||||||
|
assume it from a previous session.
|
||||||
|
- **Read-only for existing state is always fine, authorization or not.**
|
||||||
|
You may SSH in (or use `pvesm`, `qm list`, `pct list`, `qm config`, `pct
|
||||||
|
config`, the Proxmox API, etc.) to inspect the node's config, storage,
|
||||||
|
and any existing VM/container — including ones this repo didn't create.
|
||||||
|
- **Never** modify, stop, restart, delete, reconfigure, or create anything
|
||||||
|
on this node (`qm set`, `pct set`, `qm destroy`, `pct destroy`, `qm
|
||||||
|
stop`, `pct stop`, `qm create`, `pct create`, snapshot operations,
|
||||||
|
storage changes, etc.) — including scratch/test resources — without
|
||||||
|
that explicit go-ahead. Use `pve-test.sweet.home` for anything
|
||||||
|
exploratory instead; it exists precisely so `pve1` never has to be the
|
||||||
|
answer to "where do I test this."
|
||||||
|
- **This is a Claude-specific policy, not something the scripts enforce.**
|
||||||
|
`scripts/env.sh`/`create-proxmox-resource.sh` default to `pve1` exactly
|
||||||
|
as they did before `pve-test` existed, with no extra flag or prompt
|
||||||
|
required — that's deliberate, so the operator's own existing workflows
|
||||||
|
don't change. Claude, however, must never rely on that default: every
|
||||||
|
Proxmox action Claude takes on its own initiative — not explicitly
|
||||||
|
pointed at `pve1` by the operator this session — targets `pve-test`
|
||||||
|
instead (e.g. `--node "$PVE_TEST_HOST"`, or `PROXMOX_HOST=$PVE_TEST_HOST`).
|
||||||
|
Claude's own default is `pve-test`, full stop, regardless of what the
|
||||||
|
tooling's own unqualified default happens to be.
|
||||||
|
|
||||||
|
#### `pve-test.sweet.home` (sandbox — Claude's default target)
|
||||||
|
|
||||||
|
A separate Proxmox node set aside for testing. The *tooling's* default is
|
||||||
|
still production (`PROXMOX_HOST` → `PVE1_HOST`, see above) — but
|
||||||
|
**Claude's own default is this node**: absent an explicit, same-session
|
||||||
|
instruction to use `pve1`, every Proxmox action Claude initiates targets
|
||||||
|
`pve-test`. Once targeted, it's safe to create, interrogate, and destroy
|
||||||
|
resources on without asking first.
|
||||||
|
|
||||||
|
- **Test VMs/containers are allowed, but must be torn down.** Create a
|
||||||
|
scratch VM or container here (e.g. via
|
||||||
|
`scripts/proxmox/create-proxmox-resource.sh` or raw `qm`/`pct create`)
|
||||||
|
to validate something. Anything created this way must be destroyed
|
||||||
|
again in the same session, before ending the task — never leave a test
|
||||||
|
resource running. Use a VMID/name that's obviously scratch (and doesn't
|
||||||
|
collide with a real flake target) so it's unambiguous what's safe to
|
||||||
|
remove.
|
||||||
|
- **Node-level config is still not yours to change.** Creating/destroying
|
||||||
|
your own scratch guests is fine; Proxmox host config, storage pools, and
|
||||||
|
networking on `pve-test` itself are still the operator's call to make
|
||||||
|
manually, same as on `pve1`.
|
||||||
|
|
||||||
## Commands
|
## Commands
|
||||||
|
|
||||||
@@ -29,11 +107,22 @@ machines when deployed.
|
|||||||
# One-time environment bootstrap (installs Nix if missing, prints hosts)
|
# One-time environment bootstrap (installs Nix if missing, prints hosts)
|
||||||
bash scripts/codex-setup.sh
|
bash scripts/codex-setup.sh
|
||||||
|
|
||||||
# Full validation: secret grep, nixpkgs-fmt --check, statix lint, eval all hosts
|
# Changed-files-only validation: secret grep (whole repo), nixpkgs-fmt --check
|
||||||
|
# and statix on changed *.nix files, eval of the hosts/packages those changes
|
||||||
|
# can affect. This is what CI runs on every push/PR.
|
||||||
bash scripts/codex-maintenance.sh
|
bash scripts/codex-maintenance.sh
|
||||||
|
|
||||||
# Same, plus a dry-run build (no result symlink) of every host's toplevel
|
# Full sweep: nixpkgs-fmt --check/statix over the whole tree, eval every host
|
||||||
bash scripts/codex-maintenance.sh dry-run
|
# and package. Slow (minutes) -- CI never runs this; use it locally before a
|
||||||
|
# release or after touching modules/common/*, flake.nix, or variables.nix for
|
||||||
|
# extra confidence beyond the automatic full-fallback those paths already
|
||||||
|
# trigger in the default mode (see below).
|
||||||
|
bash scripts/codex-maintenance.sh --full-check
|
||||||
|
|
||||||
|
# Either mode, plus a dry-run build (no result symlink) of every host/package
|
||||||
|
# in whichever scope is active
|
||||||
|
bash scripts/codex-maintenance.sh --dry-run
|
||||||
|
bash scripts/codex-maintenance.sh --full-check --dry-run
|
||||||
|
|
||||||
# List the hosts the flake currently exposes
|
# List the hosts the flake currently exposes
|
||||||
nix eval --json .#nixosConfigurations --apply builtins.attrNames | jq -r '.[]'
|
nix eval --json .#nixosConfigurations --apply builtins.attrNames | jq -r '.[]'
|
||||||
@@ -50,58 +139,403 @@ maintenance script pulls them via `nix run github:NixOS/nixpkgs/nixos-25.11#<too
|
|||||||
There is no test suite — "correctness" here means the flake evaluates and
|
There is no test suite — "correctness" here means the flake evaluates and
|
||||||
`nixpkgs-fmt`/`statix` are clean.
|
`nixpkgs-fmt`/`statix` are clean.
|
||||||
|
|
||||||
|
With no flags, `codex-maintenance.sh` diffs against a base ref (env
|
||||||
|
`MAINT_BASE_SHA`, else the PR base SHA in CI, else `HEAD^` locally) and scopes
|
||||||
|
fmt-check/statix to the changed `*.nix` files and eval to the hosts/packages
|
||||||
|
those changes can affect — a `hosts/<name>/host.nix` edit only evals that
|
||||||
|
host's targets, a `modules/platforms/<platform>.nix` edit only evals that
|
||||||
|
platform's hosts, and so on. A change to `flake.nix`, `flake.lock`,
|
||||||
|
`variables.nix`, `modules/common/*`, or any other `modules/*.nix` file outside
|
||||||
|
`platforms/`/`build-types/` (whose blast radius isn't safely inferable from
|
||||||
|
the path alone) falls back to evaluating every host and package, same as
|
||||||
|
`--full-check` would, just without the whole-tree fmt/statix sweep. This
|
||||||
|
exists because the whole-tree sweep is what was timing out CI; **CI always
|
||||||
|
runs the plain, no-flag form and never passes `--full-check`.**
|
||||||
|
|
||||||
|
The default mode's diff is against the working tree (uncommitted and staged
|
||||||
|
edits included, not just committed ones), so it's already the right tool for
|
||||||
|
an interactive session too: after editing one or two hosts/modules, plain
|
||||||
|
`bash scripts/codex-maintenance.sh` naturally scopes to just what you
|
||||||
|
touched. Reserve `--full-check` for changes that plausibly affect every host
|
||||||
|
(`modules/common/*`, `flake.nix`, `variables.nix` — though the default mode
|
||||||
|
already falls back to evaluating everything for those paths, `--full-check`
|
||||||
|
additionally re-checks fmt/statix over the whole tree) or as a final check
|
||||||
|
before committing.
|
||||||
|
|
||||||
|
## Scripts
|
||||||
|
|
||||||
|
Beyond `codex-setup.sh`/`codex-maintenance.sh` above, `scripts/` is
|
||||||
|
organized by purpose: `scripts/secrets/` (sops/age + SSH host-key
|
||||||
|
management), `scripts/proxmox/` (Proxmox deployment), `scripts/installer/`
|
||||||
|
(the auto-installer's own shell script, templated into the image — see
|
||||||
|
below), `scripts/lib/` (shared helpers, sourced by the scripts below — not
|
||||||
|
run directly), and a handful of repo-wide scripts left at the top level
|
||||||
|
(`env.sh`, `bump-nixpkgs-release.sh`, plus `codex-setup.sh`/
|
||||||
|
`codex-maintenance.sh` above). When adding a new script, put it in the
|
||||||
|
matching subfolder rather than the top level, and if it duplicates logic
|
||||||
|
another script already has, lift the shared part into `scripts/lib/`
|
||||||
|
instead of copying it.
|
||||||
|
|
||||||
|
### `scripts/installer/`
|
||||||
|
|
||||||
|
- `scripts/installer/auto-install.sh` — the interactive install script
|
||||||
|
baked into the auto-installer image (see `docs/auto-installer.md`), kept
|
||||||
|
as a real, version-controlled shell file rather than inline in
|
||||||
|
`modules/installer/common.nix`'s Nix. It sources `scripts/env.sh` itself
|
||||||
|
for `LAN_DOMAIN` (`export LAN_DOMAIN`/`: "${LAN_DOMAIN:=...}"`, matching
|
||||||
|
`variables.nix`'s `lanDomain` — manually kept in sync, same pattern as
|
||||||
|
`NIX_CACHE_HOST` mirroring `nixCacheHost`), rather than Nix-level string
|
||||||
|
substitution — that's what makes it work identically whether run
|
||||||
|
straight from a git checkout or from inside the built installer image.
|
||||||
|
`common.nix` bakes `scripts/env.sh` in alongside it at a matching
|
||||||
|
relative path (`/etc/nixos-installer/env.sh` next to
|
||||||
|
`/etc/nixos-installer/installer/auto-install.sh`) so the script's own
|
||||||
|
`source "$(dirname ...)/../env.sh"` line resolves the same way in both
|
||||||
|
contexts — this is also why it's invoked from
|
||||||
|
`/etc/nixos-installer/installer/auto-install.sh` rather than a flat
|
||||||
|
`/etc/auto-install.sh`. `#!/usr/bin/env bash`, not
|
||||||
|
`#!/run/current-system/sw/bin/bash`: the latter only resolves on an
|
||||||
|
already-activated NixOS system, breaking the checked-out-file case
|
||||||
|
entirely (confirmed live: "cannot execute: required file not found" on
|
||||||
|
a non-NixOS box); `/usr/bin/env` is reliably present on both NixOS
|
||||||
|
(`environment.usrbinenv`'s own default) and any normal Linux distro.
|
||||||
|
|
||||||
|
### `scripts/secrets/`
|
||||||
|
|
||||||
|
- `scripts/secrets/sync-host-keys.sh` — generates/registers SSH host keys
|
||||||
|
and their `.sops.yaml`/`secrets/*.yaml` recipients for flake targets,
|
||||||
|
idempotently (`--all`, `<target>`, `--remove`, `--regenerate-all-keys`,
|
||||||
|
all with `--dry-run`). Stores keys as clan vars
|
||||||
|
(`vars/per-machine/<target>/openssh/`, committed and sops-encrypted) for
|
||||||
|
all flake targets. The primary tool for provisioning a new host's
|
||||||
|
secrets access — see "Creating a new machine" in
|
||||||
|
`docs/auto-installer.md`.
|
||||||
|
- `scripts/secrets/prepare-host-key.sh` — narrower predecessor: generates a
|
||||||
|
key by an arbitrary name without touching `.sops.yaml`. Still useful to
|
||||||
|
pre-generate a key before its flake target exists yet, since
|
||||||
|
`sync-host-keys.sh` can only act on targets `nixosConfigurations` already
|
||||||
|
has.
|
||||||
|
- `scripts/secrets/rotate-admin-key.sh <backup-admin-key> [--new-key-file
|
||||||
|
<path>] [--dry-run]` — rotates `.sops.yaml`'s `&admin` age key: decrypts
|
||||||
|
with a backed-up copy of the key currently trusted as `&admin` (verified
|
||||||
|
by deriving its public key and comparing, not taken on faith), replaces
|
||||||
|
the `&admin` line with a new key already present in the environment
|
||||||
|
(defaults to wherever sops/age itself would look), and runs
|
||||||
|
`sops updatekeys` on every `secrets/*.yaml`. One-way: the old key can no
|
||||||
|
longer decrypt anything re-encrypted this way. This is the automation
|
||||||
|
for the manual steps `sync-host-keys.sh`/`create-proxmox-resource.sh`
|
||||||
|
print when they bootstrap a brand-new, not-yet-trusted key on a machine
|
||||||
|
with no prior admin access.
|
||||||
|
- `scripts/secrets/backup-admin-key.sh <dest-path> [--key-file <path>]
|
||||||
|
[--force] [--dry-run]` — copies the local sops age key (source
|
||||||
|
resolution matches sops/age itself: `$SOPS_AGE_KEY` inline, then
|
||||||
|
`--key-file`, then `$SOPS_AGE_KEY_FILE`, then the XDG default) to an
|
||||||
|
arbitrary destination path with `0600` permissions, validating it's a
|
||||||
|
real age identity and round-tripping the public key before and after the
|
||||||
|
write. Refuses to overwrite an existing `<dest-path>` without `--force`.
|
||||||
|
Purely a local filesystem copy — never touches `.sops.yaml`/
|
||||||
|
`secrets/*.yaml` or the repo at all. The resulting file is exactly what
|
||||||
|
`rotate-admin-key.sh` expects as its backup-key argument.
|
||||||
|
- `scripts/secrets/sync-nix-cache-host-key.sh [--check] [--dry-run]
|
||||||
|
[--host <name>]` — detects drift between the ed25519 SSH host key
|
||||||
|
nix-cache is actually serving right now (via `ssh-keyscan`) and
|
||||||
|
`vars.nixCacheHostKey` (`variables.nix`), the value
|
||||||
|
`modules/nix-cache/remote-builder-client.nix` bakes into every real
|
||||||
|
client's declarative `programs.ssh.knownHosts` and
|
||||||
|
`configure-nix-cache-client.sh` hardcodes as its own default for
|
||||||
|
non-NixOS clients. That value has no automatic source of truth — it's
|
||||||
|
set once from whatever nix-cache's host key happened to be at the time,
|
||||||
|
and silently goes stale if the host is ever rebuilt/recreated with a new
|
||||||
|
key, breaking every client's distributed-build SSH trust with no error
|
||||||
|
that points back here. `--check` (used by `codex-maintenance.sh`, which
|
||||||
|
treats an unreachable nix-cache — e.g. from a non-LAN CI runner — as a
|
||||||
|
silent skip rather than a failure) only reports drift; the no-flags form
|
||||||
|
updates both files in place. Declarative clients still need a rebuild to
|
||||||
|
pick up the fix.
|
||||||
|
- `scripts/secrets/push-host-keys.sh [--all | <target>] [--dry-run]
|
||||||
|
[--skip-git-check]` — pushes newly-generated SSH host keys from
|
||||||
|
`host-keys/` to already-running NixOS hosts, so they can decrypt sops
|
||||||
|
secrets after a rebuild following `sync-host-keys.sh
|
||||||
|
--regenerate-all-keys`. Verifies that `.sops.yaml` and `secrets/*.yaml`
|
||||||
|
are committed and pushed to the remote first (hosts rebuild from the
|
||||||
|
remote Gitea flake, so recipient changes must land there before any key
|
||||||
|
push).
|
||||||
|
|
||||||
|
### `scripts/proxmox/`
|
||||||
|
|
||||||
|
- `scripts/proxmox/create-proxmox-resource.sh` — builds a `lxc-*`/
|
||||||
|
`proxmox-*` target's tarball/disk image and creates it on a real Proxmox
|
||||||
|
node (`pct create` against the tarball as a CT template / `qm create`+
|
||||||
|
`importdisk`), or reconfigures an existing resource's cores/memory/disk
|
||||||
|
size (`--modify`, always requires typing the VMID back to confirm).
|
||||||
|
Checks for an already-uploaded image on the node before building
|
||||||
|
(`--force-rebuild` to skip that and always rebuild), and probes
|
||||||
|
nix-cache's substituter/remote-builder reachability once up front rather
|
||||||
|
than letting every `nix build` call retry against it individually.
|
||||||
|
Refuses to create a target whose host identity already exists live on
|
||||||
|
the node (checked directly via `qm`/`pct`, not any file in this repo)
|
||||||
|
unless `--allow-duplicate-host` is passed. `--dry-run` throughout both
|
||||||
|
modes. The first time it has to bootstrap build tooling on a node (i.e.
|
||||||
|
`nix` wasn't already on its `PATH`), it also runs
|
||||||
|
`scripts/proxmox/configure-nix-cache-client.sh` there (non-fatally — a
|
||||||
|
failure just falls back to building from source / `cache.nixos.org`) so
|
||||||
|
the node substitutes from and can offload builds to nix-cache on every
|
||||||
|
subsequent run, not just this one.
|
||||||
|
- `scripts/proxmox/clone-pve1-to-pve-test.sh <vmid> [--new-vmid <id>]
|
||||||
|
[--mode snapshot|suspend|stop] [--dry-run]` — ad-hoc clone of a single
|
||||||
|
VM or CT from pve1 (production) to pve-test (sandbox) via vzdump +
|
||||||
|
qmrestore/pct restore. Streams the archive directly between nodes (no
|
||||||
|
local staging copy). Always restores with `--unique 1` (fresh MAC
|
||||||
|
addresses) since the original is still running on the LAN. Cleans up
|
||||||
|
the vzdump archive from both nodes after a successful restore. The
|
||||||
|
script's own default is pve1 → pve-test, matching CLAUDE.md's policy
|
||||||
|
(unlike `create-proxmox-resource.sh`, which defaults to production for
|
||||||
|
the operator's own unqualified use).
|
||||||
|
- `scripts/proxmox/configure-nix-cache-client.sh [--dry-run]
|
||||||
|
[--no-remote-builder] [--no-restart]` — the non-NixOS equivalent of
|
||||||
|
`modules/nix-cache/client.nix`/`remote-builder-client.nix`, for a plain
|
||||||
|
Debian machine with the Nix package manager (not NixOS) already
|
||||||
|
installed: run as root *on that machine* to add nix-cache as a
|
||||||
|
substituter in `/etc/nix/nix.conf` (`https://cache.nixos.org/` kept as
|
||||||
|
fallback) via `extra-substituters`/`extra-trusted-public-keys` so it
|
||||||
|
layers on top of whatever's already there instead of clobbering it, and,
|
||||||
|
if `/root/.ssh/nixremote` is already present (see docs/nix-cache.md
|
||||||
|
"Remote builder SSH keys"), configures it as a distributed-build
|
||||||
|
machine too and trusts nix-cache's SSH host key in
|
||||||
|
`/etc/ssh/ssh_known_hosts`. Idempotent (re-running replaces its own
|
||||||
|
marked block rather than duplicating it); restarts `nix-daemon` by
|
||||||
|
default so the change takes effect immediately.
|
||||||
|
|
||||||
|
### `scripts/ha/`
|
||||||
|
|
||||||
|
HA cluster lifecycle and operational scripts. All mutate real cluster state
|
||||||
|
when run for real — always run against pve-test first unless the operator
|
||||||
|
explicitly targets pve1.
|
||||||
|
|
||||||
|
- `scripts/ha/deploy.sh [--skip-*] [--destroy] [--dry-run]` — full
|
||||||
|
lifecycle manager: phases through bridge creation, key sync, VM creation
|
||||||
|
(via `create-proxmox-resource.sh`), NIC/disk attachment, and cluster
|
||||||
|
initialisation. `--destroy` tears it back down. Safe to rerun
|
||||||
|
idempotently; each phase can be individually skipped.
|
||||||
|
- `scripts/ha/cluster-init.sh` — one-time cluster bootstrap run **as root
|
||||||
|
on ha-server-1** after both VMs are booted. Generates/distributes the
|
||||||
|
Corosync authkey, initialises DRBD metadata, creates XFS on `/dev/drbd0`,
|
||||||
|
configures LIO iSCSI, and registers all Pacemaker resources (DRBD → XFS
|
||||||
|
→ iSCSI → NFS → VIPs).
|
||||||
|
- `scripts/ha/health.sh` — read-only cluster health snapshot: SSH
|
||||||
|
reachability, quorum, DRBD state, Pacemaker resources, and VIP port
|
||||||
|
reachability. Safe to run from the workstation at any time.
|
||||||
|
- `scripts/ha/failover.sh [--to node1|node2] [--force] [--timeout <s>]
|
||||||
|
[--dry-run]` — graceful failover by putting the active node into
|
||||||
|
Pacemaker standby and waiting for resources to appear on the target.
|
||||||
|
- `scripts/ha/acceptance-tests.sh` — T1–T7 acceptance tests (failover,
|
||||||
|
NFS/iSCSI connectivity, DRBD sync, etc.) that must all pass before the
|
||||||
|
cluster is considered production-ready.
|
||||||
|
- `scripts/ha/resize-data-disk.sh --size +NNg [--force] [--dry-run]` —
|
||||||
|
online data-disk resize: `qm resize` on both VMs, guest block-device
|
||||||
|
rescan, `drbdadm resize`, `xfs_growfs`. No downtime required.
|
||||||
|
- `scripts/ha/cluster-enable-stonith.sh` — enables the `fence_pve_ssh`
|
||||||
|
STONITH resource after the fence SSH key is deployed to both nodes and
|
||||||
|
authorised on the Proxmox host. Run once after `cluster-init.sh`.
|
||||||
|
- `scripts/ha/fence-pve-ssh.py` — Python STONITH fence agent for Pacemaker.
|
||||||
|
Deploy to `/etc/pacemaker/fence_pve_ssh` on both HA nodes (`chmod +x`).
|
||||||
|
SSHes to the Proxmox host and runs `qm stop/start <vmid>`.
|
||||||
|
|
||||||
|
### `scripts/ipa/`
|
||||||
|
|
||||||
|
- `scripts/ipa/create-nixos-ipa-host-account.sh [options] <hostname>` —
|
||||||
|
adds a NixOS host to the FreeIPA domain and produces a sops-encrypted
|
||||||
|
keytab at `secrets/<hostname>.keytab`, ready for `modules/ipa/client.nix`.
|
||||||
|
Replaces three error-prone manual steps: `ipa host-add`, `ipa-getkeytab`
|
||||||
|
(run on the DC, SCP'd back), and `sops encrypt` in the correct location
|
||||||
|
(must be at `secrets/<hostname>.keytab` for the creation rule to match).
|
||||||
|
|
||||||
|
### `scripts/lib/`
|
||||||
|
|
||||||
|
Sourced by the scripts above, never run directly:
|
||||||
|
|
||||||
|
- `nix-bootstrap.sh` — `NIX_CONFIG`/`ensure_nix_profile`, shared by
|
||||||
|
`codex-setup.sh`/`codex-maintenance.sh` and the remote build commands
|
||||||
|
`create-proxmox-resource.sh` runs over SSH.
|
||||||
|
- `nix-eval.sh` — `NIX_EVAL_FLAGS` plus `list_flake_targets`/
|
||||||
|
`flake_target_hostname` flake-introspection helpers.
|
||||||
|
- `nix-parallel.sh` — `run_nix_parallel`: fans out independent `nix eval`/
|
||||||
|
`nix build --dry-run` calls across up to `NIX_PARALLEL_JOBS` processes,
|
||||||
|
capped by available memory (~1 GB/job) rather than raw `nproc` to avoid
|
||||||
|
OOM on constrained CI runners. Used by `codex-maintenance.sh`.
|
||||||
|
- `clan-vars.sh` — helpers for reading/writing SSH host keys stored as clan
|
||||||
|
vars (`vars/per-machine/<target>/openssh/`, sops-encrypted) instead of
|
||||||
|
the gitignored `host-keys/` directory. Sourced by
|
||||||
|
`create-proxmox-resource.sh` and `sync-host-keys.sh`; depends on
|
||||||
|
`sops-age.sh` and `ssh-host-keys.sh` being sourced first.
|
||||||
|
- `ssh-host-keys.sh` — `generate_host_ed25519_key`/`ssh_pubkey_to_age`,
|
||||||
|
shared by `sync-host-keys.sh` and `prepare-host-key.sh`.
|
||||||
|
- `sops-age.sh` — `age_pubkey_from_identity_file`/`sops_yaml_admin_pubkey`/
|
||||||
|
`sops_updatekeys` plus the shared sops/age default key-file resolution,
|
||||||
|
shared by `backup-admin-key.sh`, `rotate-admin-key.sh`, and
|
||||||
|
`sync-host-keys.sh`.
|
||||||
|
- `confirm.sh` — `confirm_typed`, the "type X back to confirm" destructive-
|
||||||
|
action prompt shared by `create-proxmox-resource.sh` and
|
||||||
|
`sync-host-keys.sh`.
|
||||||
|
- `sync-host-keys-edit-sops.py` — the `.sops.yaml` anchor/key_groups editor
|
||||||
|
`sync-host-keys.sh` shells out to (see that script for why: precise,
|
||||||
|
idempotent YAML edits are impractical in bash).
|
||||||
|
|
||||||
|
### Top level
|
||||||
|
|
||||||
|
- `scripts/env.sh` — shared config (`PROXMOX_HOST`, storage pool, bridge,
|
||||||
|
default cores/memory, `NIX_CACHE_HOST`, `LAN_DOMAIN`) sourced by
|
||||||
|
`create-proxmox-resource.sh` and `scripts/installer/auto-install.sh`. Add
|
||||||
|
new cross-script config here instead of duplicating it per-script.
|
||||||
|
- `scripts/recover-hosts.sh [<hostname> ...]` — fixes sops/SSH-key/GitHub-token
|
||||||
|
issues on deployed NixOS hosts and triggers a `Switch-nix` rebuild on each.
|
||||||
|
With no args discovers every known hostname; with args checks only those.
|
||||||
|
Fixes applied automatically (prompts before rebuilding): SSH host key drift
|
||||||
|
(restores the registered key) and stale GitHub access tokens (empties the
|
||||||
|
rendered `nix-github-token.conf` so Nix falls back to unauthenticated requests
|
||||||
|
until sops-nix re-renders the correct token after the next successful rebuild).
|
||||||
|
- `scripts/gc-hosts.sh [--dry-run]` — runs `nix-collect-garbage -d` on all live
|
||||||
|
NixOS hosts (workstation first, then pve1, then all Proxmox guests). Excludes
|
||||||
|
`nix-cache` (gc-ing the shared binary cache evicts store paths other hosts
|
||||||
|
depend on). Uses passwordless sudo where available; falls back to user-level gc.
|
||||||
|
- `scripts/bump-nixpkgs-release.sh` — bumps `flake.nix`'s `nixpkgs.url`/
|
||||||
|
`home-manager.url` in place. Exists because flake input URLs can't
|
||||||
|
reference `variables.nix` (confirmed empirically — `nix flake metadata`
|
||||||
|
errors on it), so this is the closest equivalent to a single source of
|
||||||
|
truth for the tracked release.
|
||||||
|
|
||||||
|
`sync-host-keys.sh`, `create-proxmox-resource.sh`, and
|
||||||
|
`rotate-admin-key.sh` genuinely mutate real state when run for real (not
|
||||||
|
`--dry-run`): real `secrets/*.yaml` recipients, real Proxmox VMs/
|
||||||
|
containers, real revocation of decrypt access. They require the
|
||||||
|
operator's own SSH/sops access, which an agent session doesn't have — but
|
||||||
|
don't suggest running any of them non-dry-run without the operator's
|
||||||
|
explicit go-ahead even if it becomes technically reachable.
|
||||||
|
`backup-admin-key.sh` only writes a key copy to a path the operator gives
|
||||||
|
it — lower-stakes than the others, but it still handles a real private
|
||||||
|
key, so treat its destination path choice as the operator's call too.
|
||||||
|
|
||||||
## Architecture
|
## Architecture
|
||||||
|
|
||||||
`flake.nix` is the single entry point. It defines one `nixosConfigurations.<host>`
|
`flake.nix` is the single entry point. It generates one
|
||||||
attribute per machine, each built the same way:
|
`nixosConfigurations.<platform>-<buildtype>` attribute per target via the
|
||||||
|
`mkTarget` function, composed from:
|
||||||
|
|
||||||
```
|
```
|
||||||
nixosSystem {
|
nixosSystem {
|
||||||
modules = [
|
modules = [
|
||||||
disko.nixosModules.disko
|
disko.nixosModules.disko
|
||||||
./hosts/<host>/configuration.nix # host-specific config
|
sops-nix.nixosModules.sops
|
||||||
./modules/hardware-configuration/vm/<proxmox|linode>.nix
|
./modules/common/configuration.nix
|
||||||
|
./modules/platforms/${platform}.nix # what it runs on
|
||||||
|
./modules/build-types/${buildType}.nix # what it's for
|
||||||
|
hostPath # hosts/<name>/host.nix — per-machine identity
|
||||||
home-manager.nixosModules.home-manager { ... }
|
home-manager.nixosModules.home-manager { ... }
|
||||||
];
|
] ++ (client-only modules, for every buildType except "nix-cache" itself)
|
||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
Hosts currently defined in `flake.nix`: `nixos`, `docker`, `server`,
|
Platforms: `linode`, `proxmox`, `lxc`, `baremetal`. Build types: `minimal`,
|
||||||
`nix-cache`, `nix-minimal`, `pxe-boot`, `linode-minimal`. Treat `flake.nix` as
|
`nix-cache`, `docker`, `gui`, `pxe-boot`, `tailscale-router`, `tor-relay`,
|
||||||
the source of truth for which hosts exist — `README.md`, `AGENTS.md`,
|
`ha-server`. Not every combination is built — e.g. `pxe-boot` has no `linode`
|
||||||
|
variant (PXE/DHCP/TFTP need LAN L2 adjacency a Linode VPS doesn't have),
|
||||||
|
`tor-relay` only exists as `lxc-tor-relay`, `ha-server` only exists as
|
||||||
|
`proxmox-ha-server-{1,2}`, and `baremetal` only exists as `baremetal-gui`
|
||||||
|
(the real gui-host hardware — see `hosts/nixos/host.nix` and
|
||||||
|
`modules/platforms/baremetal.nix`). Treat
|
||||||
|
`flake.nix`'s
|
||||||
|
`generatedTargets` as the source
|
||||||
|
of truth for which hosts exist — `README.md`, `AGENTS.md`,
|
||||||
`docs/flake-lock-automation.md`, and the CI eval workflows
|
`docs/flake-lock-automation.md`, and the CI eval workflows
|
||||||
(`.github/workflows/check-nixos.yml`, `.gitea/workflows/check-nixos.yml`) list
|
(`.github/workflows/check-nixos.yml`, `.gitea/workflows/check-nixos.yml`) list
|
||||||
hosts by hand and can drift from it, so re-check them against `flake.nix` when
|
hosts by hand (or, for the CI workflows, evaluate the flake dynamically) and
|
||||||
adding or removing a host.
|
can drift from it, so re-check them against `flake.nix` when adding or
|
||||||
|
removing a host.
|
||||||
|
|
||||||
### Composition pattern
|
### Composition pattern
|
||||||
|
|
||||||
Every host's real configuration lives in `hosts/<host>/configuration.nix`,
|
- `hosts/<name>/host.nix` — per-machine identity **only**: hostname, hostId,
|
||||||
which is a thin list of `imports` pulling in reusable pieces from `modules/`:
|
per-machine secrets, `system.stateVersion`. These files carry no `imports`
|
||||||
|
of their own — all shared behavior comes from the platform/build-type modules
|
||||||
- `modules/common/configuration.nix` — base NixOS config imported by (almost)
|
composed in `flake.nix`, not from the host file.
|
||||||
every host: locale, users, nix settings, git. Nearly always the first import.
|
- `modules/platforms/{linode,proxmox,lxc,baremetal}.nix` — platform-specific
|
||||||
- `modules/common/home.nix` / `hosts/<host>/home.nix` — Home Manager config for
|
config: boot method, guest tooling, and the hardware config, imported
|
||||||
the `nixos` user; the `nixos` workstation has its own, other hosts share
|
directly by the platform module itself — **not** wired in from
|
||||||
`modules/common/home.nix`.
|
`flake.nix`. VM platforms use `../hardware-configuration/vm/{proxmox,linode}.nix`;
|
||||||
|
`baremetal.nix` uses `../hardware-configuration/baremetal.nix` (adapted
|
||||||
|
from a real `nixos-generate-config` run on the actual hardware, not a
|
||||||
|
vm/ file, since it isn't a VM) plus `hardware.enableRedistributableFirmware
|
||||||
|
= true` for real wifi/GPU/microcode firmware that VMs never needed.
|
||||||
|
`lxc.nix` has no hardware-configuration counterpart since containers
|
||||||
|
share the host kernel; instead it imports nixpkgs' own
|
||||||
|
`virtualisation/proxmox-lxc.nix`, which gives every `lxc-*` host a
|
||||||
|
`config.system.build.tarball` output — a plain rootfs tarball, used as a
|
||||||
|
`pct create ... vztmpl` CT template (**not** `pct restore`, which expects
|
||||||
|
`vzdump` backup-archive metadata this doesn't have), no install step —
|
||||||
|
see `docs/auto-installer.md`.
|
||||||
|
- `modules/build-types/*.nix` — what a system is for:
|
||||||
|
minimal/docker/gui/pxe-boot/nix-cache/tailscale-router/tor-relay/ha-server.
|
||||||
|
- `modules/common/configuration.nix` — base NixOS config imported by every
|
||||||
|
host: locale, users, nix settings, git.
|
||||||
|
- `modules/common/home.nix` / `hosts/nixos/home.nix` — Home Manager config for
|
||||||
|
the `nixos` user; the `nixos` workstation (`gui` build type) has its own,
|
||||||
|
other hosts share `modules/common/home.nix`.
|
||||||
- `modules/disko/proxmox.nix` — declarative disk layout (GPT: ESP + swap +
|
- `modules/disko/proxmox.nix` — declarative disk layout (GPT: ESP + swap +
|
||||||
ext4 root) via disko, used by all Proxmox-VM hosts.
|
ext4 root) via disko, used by all Proxmox-VM hosts (`proxmox-*`, not
|
||||||
|
`lxc-*`). Also carries `imageSize`/`imageName`, letting every `proxmox-*`
|
||||||
|
host be built as a standalone, `qm importdisk`-ready `.raw` image with no
|
||||||
|
install step — see `docs/proxmox-images.md`.
|
||||||
|
- `modules/disko/linode.nix` — `linode-*`'s disko config, deliberately
|
||||||
|
different in kind from the Proxmox one: Linode provisions and sizes
|
||||||
|
`/dev/sda`/`/dev/sdb` itself as whole, unpartitioned devices before the OS
|
||||||
|
boots, so this declares them with `destroy = false` (disko never wipes
|
||||||
|
them) and a bare `filesystem`/`swap` content type instead of a partition
|
||||||
|
table — idempotent against an already-provisioned disk, never destructive.
|
||||||
|
- `modules/disko/baremetal.nix` — `baremetal-gui`'s disko config: a ZFS
|
||||||
|
RAID0 (striped, no redundancy — disko's zpool `mode` defaults to `""`,
|
||||||
|
which is a plain stripe rather than `"mirror"`/`"raidz"`) root pool
|
||||||
|
across two disks, ESP + systemd-boot on the first. Device paths
|
||||||
|
(`vars.guiRootDisk1`/`guiRootDisk2`) are placeholders — fill in stable
|
||||||
|
`/dev/disk/by-id/...` paths before running disko for real.
|
||||||
|
`modules/platforms/baremetal.nix` also imports
|
||||||
|
`modules/services/zfs/enable-service.nix` for this (the `zfs_unstable`
|
||||||
|
package, autoScrub/autoSnapshot/trim) — the only other importer today is
|
||||||
|
`ha-server`'s NFS data pool, an unrelated non-root ZFS use.
|
||||||
- `modules/boot/efi.nix` — systemd-boot + EFI vars, paired with the disko module.
|
- `modules/boot/efi.nix` — systemd-boot + EFI vars, paired with the disko module.
|
||||||
- `modules/hardware-configuration/vm/{proxmox,linode}.nix` — hypervisor-specific
|
- `modules/installer/` — the auto-installer environment (ISO, also served as
|
||||||
hardware config, wired in from `flake.nix` (not from the host file).
|
PXE netboot): `common.nix` (shared config + the generated
|
||||||
- `modules/nix-cache/{client,server}.nix` + `modules/remote-builder-client.nix` —
|
`auto-install.sh`), `iso.nix`, `host-keys.nix` (optionally bakes
|
||||||
binary cache substituter + SSH remote-builder wiring; see `docs/nix-cache.md`
|
`host-keys/` into the image under `--impure`). See
|
||||||
for the full design (per-host local stores, no shared `/nix/store`, and how
|
`docs/auto-installer.md`.
|
||||||
the `nixremote` signing/SSH keys fit together).
|
- `modules/pxe-boot/stage-installer-artifacts.nix` — builds the installer's
|
||||||
- `modules/tailscale/`, `modules/docker/`, `modules/beszel/`,
|
netboot image and stages it on the `pxe-boot` host so its iPXE menu can
|
||||||
`modules/services/*` — single-purpose, single-host feature modules (e.g.
|
chain straight to it. See `docs/pxe-boot.md`.
|
||||||
`docker/enable-service.nix`, `services/zfs/enable-service.nix`,
|
- `modules/nix-cache/{client,server,remote-builder-client}.nix` — binary cache
|
||||||
`beszel/enable-agent.nix` for monitoring). Grep `hosts/*/configuration.nix`
|
substituter + SSH remote-builder wiring; see `docs/nix-cache.md` for the
|
||||||
for the `imports` list to see which modules apply to a given host.
|
full design (per-host local stores, no shared `/nix/store`, and how the
|
||||||
|
`nixremote` signing/SSH keys fit together).
|
||||||
|
- `modules/ha/` — HA cluster NixOS modules: `cluster-config.nix` (DRBD,
|
||||||
|
Corosync, Pacemaker, firewall rules, cluster-wide NFS/iSCSI port
|
||||||
|
authorisation — shared by both ha-server nodes), `pacemaker-stack.nix`
|
||||||
|
(Pacemaker + Corosync service enablement), and supporting modules. See
|
||||||
|
`docs/ha.md` for the cluster operational guide.
|
||||||
|
- `modules/ipa/client.nix` — FreeIPA client enrollment: sssd, Kerberos keytab,
|
||||||
|
and IPA host registration; imported by every real host via
|
||||||
|
`modules/common/configuration.nix`.
|
||||||
|
- `modules/beszel/enable-agent.nix` — enables beszel-agent, sets `HUB_URL`,
|
||||||
|
fixes the upstream `StateDirectory` bug, and wires the universal
|
||||||
|
`beszel-token` sops secret (from `secrets/common.yaml`) into the agent's
|
||||||
|
`environmentFile`; see `docs/beszel.md` for the full setup guide.
|
||||||
|
- `modules/tailscale/`, `modules/docker/`, `modules/networking/`,
|
||||||
|
`modules/traefik/`, `modules/tor/`, `modules/services/*` — single-purpose,
|
||||||
|
single-host feature modules (e.g. `docker/enable-service.nix`,
|
||||||
|
`services/zfs/enable-service.nix`). Grep `modules/build-types/*.nix` for
|
||||||
|
each build type's `imports` list to see which modules apply where.
|
||||||
|
|
||||||
New host = new `hosts/<name>/configuration.nix` + a matching block added to
|
New host = new `hosts/<name>/host.nix` + a matching
|
||||||
`flake.nix`'s `nixosConfigurations`, composed from existing `modules/*` pieces
|
`mkTarget { platform; buildType; hostPath; }` entry added to `flake.nix`'s
|
||||||
rather than duplicating config.
|
`generatedTargets`, composed from existing `modules/*` pieces rather than
|
||||||
|
duplicating config.
|
||||||
|
|
||||||
### Other docs worth reading before touching these areas
|
### Other docs worth reading before touching these areas
|
||||||
|
|
||||||
@@ -109,6 +543,14 @@ rather than duplicating config.
|
|||||||
handling.
|
handling.
|
||||||
- `docs/pxe-boot.md` — the `pxe-boot` host's iPXE/TFTP/HTTP boot chain and
|
- `docs/pxe-boot.md` — the `pxe-boot` host's iPXE/TFTP/HTTP boot chain and
|
||||||
directory layout under `/srv/pxe`.
|
directory layout under `/srv/pxe`.
|
||||||
|
- `docs/auto-installer.md` — the installer environment (ISO/netboot/Proxmox
|
||||||
|
LXC), `host-keys/` and the sops-nix pre-seeding problem it solves, and why
|
||||||
|
`lxc-*` hosts are deliberately excluded from its menu.
|
||||||
|
- `docs/proxmox-images.md` — building `proxmox-*` hosts as standalone `.raw`
|
||||||
|
disk images (disko's image builder) instead of installing, and deploying
|
||||||
|
the result to Proxmox.
|
||||||
- `docs/flake-lock-automation.md` — how `flake.lock` updates flow through CI
|
- `docs/flake-lock-automation.md` — how `flake.lock` updates flow through CI
|
||||||
(scheduled `nix flake update` PR + host-eval-on-PR workflow) and why hosts
|
(scheduled `nix flake update` PR + host-eval-on-PR workflow) and why hosts
|
||||||
should track the committed lock file rather than `nixos-rebuild --upgrade-all`.
|
should track the committed lock file rather than `nixos-rebuild --upgrade-all`.
|
||||||
|
- `docs/ha.md` — HA file-server cluster: DRBD + XFS + LIO iSCSI + NFS managed
|
||||||
|
by Corosync + Pacemaker; network topology; lifecycle scripts in `scripts/ha/`.
|
||||||
|
|||||||
@@ -8,29 +8,46 @@ workstation.
|
|||||||
Targets are named `<platform>-<buildtype>`, generated from two orthogonal
|
Targets are named `<platform>-<buildtype>`, generated from two orthogonal
|
||||||
pieces composed in `flake.nix`:
|
pieces composed in `flake.nix`:
|
||||||
|
|
||||||
- **Platforms** (what it runs on): `linode`, `proxmox`, `lxc`
|
- **Platforms** (what it runs on): `linode`, `proxmox`, `lxc`, `baremetal`
|
||||||
- **Build types** (what it's for): `minimal`, `nix-cache`, `server`, `docker`,
|
- **Build types** (what it's for): `minimal`, `nix-cache`, `docker`, `gui`,
|
||||||
`gui`, `pxe-boot`
|
`pxe-boot`, `tailscale-router`, `tor-relay`, `ha-server`
|
||||||
|
|
||||||
Not every combination exists — `pxe-boot` has no `linode` variant, since
|
Not every combination exists — `pxe-boot` has no `linode` variant, since
|
||||||
PXE/DHCP/TFTP need LAN L2 adjacency that a Linode VPS doesn't have. The full
|
PXE/DHCP/TFTP need LAN L2 adjacency that a Linode VPS doesn't have,
|
||||||
list:
|
`tor-relay` and `ha-server` currently only exist on `lxc`/`proxmox`, and
|
||||||
|
`baremetal` currently only exists as `baremetal-gui` (the real gui-host
|
||||||
|
hardware). The full list:
|
||||||
|
|
||||||
| Target | Purpose |
|
| Target | Purpose |
|
||||||
| --- | --- |
|
| --- | --- |
|
||||||
| `linode-minimal` | Minimal NixOS host profile on a Linode VPS (real, deployed) |
|
| `linode-minimal` | Minimal NixOS host profile on a Linode VPS |
|
||||||
| `proxmox-minimal` | Minimal NixOS host profile on Proxmox (real, deployed — previously the flat `nix-minimal` target) |
|
| `proxmox-minimal` | Minimal NixOS host profile on Proxmox — previously the flat `nix-minimal` target |
|
||||||
| `lxc-minimal` | Minimal NixOS host profile in a Proxmox LXC container |
|
| `lxc-minimal` | Minimal NixOS host profile in a Proxmox LXC container |
|
||||||
| `linode-nix-cache` / `proxmox-nix-cache` / `lxc-nix-cache` | Local Nix binary cache and remote builder (`proxmox-nix-cache` is the real, deployed one — previously the flat `nix-cache` target) |
|
| `linode-nix-cache` / `proxmox-nix-cache` / `lxc-nix-cache` | Local Nix binary cache and remote builder — previously the flat `nix-cache` target |
|
||||||
| `linode-server` / `proxmox-server` / `lxc-server` | Storage, NFS, backup, and monitoring exporter host (`proxmox-server` is the real, deployed one — previously the flat `server` target) |
|
| `linode-docker` / `proxmox-docker` / `lxc-docker` | Docker host for the main container stack — previously the flat `docker` target |
|
||||||
| `linode-docker` / `proxmox-docker` / `lxc-docker` | Docker host for the main container stack (`proxmox-docker` is the real, deployed one — previously the flat `docker` target) |
|
| `linode-gui` / `proxmox-gui` / `lxc-gui` | Cinnamon desktop workstation — previously the flat `nixos` target |
|
||||||
| `linode-gui` / `proxmox-gui` / `lxc-gui` | Cinnamon desktop workstation (`proxmox-gui` is the real, deployed one — previously the flat `nixos` target) |
|
| `baremetal-gui` | Same Cinnamon desktop workstation, on the real gui-host hardware — ZFS RAID0 root, systemd-boot |
|
||||||
| `proxmox-pxe-boot` / `lxc-pxe-boot` | HTTP/iPXE boot asset host (`proxmox-pxe-boot` is the real, deployed one — previously the flat `pxe-boot` target) |
|
| `proxmox-pxe-boot` / `lxc-pxe-boot` | HTTP/iPXE boot asset host — previously the flat `pxe-boot` target |
|
||||||
|
| `linode-tailscale-router` / `proxmox-tailscale-router` / `lxc-tailscale-router` | Tailscale subnet router + MagicDNS forwarder for the LAN |
|
||||||
|
| `lxc-tor-relay` | Tor middle relay |
|
||||||
|
| `proxmox-ha-server-1` / `proxmox-ha-server-2` | HA file-server cluster nodes — DRBD + XFS + iSCSI + NFS, managed by Corosync + Pacemaker |
|
||||||
|
|
||||||
|
Which variant of a given buildtype is actually deployed isn't tracked
|
||||||
|
anywhere in this repo — that's live infrastructure state, not something a
|
||||||
|
committed file can keep accurate, and it changes independently of the code.
|
||||||
|
Check the Proxmox node itself, or `/etc/flake-target` on a running host (see
|
||||||
|
below), if you need to know what's really out there right now.
|
||||||
|
`scripts/proxmox/create-proxmox-resource.sh`'s duplicate-host guard works the same
|
||||||
|
way: it checks the Proxmox node directly rather than any file here.
|
||||||
|
Real, production deployments live on `pve1.sweet.home`; there's a second
|
||||||
|
node, `pve-test.sweet.home`, set aside purely for scratch/test resources —
|
||||||
|
see `scripts/env.sh` (`PVE1_HOST` / `PVE_TEST_HOST`, and the
|
||||||
|
`--node`/`PROXMOX_HOST` targeting they feed into) and CLAUDE.md's Proxmox
|
||||||
|
section for which is which.
|
||||||
|
|
||||||
Each buildtype's `hosts/<name>/host.nix` carries the per-machine identity
|
Each buildtype's `hosts/<name>/host.nix` carries the per-machine identity
|
||||||
(hostname, hostId, per-machine secrets, `system.stateVersion`) that must stay
|
(hostname, hostId, per-machine secrets, `system.stateVersion`) that must stay
|
||||||
fixed regardless of which platform it's built for — see
|
fixed regardless of which platform it's built for. Every deployed host
|
||||||
`flake-target-refactor-spec.md` for the full rationale. Every deployed host
|
|
||||||
stamps its own active target name into `/etc/flake-target` at build time, so
|
stamps its own active target name into `/etc/flake-target` at build time, so
|
||||||
`nixos-rebuild switch --flake .#$(cat /etc/flake-target)` always picks up the
|
`nixos-rebuild switch --flake .#$(cat /etc/flake-target)` always picks up the
|
||||||
right one even after a platform migration changes the flake attribute name.
|
right one even after a platform migration changes the flake attribute name.
|
||||||
@@ -46,14 +63,18 @@ nix eval --json .#nixosConfigurations --apply builtins.attrNames | jq -r '.[]'
|
|||||||
| Path | Purpose |
|
| Path | Purpose |
|
||||||
| --- | --- |
|
| --- | --- |
|
||||||
| `flake.nix` | Flake inputs, the `mkTarget` platform × build-type generator, and `nixosConfigurations` outputs |
|
| `flake.nix` | Flake inputs, the `mkTarget` platform × build-type generator, and `nixosConfigurations` outputs |
|
||||||
|
| `variables.nix` | Single source of truth for shared values (LAN domain/CIDR, hostnames, timezone, primary username, storage root, NFS share subpaths/mountpoints, service ports, ...) — passed to every module and Home Manager config as the `vars` argument via `specialArgs`/`extraSpecialArgs` |
|
||||||
| `hosts/<name>/host.nix` | Per-machine identity: hostname, hostId, per-machine secrets, `system.stateVersion` |
|
| `hosts/<name>/host.nix` | Per-machine identity: hostname, hostId, per-machine secrets, `system.stateVersion` |
|
||||||
| `hosts/nixos/home.nix` | Workstation-specific Home Manager config (used by the `gui` build type) |
|
| `hosts/nixos/home.nix` | Workstation-specific Home Manager config (used by the `gui` build type) |
|
||||||
| `modules/platforms/` | Platform-specific config: virtualisation guest tools, boot method, hardware config (`linode.nix`, `proxmox.nix`, `lxc.nix`) |
|
| `modules/platforms/` | Platform-specific config: virtualisation guest tools, boot method, hardware config (`linode.nix`, `proxmox.nix`, `lxc.nix`, `baremetal.nix`) |
|
||||||
| `modules/build-types/` | Build-type-specific config: what makes a system minimal/server/docker/gui/pxe-boot/nix-cache |
|
| `modules/build-types/` | Build-type-specific config: what makes a system minimal/server/docker/gui/pxe-boot/nix-cache |
|
||||||
| `modules/common/` | Shared NixOS config, Home Manager, aliases imported by every host |
|
| `modules/common/` | Shared NixOS config, Home Manager, aliases imported by every host |
|
||||||
| `modules/nix-cache/` | Binary cache and remote builder client/server modules |
|
| `modules/nix-cache/` | Binary cache and remote builder client/server modules |
|
||||||
| `docs/` | Operational notes for cache, builders, lock updates, and boot services |
|
| `modules/installer/` | Auto-installer environment (ISO, also served as PXE netboot) — see `docs/auto-installer.md` |
|
||||||
| `scripts/` | Codex setup and validation helpers |
|
| `host-keys/` | Gitignored; only used by the auto-installer environment for pre-seeding SSH host keys before first boot — see `docs/auto-installer.md`. All deployed hosts use clan vars (`vars/per-machine/<target>/openssh/`) instead |
|
||||||
|
| `vars/per-machine/` | Clan vars: committed, sops-encrypted SSH host keys for all deployed hosts; read by `create-proxmox-resource.sh` at deploy time |
|
||||||
|
| `docs/` | Operational notes for cache, builders, lock updates, boot services, the auto-installer, and Proxmox image builds |
|
||||||
|
| `scripts/` | Codex setup, validation, host-key, release-bump, and Proxmox resource helpers |
|
||||||
|
|
||||||
## Validation
|
## Validation
|
||||||
|
|
||||||
@@ -61,10 +82,19 @@ Safe validation commands for Codex and local review:
|
|||||||
|
|
||||||
```bash
|
```bash
|
||||||
bash scripts/codex-setup.sh
|
bash scripts/codex-setup.sh
|
||||||
bash scripts/codex-maintenance.sh dry-run
|
|
||||||
bash scripts/codex-maintenance.sh
|
bash scripts/codex-maintenance.sh
|
||||||
```
|
```
|
||||||
|
|
||||||
|
`codex-maintenance.sh` with no flags (what CI runs on every push/PR) scopes
|
||||||
|
fmt-check/statix/eval to files changed against a base ref — fast, but only
|
||||||
|
as thorough as the diff. For the full sweep (every host, every package,
|
||||||
|
fmt-check and statix over the whole tree — slow, CI never runs this):
|
||||||
|
|
||||||
|
```bash
|
||||||
|
bash scripts/codex-maintenance.sh --full-check
|
||||||
|
bash scripts/codex-maintenance.sh --full-check --dry-run
|
||||||
|
```
|
||||||
|
|
||||||
For individual host evaluation:
|
For individual host evaluation:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
@@ -84,6 +114,29 @@ review sessions.
|
|||||||
client hosts.
|
client hosts.
|
||||||
- `pxe-boot` serves iPXE boot files over HTTP from `/srv/pxe`.
|
- `pxe-boot` serves iPXE boot files over HTTP from `/srv/pxe`.
|
||||||
|
|
||||||
|
### Deploying a new host
|
||||||
|
|
||||||
|
Three different paths depending on target, none of them involving a manual
|
||||||
|
`nixos-rebuild switch` from this repo:
|
||||||
|
|
||||||
|
- Most hosts: boot the auto-installer, pick the target from its menu — see
|
||||||
|
`docs/auto-installer.md`. Every menu target has a Disko config the
|
||||||
|
installer formats unconditionally (`docs/auto-installer.md`'s "Storage"
|
||||||
|
section covers how this stays non-destructive for `linode-*`, whose disks
|
||||||
|
Linode itself provisions ahead of time).
|
||||||
|
- `lxc-*` targets: not installed at all — build a ready-to-run container
|
||||||
|
tarball and `pct create` it as a CT template directly. `docs/auto-installer.md`
|
||||||
|
covers why (and the installer's menu excludes them for the same reason).
|
||||||
|
- `proxmox-*` targets: can alternatively be built as a standalone `.raw`
|
||||||
|
disk image and attached to a new VM with no install step — see
|
||||||
|
`docs/proxmox-images.md`.
|
||||||
|
|
||||||
|
`scripts/proxmox/create-proxmox-resource.sh --type lxc|vm --host <name>` automates
|
||||||
|
either of the last two end to end (host-key registration, building the
|
||||||
|
image directly on the Proxmox node itself, `pct create`/`qm create`), with
|
||||||
|
`--dry-run` and a guard against duplicating an already-deployed host's
|
||||||
|
identity. See its `--help`.
|
||||||
|
|
||||||
## Security Notes
|
## Security Notes
|
||||||
|
|
||||||
Do not commit tokens, private keys, live credentials, or new password hashes
|
Do not commit tokens, private keys, live credentials, or new password hashes
|
||||||
@@ -104,7 +157,15 @@ enabled via `git config core.hooksPath .githooks`, done automatically by
|
|||||||
`scripts/codex-setup.sh`) runs `gitleaks protect --staged` to catch mistakes
|
`scripts/codex-setup.sh`) runs `gitleaks protect --staged` to catch mistakes
|
||||||
before they're committed.
|
before they're committed.
|
||||||
|
|
||||||
This repository's git *history* still contains secrets committed before this
|
The auto-installer environment is the one deliberate exception to
|
||||||
migration (see `remove-sensetive-info-refactor.md`) — those are being
|
sops-nix-everywhere: it has a hardcoded login password instead (no stable
|
||||||
scrubbed and rotated separately; don't treat the repo as safe to make public
|
per-boot host key for sops-nix to derive from on ephemeral media) — see
|
||||||
until that's finished.
|
"Host keys" in `docs/auto-installer.md` for why, and how the private keys it
|
||||||
|
*does* pre-seed for target hosts stay out of git via the gitignored
|
||||||
|
`host-keys/` directory. All deployed hosts use clan vars
|
||||||
|
(`vars/per-machine/<target>/openssh/`, committed and sops-encrypted) for
|
||||||
|
their SSH host keys.
|
||||||
|
|
||||||
|
This repository's git *history* still contains secrets committed before the
|
||||||
|
sops-nix migration — those are being scrubbed and rotated separately; don't
|
||||||
|
treat the repo as safe to make public until that's finished.
|
||||||
|
|||||||
@@ -0,0 +1,25 @@
|
|||||||
|
-----BEGIN CERTIFICATE-----
|
||||||
|
MIIESDCCArCgAwIBAgIBATANBgkqhkiG9w0BAQsFADA1MRMwEQYDVQQKDApTV0VF
|
||||||
|
VC5IT01FMR4wHAYDVQQDDBVDZXJ0aWZpY2F0ZSBBdXRob3JpdHkwHhcNMjYwNzI2
|
||||||
|
MjExMzQxWhcNNDYwNzI2MjExMzQxWjA1MRMwEQYDVQQKDApTV0VFVC5IT01FMR4w
|
||||||
|
HAYDVQQDDBVDZXJ0aWZpY2F0ZSBBdXRob3JpdHkwggGiMA0GCSqGSIb3DQEBAQUA
|
||||||
|
A4IBjwAwggGKAoIBgQCzljYktbHdMGVJ6Wq0XQJuHLN6dkCSOgtoIzQtriPQkkNI
|
||||||
|
uo28LwobaiQQ8sX4kGRH/BTKnH8QlId/jug4Uc+sDHnABYu++AiOhPbBX8gCpRQ0
|
||||||
|
hebBjZiktHSBUEJR31siWOVdBoKBDJEoxehx7XUXvcxIJcaRN+LHYjO86nJN55HB
|
||||||
|
VwFU2JcYDk98c+144dFJxXdr++MjWe4Z/oVVU8JHIOtNtKhVhvij6oOSWxcYoJO/
|
||||||
|
S80LRj1vx/o6o/3G6bYug7PjY7JjZk/Oj61whijZkcsoO1MXSYI6UywJZGflv+ZB
|
||||||
|
7HyufdYAsK3WhE8O2FX3/kq64Ol83HNtoR8Dt68rTg1xpW6K45jS6iDPKueYGkb0
|
||||||
|
oSx7e++90VAW2PDhj6QQ3JJ4O5VQwrrecekJzUrAean0FOEbmgyi4PsEp1Vk6LDQ
|
||||||
|
SsIn1x0euyxVivQMlzNX2XrZL3urn1BNPqAdntXQMkR0Wl8sbUiJPe0kxG52CGXs
|
||||||
|
6yfNEXbPmVGcC0TBdGECAwEAAaNjMGEwHQYDVR0OBBYEFLh5QbI1UWMH0WR4z8bG
|
||||||
|
lhrOX3X5MB8GA1UdIwQYMBaAFLh5QbI1UWMH0WR4z8bGlhrOX3X5MA8GA1UdEwEB
|
||||||
|
/wQFMAMBAf8wDgYDVR0PAQH/BAQDAgHGMA0GCSqGSIb3DQEBCwUAA4IBgQCVodVN
|
||||||
|
owwo53OQe02QhtEbIur2PL7zIfvhvCTRD4J8gwpbMIqT7JQK0tV6Mvsg2L8yTb2O
|
||||||
|
KjrWeLKHGWaZZlhGSPTbkMFdb/Ls8M9FSnkc2bwcdWW3Z1lOiCjBYYqwLCG6JhvB
|
||||||
|
5SXVwWNJwXeasL2m7oFTSwhsqPpARJ2t25u2N35o+tqIoCjijKwkmEOT66N9EAbu
|
||||||
|
2VQjtYZWPkBtP4YCe0Ey6u4oy7sy8ThNAjOylZok+J4JW7QEFjK4Q/emhA4aQq5H
|
||||||
|
gg9qgMuG+5oi6D1g2Wy+fMTRBaukJtLYZbBpQMQhMYWg44uPp/2bbNPTID/nV1KB
|
||||||
|
GcPyHaskcVxPdYWxAPMwk3AeJXWyOq7atAPTF5sbk0kQQf2m+vyOqcli5CxRMUgV
|
||||||
|
rcyi9l6+dZW4U+38Q0ET5M3OuxNI4hA7kVY2cfTakXWNqh97+TIHnstblDhAxECK
|
||||||
|
6ZLMJQYUy7LqJTX84H27CBWLexEMjXwdr5HCV88Fj6mAK0fRufnIw5FeneA=
|
||||||
|
-----END CERTIFICATE-----
|
||||||
@@ -0,0 +1,313 @@
|
|||||||
|
# Auto-installer
|
||||||
|
|
||||||
|
This flake builds a self-contained NixOS installer environment that can
|
||||||
|
install any host exposed by its own `nixosConfigurations`. It was migrated
|
||||||
|
from a formerly-separate `nix-auto-installer` repo — everything it did now
|
||||||
|
lives here.
|
||||||
|
|
||||||
|
The installer provides a small NixOS install environment (ISO, or the same
|
||||||
|
image netbooted via PXE) with SSH access, Git support, and an interactive
|
||||||
|
installation script.
|
||||||
|
Logging in as any user (root or `nixos`) runs
|
||||||
|
`/etc/nixos-installer/installer/auto-install.sh` (the same file as
|
||||||
|
`scripts/installer/auto-install.sh` in this repo — see "Installer process"
|
||||||
|
below for why it's baked in at that path rather than a flat
|
||||||
|
`/etc/auto-install.sh`), discovers available hosts from this same flake,
|
||||||
|
lets the operator choose a target, applies that host's Disko storage
|
||||||
|
configuration, installs NixOS, and reboots.
|
||||||
|
|
||||||
|
**This applies to every `nixosConfigurations` target except `lxc-*` hosts —
|
||||||
|
see "LXC hosts" immediately below for why those are different.**
|
||||||
|
|
||||||
|
## LXC hosts
|
||||||
|
|
||||||
|
`lxc-*` targets (`lxc-minimal`, `lxc-nix-cache`, `lxc-server`, `lxc-docker`,
|
||||||
|
`lxc-gui`, `lxc-pxe-boot`, `lxc-tailscale-router`, `lxc-tor-relay`) are **not** installed via `auto-install.sh` — the
|
||||||
|
interactive menu deliberately excludes them. Don't try to select one there;
|
||||||
|
`nixos-install` would bind-mount `/` onto `/mnt` (LXC containers have no raw
|
||||||
|
disk to partition) and then refuse to touch the filesystem it's currently
|
||||||
|
running on — it's designed to protect exactly this case, so it just fails.
|
||||||
|
|
||||||
|
`modules/platforms/lxc.nix` imports nixpkgs' own
|
||||||
|
`virtualisation/proxmox-lxc.nix` module, which gives every `lxc-*` host a
|
||||||
|
`config.system.build.tarball` output — a complete, directly Proxmox-importable
|
||||||
|
container image, no install step at all:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
nix build .#nixosConfigurations.lxc-minimal.config.system.build.tarball
|
||||||
|
```
|
||||||
|
|
||||||
|
This is a plain rootfs tarball, not a `vzdump` backup archive — restoring it
|
||||||
|
with `pct restore` fails ("archive contains no configuration file"), since
|
||||||
|
that command expects backup-archive metadata this tarball doesn't have. Use
|
||||||
|
it as a CT *template* instead: drop it under Proxmox's template storage
|
||||||
|
(conventionally `/var/lib/vz/template/cache/` for the `local` storage, or
|
||||||
|
the GUI's "Create CT" → upload-as-template flow) and create a container
|
||||||
|
from it, supplying all config on the command line since a template has none
|
||||||
|
of its own:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
pct create <vmid> local:vztmpl/<file>.tar.xz \
|
||||||
|
--unprivileged 1 --features nesting=1,keyctl=1 \
|
||||||
|
--rootfs local-lvm:8 --hostname <name> --cores 2 --memory 2048 --swap 2048 \
|
||||||
|
--net0 name=eth0,bridge=vmbr0,ip=dhcp
|
||||||
|
pct start <vmid>
|
||||||
|
```
|
||||||
|
|
||||||
|
Every one of those extra flags is load-bearing, confirmed by actually
|
||||||
|
booting one:
|
||||||
|
|
||||||
|
- `--unprivileged 1` — `modules/platforms/lxc.nix` sets
|
||||||
|
`proxmoxLXC.privileged = false`, so the image assumes it's running
|
||||||
|
unprivileged. `pct create`'s own CLI default for this flag is
|
||||||
|
privileged (unlike the web UI, whose checkbox defaults the other way)
|
||||||
|
— omit it and you get a privileged container running a NixOS config
|
||||||
|
that assumes unprivileged, a real mismatch.
|
||||||
|
- `--features nesting=1,keyctl=1` — required for a modern (v247+)
|
||||||
|
systemd guest to boot unprivileged at all. Without it, AppArmor denies
|
||||||
|
the nested user namespaces and credential mounts systemd routinely
|
||||||
|
uses (even plain getty units) — every getty crash-loops on a denied
|
||||||
|
`/run/credentials/*` mount every ~3s (this is what garbage on the
|
||||||
|
console turns out to be) while core services like `nsncd` fail the
|
||||||
|
same way, and the system never finishes activating.
|
||||||
|
- `--swap 2048` — `--memory` doesn't touch swap; it silently stays at
|
||||||
|
Proxmox's own 512M default otherwise. Match it to `--memory` unless
|
||||||
|
you deliberately want otherwise.
|
||||||
|
|
||||||
|
First boot runs `boot.postBootCommands` (registers the Nix store DB and
|
||||||
|
system profile) — there's no separate activation step to run yourself.
|
||||||
|
`scripts/proxmox/create-proxmox-resource.sh --type lxc --host <name>` automates all
|
||||||
|
of this (host-key handling, building the tarball directly on the Proxmox
|
||||||
|
node itself, `pct create` with the flags above) — see its `--help`.
|
||||||
|
|
||||||
|
Host keys still need pre-seeding the same way as any other host — the
|
||||||
|
sops-nix activation-vs-first-boot race is identical regardless of how the
|
||||||
|
image reaches the machine. Unlike the ISO/PXE installer (where
|
||||||
|
`modules/installer/host-keys.nix` bakes *every* `host-keys/` entry into
|
||||||
|
`/etc/host-keys/` for `auto-install.sh` to pick from and copy at install
|
||||||
|
time — see "Host keys" below), an `lxc-*` tarball has no install step to
|
||||||
|
copy anything during, so `modules/platforms/lxc.nix` bakes this *one*
|
||||||
|
target's key straight into `/etc/ssh/ssh_host_ed25519_key(.pub)` directly,
|
||||||
|
keyed by its own exact flake target name (`config.environment.etc` can't
|
||||||
|
be read back from within a module still contributing to it, so this comes
|
||||||
|
in via `specialArgs.flakeTarget`, set by `flake.nix`'s `mkTarget`):
|
||||||
|
|
||||||
|
```sh
|
||||||
|
NIXOS_HOST_KEYS_DIR="$(pwd)/host-keys" \
|
||||||
|
nix build .#nixosConfigurations.lxc-nix-cache.config.system.build.tarball --impure
|
||||||
|
```
|
||||||
|
|
||||||
|
Confirmed the hard way: without this, the tarball's own built-in system
|
||||||
|
just generates a fresh host key at first boot like any host would, which
|
||||||
|
can never match whatever `.sops.yaml` actually trusts for that target —
|
||||||
|
`sops-install-secrets` fails with `Error getting data key: 0 successful
|
||||||
|
groups required, got 0`, and *every* secret (including this host's own
|
||||||
|
login) permanently fails to decrypt, silently — no error in the boot log
|
||||||
|
at all, since the activation step that would install secrets only runs on
|
||||||
|
a from-scratch first activation and skips silently once `/run/current-system`
|
||||||
|
already exists. `scripts/proxmox/create-proxmox-resource.sh` always builds with
|
||||||
|
`NIXOS_HOST_KEYS_DIR` set for this reason.
|
||||||
|
|
||||||
|
## Layout
|
||||||
|
|
||||||
|
- `modules/installer/common.nix` — shared by every installer target: SSH
|
||||||
|
access, users, the generated `/etc/auto-install.sh` script, and the
|
||||||
|
`programs.bash.loginShellInit` hook that runs it on login.
|
||||||
|
- `modules/installer/iso.nix` — ISO/netboot-specific: imports the stock
|
||||||
|
`installation-cd-minimal.nix` module plus `common.nix`. Also used, paired
|
||||||
|
with `netboot-minimal.nix`, to build the PXE netboot variant (see
|
||||||
|
`docs/pxe-boot.md`).
|
||||||
|
- `modules/installer/host-keys.nix` — optionally bakes pre-generated SSH
|
||||||
|
host keys into the image; see "Host keys" below.
|
||||||
|
- `scripts/secrets/sync-host-keys.sh` — admin-workstation tool that generates,
|
||||||
|
registers, and (via `--remove`/`--regenerate-all-keys`) retires host
|
||||||
|
keys; see "Creating a New Machine" below.
|
||||||
|
- `scripts/secrets/prepare-host-key.sh` — narrower predecessor: generates a single
|
||||||
|
key by an arbitrary name without touching `.sops.yaml`. Still useful for
|
||||||
|
pre-generating a key *before* its flake target exists (`sync-host-keys.sh`
|
||||||
|
can only act on targets `nixosConfigurations` already has); otherwise
|
||||||
|
`sync-host-keys.sh` does the same thing and more.
|
||||||
|
|
||||||
|
Flake outputs:
|
||||||
|
|
||||||
|
```nix
|
||||||
|
nixosConfigurations.installer # ISO/netboot installer image
|
||||||
|
|
||||||
|
packages.x86_64-linux.iso # installer ISO/netboot image
|
||||||
|
packages.x86_64-linux.pxe # auto-installer netboot bundle (kernel + initrd + ipxe script)
|
||||||
|
packages.x86_64-linux.pxe-minimal # vanilla NixOS minimal netboot bundle (no installer wiring)
|
||||||
|
```
|
||||||
|
|
||||||
|
```sh
|
||||||
|
nix build .#iso
|
||||||
|
nix build .#pxe
|
||||||
|
nix build .#pxe-minimal
|
||||||
|
```
|
||||||
|
|
||||||
|
There's no `nixosConfigurations.proxmox-lxc` (installer-boots-as-an-LXC-
|
||||||
|
container) or `packages.x86_64-linux.lxc`/`.all` anymore. Both existed only
|
||||||
|
to let the installer itself run as an LXC container so you could
|
||||||
|
`nixos-install` some *other* host from within it — but LXC targets are
|
||||||
|
excluded from the install menu (same bind-mount problem as any LXC
|
||||||
|
`nixos-install`), and now have their own direct tarball path anyway (see
|
||||||
|
"LXC hosts" above), which left the installer's own LXC form with no real
|
||||||
|
use case.
|
||||||
|
|
||||||
|
The `pxe` variant is also built automatically as part of the `pxe-boot` host
|
||||||
|
itself (`modules/pxe-boot/stage-installer-artifacts.nix`) and served over
|
||||||
|
iPXE as the menu's "NixOS Auto-Installer" entry — see `docs/pxe-boot.md`.
|
||||||
|
That same host also builds and serves `packages.x86_64-linux.pxe-minimal`,
|
||||||
|
a vanilla NixOS minimal netboot image with none of this auto-installer's
|
||||||
|
wiring, as a separate "NixOS Minimal" menu entry — also documented in
|
||||||
|
`docs/pxe-boot.md`, not covered further here since it's not this installer.
|
||||||
|
|
||||||
|
## Host keys
|
||||||
|
|
||||||
|
`sops-nix` derives each host's decryption key from its own
|
||||||
|
`/etc/ssh/ssh_host_ed25519_key`, generated at **activation** time — before
|
||||||
|
systemd would otherwise generate one on first boot. Without pre-seeding this
|
||||||
|
key, secrets (including the root/nixos login password) fail to decrypt on a
|
||||||
|
genuinely fresh install.
|
||||||
|
|
||||||
|
Generated host keys live in `host-keys/` at the repo root (`ssh_host_ed25519_key`
|
||||||
|
+ `.pub` pairs per hostname). This directory is **gitignored on purpose** —
|
||||||
|
private key material must never be committed — which also means flakes can't
|
||||||
|
see it through a normal relative path. `modules/installer/host-keys.nix`
|
||||||
|
reads it through `builtins.getEnv`, which Nix silently returns as an empty
|
||||||
|
string under normal (non-`--impure`) evaluation, so the module is a no-op —
|
||||||
|
safe by default, including in CI — unless explicitly opted into:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
NIXOS_HOST_KEYS_DIR="$(pwd)/host-keys" nix build .#iso --impure
|
||||||
|
```
|
||||||
|
|
||||||
|
When built this way, every key currently in `host-keys/` is baked into the
|
||||||
|
image at `/etc/host-keys/<hostname>_ssh_host_ed25519_key(.pub)`, and
|
||||||
|
`auto-install.sh` automatically installs whichever one matches the flake
|
||||||
|
target selected at install time — no manual per-host scp step needed.
|
||||||
|
|
||||||
|
**Trade-off, accepted deliberately for this LAN-only setup:** baking keys in
|
||||||
|
means every key present in `host-keys/` at build time becomes readable by
|
||||||
|
anyone who can reach the built image — including, for the PXE variant, anyone
|
||||||
|
who can reach the `pxe-boot` host's unauthenticated HTTP server. This is
|
||||||
|
considered acceptable here because `pxe-boot` sits behind LAN-only network
|
||||||
|
infrastructure, not the open internet. If that ever changes, reconsider this
|
||||||
|
default.
|
||||||
|
|
||||||
|
`auto-install.sh` still supports the older manual path as a fallback: if a
|
||||||
|
host's key isn't baked in (`/etc/host-keys`), it checks `/root/host-keys`
|
||||||
|
next, where you can `scp` a key in after boot, same as before this migration.
|
||||||
|
If neither has it and the script is running interactively (an actual
|
||||||
|
operator at the other end of stdin, not an unattended run), it prompts for
|
||||||
|
an arbitrary directory to check (a mounted USB stick, another filesystem,
|
||||||
|
etc.) and copies the key pair into `/root/host-keys` from there if found.
|
||||||
|
|
||||||
|
## Storage
|
||||||
|
|
||||||
|
Disk partitioning is handled by Disko — the installer has no hardcoded
|
||||||
|
`parted`/`mkfs`/`mkswap`/`mount` commands, and `auto-install.sh` runs
|
||||||
|
`disko --mode destroy,format,mount` unconditionally, no branching on whether
|
||||||
|
the target has a Disko config. Every host reachable through this menu has
|
||||||
|
one:
|
||||||
|
|
||||||
|
- `proxmox-*` (`modules/disko/proxmox.nix`): a real GPT partition table
|
||||||
|
(ESP + swap + root) on `/dev/sda`.
|
||||||
|
- `linode-*` (`modules/disko/linode.nix`): Linode provisions and sizes
|
||||||
|
`/dev/sda`/`/dev/sdb` itself as whole, unpartitioned block devices before
|
||||||
|
the OS ever boots, so this declares them with `destroy = false` (skips
|
||||||
|
disko's wipe stage for these disks entirely — see the option's own docs)
|
||||||
|
and a bare `filesystem`/`swap` content type with no partition table, and
|
||||||
|
the format step it does run only calls `mkfs`/`mkswap` if `blkid` shows
|
||||||
|
the device isn't already formatted — a re-run against an
|
||||||
|
already-provisioned Linode disk is a no-op, not a wipe.
|
||||||
|
|
||||||
|
`lxc-*` is the only category without one — it's excluded from this menu
|
||||||
|
entirely (see "LXC hosts" above), so it never reaches this code path.
|
||||||
|
|
||||||
|
## Installer process
|
||||||
|
|
||||||
|
`scripts/installer/auto-install.sh` is a real, version-controlled shell
|
||||||
|
script — not an inline Nix string. It sources `scripts/env.sh` for
|
||||||
|
`LAN_DOMAIN` itself (same as every other script in `scripts/`), so it
|
||||||
|
behaves identically whether it's run straight from a git checkout (e.g.
|
||||||
|
manually, from a stock NixOS ISO that isn't this repo's own installer
|
||||||
|
image) or from inside the built installer image. That's also why it's
|
||||||
|
baked in at `/etc/nixos-installer/installer/auto-install.sh` rather than a
|
||||||
|
flat `/etc/auto-install.sh` — `modules/installer/common.nix` bakes
|
||||||
|
`scripts/env.sh` in alongside it at `/etc/nixos-installer/env.sh`,
|
||||||
|
preserving the same relative layout (`installer/auto-install.sh` ->
|
||||||
|
`../env.sh`) the checked-out repo has, so the script's own
|
||||||
|
`source ".../env.sh"` line resolves correctly in both places without any
|
||||||
|
Nix-level templating.
|
||||||
|
|
||||||
|
Once running, it:
|
||||||
|
|
||||||
|
1. Queries `nixosConfigurations` from this flake over the network (`git+https://<lanDomain>/beatzaplenty/nixos.git`) — this happens at *install* time, not build time, so a generic installer image always sees whatever hosts are currently committed, without needing a rebuild.
|
||||||
|
2. Presents them as a menu; confirms the choice.
|
||||||
|
3. Skips the `nix-cache` substituter when installing a `nix-cache` host itself (consistent with that host's own runtime config).
|
||||||
|
4. Runs `disko --mode destroy,format,mount` (see "Storage" above — every host reachable through this menu has a Disko config, so this is unconditional).
|
||||||
|
5. Installs the target's SSH host key from `/etc/host-keys` or `/root/host-keys` (see "Host keys" above).
|
||||||
|
6. Runs `nixos-install --flake <url>#<choice> --no-root-password`.
|
||||||
|
7. Cleans up and reboots.
|
||||||
|
|
||||||
|
## Creating a new machine
|
||||||
|
|
||||||
|
Do this instead of jumping straight to a plain install whenever the target
|
||||||
|
host consumes any sops-nix secret — as of this writing, that's every host
|
||||||
|
(`modules/common/configuration.nix` puts the root/nixos password hash and the
|
||||||
|
GitHub token behind sops-nix for all of them).
|
||||||
|
|
||||||
|
1. **Add the flake target** — `hosts/<name>/host.nix` plus the matching
|
||||||
|
`mkTarget { ... }` entry in `flake.nix`'s `generatedTargets` (see
|
||||||
|
"Composition pattern" in `CLAUDE.md`). No secrets involved yet, so this
|
||||||
|
is safe to commit on its own if you want a clean history.
|
||||||
|
|
||||||
|
2. **On your admin workstation, generate and register its host key:**
|
||||||
|
|
||||||
|
```sh
|
||||||
|
./scripts/secrets/sync-host-keys.sh <flake-target>
|
||||||
|
```
|
||||||
|
|
||||||
|
This generates `host-keys/<flake-target>_ssh_host_ed25519_key(.pub)`,
|
||||||
|
adds it as a new `.sops.yaml` anchor, works out which `secrets/*.yaml`
|
||||||
|
files this specific host actually references (from its own
|
||||||
|
`config.sops.secrets`, not guessed), adds it to each one's
|
||||||
|
`key_groups`, and re-encrypts them with `sops updatekeys` — no manual
|
||||||
|
YAML editing. Safe to re-run; it only fills in what's missing.
|
||||||
|
|
||||||
|
Doing this for every host that needs one at once — after adding several
|
||||||
|
new targets, or just to catch up any that were missed — is
|
||||||
|
`./scripts/secrets/sync-host-keys.sh --all`. See `scripts/secrets/sync-host-keys.sh --help`
|
||||||
|
for its other modes (`--remove`, `--regenerate-all-keys`).
|
||||||
|
|
||||||
|
3. **Commit and push.** The flake build the installer uses has to see the
|
||||||
|
new recipient before you install, or decryption fails on first boot
|
||||||
|
regardless of the next step.
|
||||||
|
|
||||||
|
4. **Build the installer image with keys baked in** (or reuse an already-serving `pxe-boot` host, which does this automatically once redeployed):
|
||||||
|
|
||||||
|
```sh
|
||||||
|
NIXOS_HOST_KEYS_DIR="$(pwd)/host-keys" nix build .#iso --impure
|
||||||
|
```
|
||||||
|
|
||||||
|
5. **Boot it on the target machine**, log in, select the new host's flake
|
||||||
|
target from the menu, confirm. `auto-install.sh` finds the baked-in key,
|
||||||
|
runs Disko + `nixos-install`, and reboots.
|
||||||
|
|
||||||
|
6. **Verify after reboot:**
|
||||||
|
|
||||||
|
```sh
|
||||||
|
ssh <new-host> ls /run/secrets/
|
||||||
|
```
|
||||||
|
|
||||||
|
If that's empty or login fails, the host's age key most likely wasn't in
|
||||||
|
`.sops.yaml` (or wasn't re-encrypted into the secrets file it needs) when
|
||||||
|
`nixos-install` ran — fix `.sops.yaml`/`secrets/*.yaml`, push, then re-run
|
||||||
|
`nixos-install --flake .#<hostname> --no-root-password` from a rescue
|
||||||
|
environment against the existing `/mnt`, or just redo the install.
|
||||||
|
|
||||||
|
## Safety
|
||||||
|
|
||||||
|
This installer is destructive: `disko --mode destroy,format,mount` erases
|
||||||
|
any disk defined by the selected host's Disko configuration. Always verify
|
||||||
|
the selected host profile and target machine before confirming.
|
||||||
+104
@@ -0,0 +1,104 @@
|
|||||||
|
# Beszel agent
|
||||||
|
|
||||||
|
[Beszel](https://github.com/henrygd/beszel) is the monitoring dashboard used
|
||||||
|
in this LAN. The hub runs as a Docker container on `docker.sweet.home` (port
|
||||||
|
`vars.ports.beszelHub`, 8090). Each monitored NixOS host runs a
|
||||||
|
`beszel-agent` that connects back to the hub.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## How it works
|
||||||
|
|
||||||
|
Everything is handled by a single module:
|
||||||
|
|
||||||
|
**`modules/beszel/enable-agent.nix`** — imported by a build type. It:
|
||||||
|
- Enables `beszel-agent`
|
||||||
|
- Sets `HUB_URL` to `docker.sweet.home:8090`
|
||||||
|
- Sets `KEY` from `vars.beszelHubKey` (`variables.nix`) — the hub's SSH
|
||||||
|
public key, shared by every agent. Update `beszelHubKey` if the docker
|
||||||
|
host is ever rebuilt and the hub generates a new keypair.
|
||||||
|
- Reads the universal `beszel-token` from `secrets/common.yaml` via sops
|
||||||
|
and passes it to the agent as `TOKEN` in an env file
|
||||||
|
- Fixes an upstream bug where the agent couldn't persist its hub-pairing
|
||||||
|
fingerprint across restarts (adds a real `StateDirectory`)
|
||||||
|
|
||||||
|
A host file needs no beszel configuration at all — just import the module
|
||||||
|
in the build type and add the system in the hub UI.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Adding beszel to a new build type
|
||||||
|
|
||||||
|
Add `../beszel/enable-agent.nix` to the `imports` list in
|
||||||
|
`modules/build-types/<type>.nix`:
|
||||||
|
|
||||||
|
```nix
|
||||||
|
imports = [
|
||||||
|
../beszel/enable-agent.nix
|
||||||
|
# ... other imports
|
||||||
|
];
|
||||||
|
```
|
||||||
|
|
||||||
|
That's the only change required. The host file needs nothing.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Adding a new system to the hub
|
||||||
|
|
||||||
|
1. Rebuild and deploy the host with its build type importing `enable-agent.nix`.
|
||||||
|
2. Open the beszel hub (`http://docker.sweet.home:8090`).
|
||||||
|
3. Go to **Systems → Add system**, enter the host's IP and the default port
|
||||||
|
(45876). The agent will connect and the system will appear as active.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## One-time setup: add the token to `secrets/common.yaml`
|
||||||
|
|
||||||
|
The universal token is stored once in the common secrets file, shared by all
|
||||||
|
agents. Only needed once, not per-host:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
sops secrets/common.yaml
|
||||||
|
```
|
||||||
|
|
||||||
|
Add:
|
||||||
|
```yaml
|
||||||
|
beszel-token: <token from the beszel hub Settings → Keys>
|
||||||
|
```
|
||||||
|
|
||||||
|
`secrets/common.yaml` is already a sops recipient for every host via their
|
||||||
|
SSH host keys, so no additional sops recipient setup is needed.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Optional: monitoring extra filesystems
|
||||||
|
|
||||||
|
To report disk usage for a mount beyond the root filesystem, add
|
||||||
|
`EXTRA_FILESYSTEMS` in the host file:
|
||||||
|
|
||||||
|
```nix
|
||||||
|
services.beszel.agent.environment = {
|
||||||
|
EXTRA_FILESYSTEMS = "/mnt/data"; # colon-separated for multiple paths
|
||||||
|
};
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Optional: monitoring Docker containers
|
||||||
|
|
||||||
|
`enable-agent.nix` has a commented-out line for Docker monitoring:
|
||||||
|
|
||||||
|
```nix
|
||||||
|
#DOCKER_HOST = "tcp://docker-socket-proxy:2375";
|
||||||
|
```
|
||||||
|
|
||||||
|
Uncomment it if the host runs docker-socket-proxy and you want per-container
|
||||||
|
stats. Hosts without Docker should leave it commented out.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## If the hub key changes
|
||||||
|
|
||||||
|
If the docker host is ever rebuilt and beszel generates a new SSH keypair,
|
||||||
|
update `beszelHubKey` in `variables.nix` and rebuild all beszel-enabled hosts.
|
||||||
|
The new key is visible in the beszel hub under **Settings → Keys**.
|
||||||
@@ -8,9 +8,14 @@ and to verify that declared NixOS hosts still evaluate after dependency updates.
|
|||||||
- A scheduled workflow runs `nix flake update` once per week.
|
- A scheduled workflow runs `nix flake update` once per week.
|
||||||
- On GitHub, any resulting `flake.lock` change is proposed through a pull request.
|
- On GitHub, any resulting `flake.lock` change is proposed through a pull request.
|
||||||
- On Gitea, the workflow can commit and push `flake.lock` directly when PR automation is not configured.
|
- On Gitea, the workflow can commit and push `flake.lock` directly when PR automation is not configured.
|
||||||
- A separate CI workflow evaluates every configured host before merge, listed
|
- A separate CI workflow runs `scripts/codex-maintenance.sh` before merge.
|
||||||
dynamically via `nix eval --json .#nixosConfigurations --apply builtins.attrNames`
|
Its default mode scopes eval to the hosts/packages a change can affect,
|
||||||
rather than hand-enumerated, so it can't drift as `<platform>-<buildtype>`
|
determined from a git diff against the PR base — but a `flake.lock` change
|
||||||
|
is treated as repo-wide and always falls back to evaluating every host, so
|
||||||
|
a lock-file update PR still gets full coverage. Hosts are still listed
|
||||||
|
dynamically via
|
||||||
|
`nix eval --json .#nixosConfigurations --apply builtins.attrNames` rather
|
||||||
|
than hand-enumerated, so that fallback can't drift as `<platform>-<buildtype>`
|
||||||
targets are added or removed. See `README.md` for the current target list.
|
targets are added or removed. See `README.md` for the current target list.
|
||||||
|
|
||||||
## Why hosts should stop using `--upgrade-all`
|
## Why hosts should stop using `--upgrade-all`
|
||||||
|
|||||||
+160
@@ -0,0 +1,160 @@
|
|||||||
|
# HA File-Server Cluster
|
||||||
|
|
||||||
|
Two `proxmox-ha-server-{1,2}` VMs form an active/passive file-server cluster:
|
||||||
|
DRBD replicates a block device between nodes; Corosync + Pacemaker manage
|
||||||
|
failover; XFS, LIO iSCSI, and NFS are brought up as a collocated resource
|
||||||
|
group on whichever node holds the DRBD Primary role.
|
||||||
|
|
||||||
|
NixOS modules: `modules/ha/`. Lifecycle scripts: `scripts/ha/`.
|
||||||
|
Cluster-wide constants: `variables.nix` (`haServer*` vars).
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Network layout
|
||||||
|
|
||||||
|
Three subnets — all internal to pve1 (`vmbr0`/`vmbr1`/`vmbr2`):
|
||||||
|
|
||||||
|
| Subnet | VLAN | CIDR | Bridge | Purpose |
|
||||||
|
|---|---|---|---|---|
|
||||||
|
| LAN | 2 | `192.168.2.0/24` | `vmbr0` | Management, LAN NFS |
|
||||||
|
| Cluster | 10 | `192.168.10.224/29` | `vmbr1` | Corosync ring0 + DRBD replication |
|
||||||
|
| Storage-client | 20 | `192.168.20.0/24` | `vmbr2` | NFS + iSCSI for docker/swarm |
|
||||||
|
|
||||||
|
Each HA VM has three NICs: `ens18` (LAN/vmbr0), `ens19` (cluster/vmbr1),
|
||||||
|
`ens20` (storage-client/vmbr2). See `docs/ip-addressing.md` for all IPs.
|
||||||
|
|
||||||
|
Corosync ring0 uses the cluster NIC; ring1 (backup heartbeat) uses the LAN
|
||||||
|
NIC. DRBD replicates over the cluster NIC. No storage traffic crosses the LAN.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Pacemaker resources
|
||||||
|
|
||||||
|
All resources run collocated on whichever node is Primary, in this order:
|
||||||
|
|
||||||
|
```
|
||||||
|
ms-drbd0 (promotable DRBD clone)
|
||||||
|
→ xfs-data (XFS mount on /dev/drbd0 → /srv/ha-data)
|
||||||
|
→ iscsi-target (targetctl)
|
||||||
|
→ nfs-server (nfs-server.service)
|
||||||
|
→ vip-lan (192.168.2.229/24 on vmbr0 — NFS for LAN clients)
|
||||||
|
→ vip-storage (192.168.20.229/24 on vmbr2 — NFS + iSCSI for VLAN 20)
|
||||||
|
```
|
||||||
|
|
||||||
|
`vip-lan` serves pxe-boot and other LAN-only NFS clients.
|
||||||
|
`vip-storage` serves docker and any future swarm nodes; iSCSI is available on
|
||||||
|
VLAN 20 but NFS is preferred for multi-host volume sharing.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## DRBD fencing
|
||||||
|
|
||||||
|
`fencing resource-only` with `crm-fence-peer.sh`/`crm-unfence-peer.sh`
|
||||||
|
wrappers (`modules/ha/cluster-config.nix`). The DRBD kernel module invokes
|
||||||
|
these via the User Mode Helper with a minimal PATH; the wrappers prepend
|
||||||
|
`/run/current-system/sw/bin` before exec-ing the real handlers so Pacemaker
|
||||||
|
tools (`cibadmin`, `crm_mon`, etc.) are found.
|
||||||
|
|
||||||
|
STONITH is initially disabled (`stonith-enabled: false`,
|
||||||
|
`no-quorum-policy: ignore`). Enable it once the `fence_pve_ssh` fence agent
|
||||||
|
(`scripts/ha/fence-pve-ssh.py`) is deployed and authorised:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
scripts/ha/cluster-enable-stonith.sh # run as root on ha-server-1
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Deploying the cluster from scratch
|
||||||
|
|
||||||
|
Use `scripts/ha/deploy.sh` — it orchestrates all phases:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# Against pve-test (safe — Claude's default target):
|
||||||
|
scripts/ha/deploy.sh --node "$PVE_TEST_HOST" [--dry-run]
|
||||||
|
|
||||||
|
# Against pve1 (production — requires explicit operator go-ahead):
|
||||||
|
scripts/ha/deploy.sh --node "$PVE1_HOST"
|
||||||
|
```
|
||||||
|
|
||||||
|
Phases (each skippable with `--skip-<phase>`):
|
||||||
|
1. `ensure-bridge` — creates `vmbr1`/`vmbr2` on the Proxmox node if absent
|
||||||
|
2. `sync-keys` — generates SSH host keys for both nodes; registers sops recipients
|
||||||
|
3. `create-vms` — builds disk images, creates VMs via `create-proxmox-resource.sh`
|
||||||
|
4. `add-hardware` — attaches storage NIC and DRBD data disk to each VM
|
||||||
|
5. `init-cluster` — runs `scripts/ha/cluster-init.sh` on ha-server-1
|
||||||
|
|
||||||
|
`--destroy` runs the teardown sequence.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Day-to-day operations
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# Read-only health check (safe from workstation):
|
||||||
|
scripts/ha/health.sh
|
||||||
|
|
||||||
|
# Graceful failover (prompts for confirmation):
|
||||||
|
scripts/ha/failover.sh [--to node1|node2]
|
||||||
|
|
||||||
|
# Online data-disk growth (no downtime):
|
||||||
|
scripts/ha/resize-data-disk.sh --size +20G
|
||||||
|
|
||||||
|
# Acceptance tests (run after any significant change):
|
||||||
|
scripts/ha/acceptance-tests.sh
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Adding FreeIPA host accounts
|
||||||
|
|
||||||
|
IPA host registration is automated:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
scripts/ipa/create-nixos-ipa-host-account.sh <hostname>
|
||||||
|
```
|
||||||
|
|
||||||
|
This runs `ipa host-add`, fetches a keytab from the domain controller, and
|
||||||
|
writes a sops-encrypted `secrets/<hostname>.keytab` in one step. The module
|
||||||
|
`modules/ipa/client.nix` (imported by every host via
|
||||||
|
`modules/common/configuration.nix`) consumes the keytab via sops-nix.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Storage layout
|
||||||
|
|
||||||
|
```
|
||||||
|
/srv/ha-data/
|
||||||
|
docker/
|
||||||
|
config/ NFS → docker:/mnt/docker/config
|
||||||
|
databases/ NFS → docker:/mnt/docker/databases
|
||||||
|
volumes/ NFS → docker:/mnt/docker/volumes
|
||||||
|
nextcloud-data/ NFS → docker:/mnt/docker/nextcloud-data
|
||||||
|
proxmox/
|
||||||
|
iso/ NFS → pve1 ISO storage
|
||||||
|
lxc/ NFS → pve1 CT template storage
|
||||||
|
pxe-boot/
|
||||||
|
images/ NFS → pxe-boot:/srv/pxe/http/images (PXE assets)
|
||||||
|
raspi/
|
||||||
|
volumes/ NFS → raspi NFS mounts
|
||||||
|
iscsi-lun.img iSCSI fileio backstore (VLAN 20 only, not in active use)
|
||||||
|
```
|
||||||
|
|
||||||
|
All shares are defined in `variables.nix` (`vars.nfsShares.*`). The NFS
|
||||||
|
export list lives in `modules/ha/nfs-exports.nix`.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Key variables
|
||||||
|
|
||||||
|
| Variable | Description |
|
||||||
|
|---|---|
|
||||||
|
| `vars.haServer1Ip` / `vars.haServer2Ip` | LAN management IPs |
|
||||||
|
| `vars.haServer1StorageIp` / `vars.haServer2StorageIp` | Cluster NIC IPs (DRBD/Corosync ring0) |
|
||||||
|
| `vars.haServerLanVip` | Pacemaker `vip-lan` — NFS for LAN (192.168.2.229) |
|
||||||
|
| `vars.haServerVip` | Pacemaker `vip-storage` — NFS + iSCSI for VLAN 20 (192.168.20.229) |
|
||||||
|
| `vars.haLanNfsFqdn` | FQDN of `vip-lan`: `ha-vip-lan.sweet.home` |
|
||||||
|
| `vars.haStorageRoot` | XFS mount point: `/srv/ha-data` |
|
||||||
|
| `vars.haServerDrbdDisk` | Block device for DRBD backing store |
|
||||||
|
| `vars.haStorageCidr` | Cluster subnet CIDR (`192.168.10.224/29`) |
|
||||||
|
| `vars.haClientCidr` | Storage-client subnet CIDR (`192.168.20.0/24`) |
|
||||||
@@ -0,0 +1,330 @@
|
|||||||
|
# Docker Swarm Cutover Plan
|
||||||
|
|
||||||
|
Migration guide for moving containerised services from the existing single-host
|
||||||
|
Docker LXC container (CT 105, `docker.sweet.home`, 192.168.2.225) to the new
|
||||||
|
Docker Swarm cluster (`ha-docker-1` / `ha-docker-2`, 192.168.2.230–231).
|
||||||
|
|
||||||
|
CT 105 stays running throughout. Services migrate one stack at a time.
|
||||||
|
Roll back any stack by restarting it on CT 105 if anything goes wrong.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Prerequisites
|
||||||
|
|
||||||
|
- Swarm cluster deployed and healthy (`scripts/docker-swarm/deploy.sh`).
|
||||||
|
- Both nodes show `Ready / Active / Manager` in `docker node ls`.
|
||||||
|
- NFS mounts healthy on both swarm nodes (`/mnt/docker/config`, `/mnt/docker/databases`, `/mnt/docker/volumes`).
|
||||||
|
- Access to FreeIPA DNS admin to update A records during cutover.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 1. Traefik — switch to Docker log rotation
|
||||||
|
|
||||||
|
**Current state (CT 105):** Traefik writes access logs to the NFS volume at
|
||||||
|
`/mnt/docker/volumes/traefik-data/logs/`. `modules/traefik/rotate-logs.nix`
|
||||||
|
rotates those files via `logrotate`.
|
||||||
|
|
||||||
|
**Swarm approach:** Remove file-based access logging from Traefik's static
|
||||||
|
config and rely on Docker's json-file log driver with built-in rotation.
|
||||||
|
Traefik container logs (including access events) then live under
|
||||||
|
`/var/lib/docker/containers/<id>/` on the node running Traefik.
|
||||||
|
|
||||||
|
### Steps
|
||||||
|
|
||||||
|
**1a.** In the Traefik stack definition, add logging config to the service:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
services:
|
||||||
|
traefik:
|
||||||
|
logging:
|
||||||
|
driver: "json-file"
|
||||||
|
options:
|
||||||
|
max-size: "100m"
|
||||||
|
max-file: "20"
|
||||||
|
```
|
||||||
|
|
||||||
|
**1b.** In `traefik.yml` (Traefik's static config), remove the `accessLog`
|
||||||
|
file path if present. To keep structured access logs, use Traefik's
|
||||||
|
`accessLog.format: json` with no `filePath` — logs then go to stdout and are
|
||||||
|
captured by the json-file driver above.
|
||||||
|
|
||||||
|
**1c.** Deploy Traefik to the swarm:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# On either swarm manager:
|
||||||
|
docker stack deploy -c /mnt/docker/config/traefik/docker-compose.yml traefik
|
||||||
|
```
|
||||||
|
|
||||||
|
Traefik should be deployed as a **global mode** service so it runs on all
|
||||||
|
swarm nodes and handles ingress on whichever node a request arrives at:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
services:
|
||||||
|
traefik:
|
||||||
|
deploy:
|
||||||
|
mode: global
|
||||||
|
placement:
|
||||||
|
constraints:
|
||||||
|
- node.role == manager
|
||||||
|
```
|
||||||
|
|
||||||
|
**1d.** After confirming Traefik works on the swarm, remove
|
||||||
|
`traefik/rotate-logs.nix` from the `docker` build type in
|
||||||
|
`modules/build-types/docker.nix` and rebuild CT 105.
|
||||||
|
|
||||||
|
**DNS:** Update `docker.sweet.home` and any service FQDNs that point at
|
||||||
|
192.168.2.225 to a swarm VIP or round-robin A records once Traefik is running
|
||||||
|
on the swarm. See section 8 (DNS cutover).
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 2. Nextcloud — migrate cron job to sidecar container
|
||||||
|
|
||||||
|
**Current state (CT 105):** `modules/docker/nextcloud-cron-job.nix` runs a
|
||||||
|
systemd timer every 5 minutes that calls:
|
||||||
|
```bash
|
||||||
|
docker exec nextcloud-webapp php ./cron.php
|
||||||
|
```
|
||||||
|
|
||||||
|
**Swarm problem:** `docker exec` only works against the local daemon. If
|
||||||
|
Nextcloud is scheduled on the other swarm node, the exec fails silently and
|
||||||
|
cron never runs.
|
||||||
|
|
||||||
|
**Swarm approach:** Add a `nextcloud-cron` sidecar container to the Nextcloud
|
||||||
|
stack definition, pinned to the same node as the main Nextcloud container via
|
||||||
|
placement constraints.
|
||||||
|
|
||||||
|
### Steps
|
||||||
|
|
||||||
|
**2a.** Choose which swarm node will host Nextcloud (e.g. `ha-docker-1`).
|
||||||
|
Label that node:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# On either swarm manager:
|
||||||
|
docker node update --label-add nextcloud=true ha-docker-1
|
||||||
|
```
|
||||||
|
|
||||||
|
**2b.** In the Nextcloud stack compose file, add the sidecar and pin both
|
||||||
|
services to the labelled node:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
services:
|
||||||
|
nextcloud-webapp:
|
||||||
|
image: nextcloud:production # pin same version as CT 105
|
||||||
|
deploy:
|
||||||
|
replicas: 1
|
||||||
|
placement:
|
||||||
|
constraints:
|
||||||
|
- node.labels.nextcloud == true
|
||||||
|
# ... existing volumes, env, networks ...
|
||||||
|
|
||||||
|
nextcloud-cron:
|
||||||
|
image: nextcloud:production # same image, different entrypoint
|
||||||
|
entrypoint: /cron.sh
|
||||||
|
deploy:
|
||||||
|
replicas: 1
|
||||||
|
placement:
|
||||||
|
constraints:
|
||||||
|
- node.labels.nextcloud == true # must co-locate with webapp
|
||||||
|
volumes:
|
||||||
|
# Same data volume as nextcloud-webapp so cron sees the same files.
|
||||||
|
- nextcloud-data:/var/www/html
|
||||||
|
# No ports exposed — cron only runs PHP inside the container.
|
||||||
|
```
|
||||||
|
|
||||||
|
`/cron.sh` is Nextcloud's built-in cron entrypoint. It runs
|
||||||
|
`php -f /var/www/html/cron.php` in a loop, sleeping for 5 minutes between
|
||||||
|
runs — identical to the current systemd timer.
|
||||||
|
|
||||||
|
**2c.** Migrate Nextcloud's data volume to the swarm:
|
||||||
|
|
||||||
|
```
|
||||||
|
/mnt/docker/volumes/nextcloud-data/ → already on NFS, no migration needed
|
||||||
|
/mnt/docker/databases/nextcloud/ → already on NFS, no migration needed
|
||||||
|
```
|
||||||
|
|
||||||
|
The NFS paths are identical on the swarm nodes (`mount-data.nix` mounts the
|
||||||
|
same shares from the same VIP). Stop Nextcloud on CT 105, deploy on the
|
||||||
|
swarm, confirm it starts cleanly.
|
||||||
|
|
||||||
|
**2d.** Remove `nextcloud-cron-job.nix` from `modules/build-types/docker.nix`
|
||||||
|
and rebuild CT 105 after confirming Nextcloud works on the swarm.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 3. docker-health-to-gotify — update for swarm awareness
|
||||||
|
|
||||||
|
**Current state (CT 105):** The script at
|
||||||
|
`/home/nixos/docker/monitoring/gotify/docker-health-to-gotify.sh` runs every
|
||||||
|
minute, calls `docker ps --filter health=unhealthy`, and notifies Gotify.
|
||||||
|
|
||||||
|
**Swarm behaviour:** The same script runs on both swarm nodes independently,
|
||||||
|
each monitoring its own local Docker daemon. This gives per-node coverage
|
||||||
|
across the swarm.
|
||||||
|
|
||||||
|
**Changes needed in the script** (edit the copy on the NFS volume — it takes
|
||||||
|
effect on both nodes simultaneously on the next timer fire):
|
||||||
|
|
||||||
|
### 3a. Strip the Swarm task suffix from service names
|
||||||
|
|
||||||
|
In swarm mode, `docker ps --format '{{.Names}}'` returns names like
|
||||||
|
`nextcloud-webapp.1.abc123xyz`. The notification should show `nextcloud-webapp`,
|
||||||
|
not the full task name.
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# Before:
|
||||||
|
CONTAINER_NAME=$(docker ps --format '{{.Names}}' ...)
|
||||||
|
|
||||||
|
# After:
|
||||||
|
CONTAINER_NAME=$(docker ps --format '{{.Names}}' ... | cut -d. -f1)
|
||||||
|
```
|
||||||
|
|
||||||
|
### 3b. Include the reporting node in the Gotify message
|
||||||
|
|
||||||
|
Add `$(hostname)` to the notification payload so you know which swarm node
|
||||||
|
detected the problem:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
MESSAGE="[$(hostname)] ${CONTAINER_NAME} is unhealthy"
|
||||||
|
```
|
||||||
|
|
||||||
|
### 3c. Extend to catch swarm service replica failures
|
||||||
|
|
||||||
|
`docker ps` only shows what's running locally. If a service has zero healthy
|
||||||
|
replicas (task crash-looping) it may not show up on either node's `docker ps`
|
||||||
|
at the same moment. Add a swarm-level check:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# Run only on managers (both ha-docker nodes are managers):
|
||||||
|
if docker info --format '{{.Swarm.ControlAvailable}}' 2>/dev/null | grep -q true; then
|
||||||
|
# Find services where running replicas < desired replicas
|
||||||
|
docker service ls --format '{{.Name}}\t{{.Replicas}}' | \
|
||||||
|
awk -F'\t' '$2 !~ /^[0-9]+\/[0-9]+$/ || split($2,a,"/") && a[1] < a[2] { print $1, $2 }' | \
|
||||||
|
while read -r svc_name replicas; do
|
||||||
|
# Send Gotify notification for degraded service
|
||||||
|
curl -s -X POST "${GOTIFY_URL}/message" \
|
||||||
|
-H "X-Gotify-Key: ${GOTIFY_TOKEN}" \
|
||||||
|
-d "title=Swarm service degraded" \
|
||||||
|
-d "message=[$(hostname)] ${svc_name}: ${replicas} replicas"
|
||||||
|
done
|
||||||
|
fi
|
||||||
|
```
|
||||||
|
|
||||||
|
This catches the case where a service's desired replicas are not running
|
||||||
|
(e.g. OOM kill, image pull failure) — a failure mode that doesn't produce a
|
||||||
|
Docker health event on any node.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 4. Passbolt migration
|
||||||
|
|
||||||
|
Passbolt has strict data integrity requirements. Migrate with care:
|
||||||
|
|
||||||
|
1. **Backup first** — `docker exec passbolt-webapp php /usr/share/php/passbolt/bin/cake passbolt export_keys` and a database dump.
|
||||||
|
2. Database is on NFS (`/mnt/docker/databases/passbolt/`) — no data copy needed.
|
||||||
|
3. Pin Passbolt to a specific node: `docker node update --label-add passbolt=true ha-docker-1`
|
||||||
|
4. Add placement constraint `node.labels.passbolt == true` to the Passbolt stack.
|
||||||
|
5. Stop on CT 105, deploy on swarm, verify login works.
|
||||||
|
6. Test email delivery and 2FA.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 5. Gitea migration
|
||||||
|
|
||||||
|
Gitea's data directory is on NFS (`/mnt/docker/volumes/gitea-data/`).
|
||||||
|
|
||||||
|
1. Stop Gitea on CT 105: `docker stop gitea`
|
||||||
|
2. Deploy to swarm with placement constraint (pin to `ha-docker-1` initially).
|
||||||
|
3. Verify web UI and SSH clone/push work.
|
||||||
|
4. Update DNS: `gitea.lan.ddnsgeek.com` → swarm Traefik endpoint.
|
||||||
|
5. Update the flake remote URL in `variables.nix` (`giteaDomain`) if the address changes.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 6. Other services
|
||||||
|
|
||||||
|
Deploy remaining services (Grafana, InfluxDB, NodeRed, Prometheus, etc.)
|
||||||
|
as swarm stacks. Most have no special migration concern — they use NFS
|
||||||
|
volumes already on the shared storage.
|
||||||
|
|
||||||
|
Services with stateful databases (PostgreSQL, MariaDB) should follow the
|
||||||
|
pattern: stop on CT 105, confirm NFS database directory is intact, deploy on
|
||||||
|
swarm, verify.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 7. Monitoring — Beszel
|
||||||
|
|
||||||
|
The Beszel hub runs on CT 105 (`docker.sweet.home:8090`). Both swarm nodes
|
||||||
|
run `beszel-agent` (from `modules/beszel/enable-agent.nix`), pointing at the
|
||||||
|
existing hub URL.
|
||||||
|
|
||||||
|
No migration needed for Beszel itself during the container migration. Once
|
||||||
|
all services are on the swarm, you may wish to move the Beszel hub too (as a
|
||||||
|
swarm service with a placement constraint) but this is optional.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 8. DNS cutover
|
||||||
|
|
||||||
|
When a service is confirmed working on the swarm, update the FreeIPA DNS
|
||||||
|
A record from the CT 105 IP (192.168.2.225) to a swarm node IP or, when a
|
||||||
|
shared Traefik frontend is in place, to a round-robin record across both nodes.
|
||||||
|
|
||||||
|
**Recommended approach — Traefik as the single entry point:**
|
||||||
|
|
||||||
|
```
|
||||||
|
service.lan.ddnsgeek.com → Traefik on swarm (global mode)
|
||||||
|
docker.sweet.home → keep as 192.168.2.225 (CT 105) until fully decommissioned
|
||||||
|
```
|
||||||
|
|
||||||
|
For LAN-only services using `*.sweet.home` names, update FreeIPA directly:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# On domain-controller (or via SSH):
|
||||||
|
ipa dnsrecord-mod sweet.home nextcloud --a-rec=192.168.2.230
|
||||||
|
# Add 192.168.2.231 as a second A record for round-robin (optional):
|
||||||
|
ipa dnsrecord-add sweet.home nextcloud --a-rec=192.168.2.231
|
||||||
|
```
|
||||||
|
|
||||||
|
Services behind Traefik don't need their own DNS updates — only Traefik's
|
||||||
|
own entry point IPs need to change.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 9. NixOS cleanup — CT 105
|
||||||
|
|
||||||
|
Once all services are migrated:
|
||||||
|
|
||||||
|
**Remove from `modules/build-types/docker.nix`:**
|
||||||
|
- `../docker/nextcloud-cron-job.nix` — replaced by sidecar container
|
||||||
|
- `../traefik/rotate-logs.nix` — replaced by Docker log driver
|
||||||
|
|
||||||
|
**Keep in `modules/build-types/docker.nix` until CT 105 is decommissioned:**
|
||||||
|
- `../docker/docker-health-to-gotify.nix` — still monitors CT 105's own daemon
|
||||||
|
- Everything else
|
||||||
|
|
||||||
|
**When decommissioning CT 105:**
|
||||||
|
1. Confirm all NFS volumes are in use only by swarm services (not CT 105).
|
||||||
|
2. Stop CT 105: `pct stop 105` on pve1.
|
||||||
|
3. Archive/remove the `lxc-docker` and `proxmox-docker` targets from `flake.nix`.
|
||||||
|
4. Remove `hosts/docker/`, `modules/build-types/docker.nix`, and `modules/docker/`.
|
||||||
|
5. Update `variables.nix` to remove `dockerIp`, `dockerStorageIp`, `dockerHost`
|
||||||
|
(or reassign `dockerHost` to point at a swarm node for Beszel hub resolution).
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Rollback
|
||||||
|
|
||||||
|
Any stack can be rolled back to CT 105 independently:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# On CT 105:
|
||||||
|
docker start <service-name>
|
||||||
|
# Update DNS A record back to 192.168.2.225
|
||||||
|
ipa dnsrecord-mod sweet.home <service> --a-rec=192.168.2.225
|
||||||
|
```
|
||||||
|
|
||||||
|
CT 105 remains running throughout the cutover. Only decommission it after
|
||||||
|
every service is confirmed stable on the swarm and you have run one full
|
||||||
|
backup cycle from the new hosts.
|
||||||
@@ -0,0 +1,228 @@
|
|||||||
|
# IP Addressing Scheme
|
||||||
|
|
||||||
|
## Subnets
|
||||||
|
|
||||||
|
| Subnet | VLAN | CIDR | Purpose | Routed? |
|
||||||
|
|---|---|---|---|---|
|
||||||
|
| LAN | 2 (native/untagged) | `192.168.2.0/24` | General LAN — clients and infrastructure | Yes (gateway .254) |
|
||||||
|
| Cluster | 10 | `192.168.10.224/29` | HA file server DRBD replication + Corosync heartbeat | No — internal `vmbr1` only, no uplink |
|
||||||
|
| Storage client | 20 | `192.168.20.0/24` | HA file server NFS (and iSCSI if needed) — docker and swarm nodes mount from VIP here | No — internal `vmbr2` only, no uplink |
|
||||||
|
| Swarm cluster | 30 | `192.168.30.0/24` | Docker Swarm gossip (TCP/UDP 7946) + VXLAN overlay (UDP 4789) | No — internal `vmbr3` only, no uplink |
|
||||||
|
|
||||||
|
When expanded to a second Proxmox node, VLAN 10 (cluster), VLAN 20 (storage-client), and VLAN 30 (swarm) all share
|
||||||
|
the same inter-node trunk NIC via 802.1q VLAN tagging — different VLAN IDs, same physical cable.
|
||||||
|
|
||||||
|
The cluster and storage-client subnets never leave pve1. `vmbr1` and `vmbr2` are Proxmox Linux
|
||||||
|
bridges with no physical port attached; traffic between guests on each bridge stays in-kernel.
|
||||||
|
|
||||||
|
VLAN IDs match the third octet of each subnet (VLAN 2 → 192.168.**2**.x, VLAN 10 → 192.168.**10**.x,
|
||||||
|
VLAN 20 → 192.168.**20**.x). The host octet is consistent across all subnets — e.g. ha-node1
|
||||||
|
is always `.228`: `192.168.2.228` (LAN), `192.168.10.228` (cluster), `192.168.20.228` (storage client).
|
||||||
|
|
||||||
|
**Protocol separation** (enforced by firewall on HA nodes):
|
||||||
|
- NFS (ports 111, 2049, 20048): both subnets, each restricted to its own CIDR
|
||||||
|
- VLAN 2 only → `vip-lan` (192.168.2.229) — pxe-boot and other LAN clients
|
||||||
|
- VLAN 20 only → `vip-storage` (192.168.20.229) — docker, future swarm nodes
|
||||||
|
- iSCSI (port 3260): VLAN 20 only — available but not in active use; NFS is preferred
|
||||||
|
for multi-host access (shared volumes across a Docker Swarm require a shared filesystem,
|
||||||
|
not per-host block devices)
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## DNS Zones
|
||||||
|
|
||||||
|
FreeIPA (domain-controller.sweet.home) is authoritative for all zones.
|
||||||
|
|
||||||
|
Four zones correspond to the four subnets. All zones are internal only; no external delegation.
|
||||||
|
|
||||||
|
### sweet.home — VLAN 2 (192.168.2.x)
|
||||||
|
|
||||||
|
General LAN zone. All infrastructure hostnames live here.
|
||||||
|
|
||||||
|
| Hostname | A record | Notes |
|
||||||
|
|---|---|---|
|
||||||
|
| `domain-controller.sweet.home` | `192.168.2.253` | FreeIPA / KDC / DNS |
|
||||||
|
| `ha-vip-lan.sweet.home` | `192.168.2.229` | Pacemaker `vip-lan` — NFS for LAN clients |
|
||||||
|
| `ha-server-1.sweet.home` | `192.168.2.228` | HA node 1 management NIC |
|
||||||
|
| `ha-server-2.sweet.home` | `192.168.2.227` | HA node 2 management NIC |
|
||||||
|
| `server.sweet.home` | `192.168.2.226` | Current ZFS/NFS server (retiring) |
|
||||||
|
| `docker.sweet.home` | `192.168.2.225` | Docker/Traefik host |
|
||||||
|
| `nix-cache.sweet.home` | `192.168.2.224` | Nix binary cache + remote builder |
|
||||||
|
| `pxe-boot.sweet.home` | `192.168.2.223` | PXE / TFTP / HTTP netboot |
|
||||||
|
| `tailscale-router.sweet.home` | `192.168.2.222` | Tailscale exit node |
|
||||||
|
| `tor-relay.sweet.home` | `192.168.2.221` | Tor relay |
|
||||||
|
| `pdm.sweet.home` | `192.168.2.220` | Proxmox Deploy Manager |
|
||||||
|
| `nixos.sweet.home` | `192.168.2.39` | Bare-metal workstation (DHCP) |
|
||||||
|
| `pve1.sweet.home` | `192.168.2.245` | Proxmox VE hypervisor |
|
||||||
|
| `pbs.sweet.home` | `192.168.2.244` | Proxmox Backup Server |
|
||||||
|
|
||||||
|
PTR records exist for all static hosts. The workstation (`nixos.sweet.home`) is
|
||||||
|
DHCP-assigned; its PTR is omitted.
|
||||||
|
|
||||||
|
### cluster.home — VLAN 10 (192.168.10.x)
|
||||||
|
|
||||||
|
Internal only — Corosync ring0 heartbeat and DRBD replication between HA nodes.
|
||||||
|
No VIP exists on this subnet (DRBD/Corosync endpoints are static per-node IPs).
|
||||||
|
|
||||||
|
| Hostname | A record | Notes |
|
||||||
|
|---|---|---|
|
||||||
|
| `ha-server-1.cluster.home` | `192.168.10.228` | HA node 1 cluster NIC (ens19 / vmbr1) |
|
||||||
|
| `ha-server-2.cluster.home` | `192.168.10.227` | HA node 2 cluster NIC (ens19 / vmbr1) |
|
||||||
|
|
||||||
|
PTR records exist for both. DNS here is for debugging convenience — DRBD and
|
||||||
|
Corosync use the IPs from the NixOS config directly, not DNS.
|
||||||
|
|
||||||
|
### storage.home — VLAN 20 (192.168.20.x)
|
||||||
|
|
||||||
|
Internal only — NFS (and iSCSI) client access to the HA storage VIP. NFS clients
|
||||||
|
mount from **`nfs.storage.home`** (the Pacemaker floating VIP) so mounts survive
|
||||||
|
failover transparently without reconfiguration.
|
||||||
|
|
||||||
|
| Hostname | A record | Notes |
|
||||||
|
|---|---|---|
|
||||||
|
| `nfs.storage.home` | `192.168.20.229` | Pacemaker `vip-storage` — NFS + iSCSI VIP |
|
||||||
|
| `ha-server-1.storage.home` | `192.168.20.228` | HA node 1 storage-client NIC (ens20 / vmbr2) |
|
||||||
|
| `ha-server-2.storage.home` | `192.168.20.227` | HA node 2 storage-client NIC (ens20 / vmbr2) |
|
||||||
|
| `docker.storage.home` | `192.168.20.225` | Docker host storage-client NIC (eth1 / vmbr2) |
|
||||||
|
| `server.storage.home` | `192.168.20.226` | server VM storage-client NIC (decommissioned — remove DNS record after VM is destroyed) |
|
||||||
|
|
||||||
|
PTR records exist for all five. Remove `server.storage.home`, `server.sweet.home`,
|
||||||
|
and their PTRs from FreeIPA DNS once the server VM is destroyed.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## LAN — 192.168.2.0/24
|
||||||
|
|
||||||
|
### Address map
|
||||||
|
|
||||||
|
| Range | Purpose |
|
||||||
|
|---|---|
|
||||||
|
| .1–.9 | Reserved, never assign |
|
||||||
|
| .10–.59 | Client DHCP pool (router-assigned) |
|
||||||
|
| .60–.219 | Unallocated buffer |
|
||||||
|
| .220–.229 | Virtual nodes (VMs / LXC containers) |
|
||||||
|
| .230–.239 | Expansion buffer (reserved, unallocated) |
|
||||||
|
| .240–.249 | Physical nodes (bare-metal hosts) |
|
||||||
|
| .250–.253 | Network services |
|
||||||
|
| .254 | Router / gateway |
|
||||||
|
|
||||||
|
### Network services (.250–.253)
|
||||||
|
|
||||||
|
| IP | Hostname | Role |
|
||||||
|
|---|---|---|
|
||||||
|
| `192.168.2.254` | router | Gateway (TP-Link) |
|
||||||
|
| `192.168.2.253` | domain-controller | FreeIPA — authoritative DNS for `sweet.home`, Kerberos, LDAP |
|
||||||
|
| `192.168.2.250`–`.252` | — | Reserved for future network services |
|
||||||
|
|
||||||
|
### Physical nodes (.240–.249)
|
||||||
|
|
||||||
|
| IP | Hostname | Role |
|
||||||
|
|---|---|---|
|
||||||
|
| `192.168.2.245` | pve1 | Proxmox VE hypervisor |
|
||||||
|
| `192.168.2.244` | pbs | Proxmox Backup Server |
|
||||||
|
| `192.168.2.243` | nixos | Bare-metal workstation (`baremetal-gui`) |
|
||||||
|
| `192.168.2.246`–`.249` | — | Reserved — second Proxmox node and associated services |
|
||||||
|
| `192.168.2.240`–`.242` | — | Reserved |
|
||||||
|
|
||||||
|
pve1 sits mid-range deliberately so a second Proxmox node can slot in on either side.
|
||||||
|
|
||||||
|
### Virtual nodes (.220–.229)
|
||||||
|
|
||||||
|
All VMs and LXC containers run on pve1.
|
||||||
|
|
||||||
|
| IP | Hostname | Role | Status |
|
||||||
|
|---|---|---|---|
|
||||||
|
| `192.168.2.229` | ha-vip-lan | HA file server LAN floating VIP (Pacemaker `vip-lan`) — LAN iSCSI + NFS | Active |
|
||||||
|
| `192.168.2.228` | ha-node1 | HA file server node 1 — management NIC | Active |
|
||||||
|
| `192.168.2.227` | ha-node2 | HA file server node 2 — management NIC | Active |
|
||||||
|
| `192.168.2.226` | server | Former NFS/ZFS file server — decommissioned | Removed from flake |
|
||||||
|
| `192.168.2.225` | docker | Docker / Traefik stack (CT 105 — existing single-host) | Active |
|
||||||
|
| `192.168.2.224` | nix-cache | Nix binary cache + remote builder | Active |
|
||||||
|
| `192.168.2.223` | pxe-boot | PXE / TFTP / HTTP netboot server | Active |
|
||||||
|
| `192.168.2.222` | tailscale-router | Tailscale exit node / router | Active |
|
||||||
|
| `192.168.2.221` | tor-relay | Tor relay | Active |
|
||||||
|
| `192.168.2.220` | pdm | Proxmox Deploy Manager | Active |
|
||||||
|
| `192.168.2.231` | ha-docker-2 | Docker Swarm node 2 — management NIC | Active |
|
||||||
|
| `192.168.2.230` | ha-docker-1 | Docker Swarm node 1 — management NIC | Active |
|
||||||
|
|
||||||
|
### Client DHCP pool (.10–.59)
|
||||||
|
|
||||||
|
Assigned by the router. DNS option points to `192.168.2.253` (domain-controller).
|
||||||
|
|
||||||
|
Devices in this range: phones, laptops, IoT, Canon printer, any non-infrastructure host.
|
||||||
|
No static reservations for infrastructure hosts — all infra uses static IP configuration
|
||||||
|
on the guest itself (not DHCP reservations), so IPs survive VM recreation regardless of
|
||||||
|
MAC address churn.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Cluster network — VLAN 10 — 192.168.10.224/29
|
||||||
|
|
||||||
|
Internal to pve1 only. Proxmox bridge `vmbr1`, no physical NIC attached.
|
||||||
|
|
||||||
|
| IP | Hostname | Interface role |
|
||||||
|
|---|---|---|
|
||||||
|
| `192.168.10.228` | ha-node1 | DRBD replication + Corosync ring0 (primary heartbeat) |
|
||||||
|
| `192.168.10.227` | ha-node2 | DRBD replication + Corosync ring0 (primary heartbeat) |
|
||||||
|
| — | no gateway | Isolated — not routed to LAN or internet |
|
||||||
|
|
||||||
|
Corosync ring1 (backup heartbeat only) uses the LAN IPs (`192.168.2.228` / `192.168.2.227`)
|
||||||
|
over `vmbr0` — no additional bridge needed, and DRBD traffic never crosses ring1.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Storage-client network — VLAN 20 — 192.168.20.0/24
|
||||||
|
|
||||||
|
Internal to pve1 only. Proxmox bridge `vmbr2`, no physical NIC attached.
|
||||||
|
|
||||||
|
| IP | Hostname | Interface / role |
|
||||||
|
|---|---|---|
|
||||||
|
| `192.168.20.229` | ha-vip-storage | Pacemaker floating VIP — NFS + iSCSI endpoint |
|
||||||
|
| `192.168.20.228` | ha-node1 | Storage-client NIC (ens20 / vmbr2) |
|
||||||
|
| `192.168.20.227` | ha-node2 | Storage-client NIC (ens20 / vmbr2) |
|
||||||
|
| `192.168.20.226` | server | Storage-client NIC (ens19 / vmbr2) — decommissioned |
|
||||||
|
| `192.168.20.225` | docker | Storage-client NIC (eth1 / vmbr2) — NFS client (CT 105) |
|
||||||
|
| `192.168.20.231` | ha-docker-2 | Storage-client NIC (ens19 / vmbr2) — NFS client |
|
||||||
|
| `192.168.20.230` | ha-docker-1 | Storage-client NIC (ens19 / vmbr2) — NFS client |
|
||||||
|
| — | no gateway | Isolated — not routed to LAN or internet |
|
||||||
|
|
||||||
|
NFS clients mount from `192.168.20.229` (surviving failover transparently via the VIP).
|
||||||
|
Firewall on each HA node restricts NFS and iSCSI ports to `192.168.20.0/24` — LAN hosts
|
||||||
|
cannot reach either service on this VIP. The `vip-storage` endpoint is not reachable
|
||||||
|
from the workstation directly (internal bridge only); health checks proxy through the
|
||||||
|
active HA node.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Swarm cluster network — VLAN 30 — 192.168.30.0/24
|
||||||
|
|
||||||
|
Internal to pve1 only. Proxmox bridge `vmbr3`, no physical NIC attached.
|
||||||
|
Carries Docker Swarm inter-node traffic only: Raft consensus (TCP 2377),
|
||||||
|
Serf gossip (TCP/UDP 7946), and VXLAN overlay data path (UDP 4789).
|
||||||
|
Docker Swarm is initialised with `--advertise-addr` and `--data-path-addr`
|
||||||
|
both pointing to this subnet so all cluster traffic stays on `vmbr3` and
|
||||||
|
never crosses the LAN.
|
||||||
|
|
||||||
|
| IP | Hostname | Interface / role |
|
||||||
|
|---|---|---|
|
||||||
|
| `192.168.30.231` | ha-docker-2 | Swarm cluster NIC (ens20 / vmbr3) |
|
||||||
|
| `192.168.30.230` | ha-docker-1 | Swarm cluster NIC (ens20 / vmbr3) |
|
||||||
|
| — | no gateway | Isolated — not routed to LAN or internet |
|
||||||
|
|
||||||
|
### DNS zone: `swarm.home` — VLAN 30 (192.168.30.x)
|
||||||
|
|
||||||
|
| Hostname | A record | Notes |
|
||||||
|
|---|---|---|
|
||||||
|
| `ha-docker-1.swarm.home` | `192.168.30.230` | Swarm NIC — debugging only |
|
||||||
|
| `ha-docker-2.swarm.home` | `192.168.30.231` | Swarm NIC — debugging only |
|
||||||
|
|
||||||
|
Operators reach the Docker API on the LAN IPs (`192.168.2.230`/`.231`), not these addresses.
|
||||||
|
The `swarm.home` records exist for diagnostic convenience (e.g. confirming `vmbr3` routing).
|
||||||
|
|
||||||
|
### Multi-node Proxmox expansion
|
||||||
|
|
||||||
|
When a second Proxmox node (pve2) is added, VLAN 10 (cluster), VLAN 20 (storage-client),
|
||||||
|
and VLAN 30 (swarm) all extend to pve2 via 802.1q VLAN tagging on the inter-node trunk
|
||||||
|
link. All three internal networks share the same physical NIC between hypervisors —
|
||||||
|
VLAN tags provide the logical separation.
|
||||||
|
|
||||||
+56
-20
@@ -8,37 +8,73 @@ This repository configures `nix-cache` as a **binary cache server** and a **remo
|
|||||||
- Every machine still keeps and uses its own local `/nix/store`.
|
- Every machine still keeps and uses its own local `/nix/store`.
|
||||||
- Clients prefer `http://nix-cache` for substitutes and keep `https://cache.nixos.org/` as fallback.
|
- Clients prefer `http://nix-cache` for substitutes and keep `https://cache.nixos.org/` as fallback.
|
||||||
- Clients can offload builds to `nix-cache` through SSH (`nix.distributedBuilds`).
|
- Clients can offload builds to `nix-cache` through SSH (`nix.distributedBuilds`).
|
||||||
- Client hosts import `modules/nix/cache-client.nix` and, when remote building is enabled, `modules/nix/remote-builder-client.nix`.
|
- Client hosts import `modules/nix-cache/client.nix` and, when remote building is enabled, `modules/nix-cache/remote-builder-client.nix`.
|
||||||
- The `nix-cache` host imports `modules/nix/cache-server.nix`.
|
- The `nix-cache` host imports `modules/nix-cache/server.nix`.
|
||||||
|
|
||||||
## Binary cache signing keys (on nix-cache)
|
## Binary cache signing key
|
||||||
|
|
||||||
|
`modules/nix-cache/client.nix` hardcodes every client's trust in one
|
||||||
|
specific public key (`cache.local-1:usoWYanY3Kpq2+kDIS2nhWoLZiRxanmdysdzqCFBHW4=`).
|
||||||
|
That means whichever host is currently playing the `nix-cache` role has to
|
||||||
|
use that *exact* keypair — not a freshly generated one — or no client will
|
||||||
|
accept substitutes from it (they'd just silently fall back to building
|
||||||
|
from source). So unlike most per-host secrets, this one can't be
|
||||||
|
self-generated on first boot; it's managed via sops-nix like every other
|
||||||
|
secret in this repo, sourced from `secrets/nix-cache.yaml`'s
|
||||||
|
`cache-priv-key` entry (`modules/nix-cache/server.nix`).
|
||||||
|
|
||||||
|
**Adding or rotating the value:**
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
sudo install -d -m 0700 /etc/nix
|
nix-shell -p sops --run 'sops secrets/nix-cache.yaml'
|
||||||
sudo nix-store --generate-binary-cache-key nix-cache-1 /etc/nix/cache-priv.pem /etc/nix/cache-pub.pem
|
|
||||||
sudo chmod 0600 /etc/nix/cache-priv.pem
|
|
||||||
sudo chmod 0644 /etc/nix/cache-pub.pem
|
|
||||||
cat /etc/nix/cache-pub.pem
|
|
||||||
```
|
```
|
||||||
|
|
||||||
Do not commit private keys.
|
Add (or replace) a `cache-priv-key` entry with the private key file's exact
|
||||||
Do not commit new password hashes or live credentials. Existing committed hashes
|
contents. If you don't have it yet, generate a keypair once:
|
||||||
should be rotated and moved to host-local secret management.
|
|
||||||
|
```bash
|
||||||
|
nix-store --generate-binary-cache-key nix-cache-1 cache-priv.pem cache-pub.pem
|
||||||
|
```
|
||||||
|
|
||||||
|
— paste `cache-priv.pem`'s contents into the `cache-priv-key` entry above,
|
||||||
|
delete both local files afterward, and update
|
||||||
|
`trusted-public-keys` in `modules/nix-cache/client.nix` (and every already-built
|
||||||
|
client) to match `cache-pub.pem` if this is a genuine rotation rather than
|
||||||
|
a first-time bootstrap. Any `nixos-configurations.*-nix-cache` host picks
|
||||||
|
the new key up automatically on next activation — no more manual
|
||||||
|
`/etc/nix/cache-priv.pem` install step.
|
||||||
|
|
||||||
## Remote builder SSH keys
|
## Remote builder SSH keys
|
||||||
|
|
||||||
On each client, install the private key used to authenticate as `nixremote`:
|
Each client authenticates as `nixremote` using its **own default root SSH
|
||||||
|
identity** (`/root/.ssh/id_ed25519`) — not a separately-named or shared
|
||||||
|
keypair. If a client doesn't have one yet:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
sudo install -d -m 0700 /root/.ssh
|
sudo ssh-keygen -t ed25519 -N '' -f /root/.ssh/id_ed25519
|
||||||
sudo install -m 0600 ./nixremote /root/.ssh/nixremote
|
|
||||||
sudo ssh -i /root/.ssh/nixremote nixremote@nix-cache nix-store --version
|
|
||||||
```
|
```
|
||||||
|
|
||||||
On `nix-cache`, install the matching public key used by `nixremote` authorized keys.
|
Then add its `.pub` contents as a new entry in `vars.remoteBuilderAuthorizedKeys`
|
||||||
|
(`variables.nix`) and rebuild `nix-cache` to pick it up (that list is
|
||||||
|
declarative — an imperative `ssh-copy-id nixremote@nix-cache` won't stick;
|
||||||
|
it gets overwritten on every rebuild). Verify with:
|
||||||
|
|
||||||
The committed `nixremote` authorized keys are public SSH keys only. Keep the
|
```bash
|
||||||
matching private keys on client hosts and out of the repository.
|
sudo ssh -i /root/.ssh/id_ed25519 nixremote@nix-cache nix-store --version
|
||||||
|
```
|
||||||
|
|
||||||
|
The committed `remoteBuilderAuthorizedKeys` entries are public SSH keys
|
||||||
|
only. Keep the matching private keys on client hosts and out of the
|
||||||
|
repository.
|
||||||
|
|
||||||
|
nix-cache's own SSH *host* key is trusted declaratively via
|
||||||
|
`programs.ssh.knownHosts` in `modules/nix-cache/remote-builder-client.nix`,
|
||||||
|
sourced from `vars.nixCacheHostKey` (`variables.nix`) — every client rebuild
|
||||||
|
picks it up automatically, so distributed builds don't fail with "Host key
|
||||||
|
verification failed" on a client that has never manually SSH'd to nix-cache
|
||||||
|
before. If nix-cache's host key is ever rotated or the host rebuilt from
|
||||||
|
scratch, update `vars.nixCacheHostKey` to match its new
|
||||||
|
`/etc/ssh/ssh_host_ed25519_key.pub`.
|
||||||
|
|
||||||
## Manual verification
|
## Manual verification
|
||||||
|
|
||||||
@@ -48,8 +84,8 @@ After deployment:
|
|||||||
curl http://nix-cache/nix-cache-info
|
curl http://nix-cache/nix-cache-info
|
||||||
nix store ping --store http://nix-cache
|
nix store ping --store http://nix-cache
|
||||||
nix show-config | grep -E 'substituters|trusted-public-keys|builders-use-substitutes'
|
nix show-config | grep -E 'substituters|trusted-public-keys|builders-use-substitutes'
|
||||||
sudo ssh -i /root/.ssh/nixremote nixremote@nix-cache nix-store --version
|
sudo ssh -i /root/.ssh/id_ed25519 nixremote@nix-cache nix-store --version
|
||||||
nix build nixpkgs#hello --builders 'ssh://nixremote@nix-cache x86_64-linux /root/.ssh/nixremote 4 2 big-parallel,kvm,nixos-test,benchmark' -L
|
nix build nixpkgs#hello --builders 'ssh://nixremote@nix-cache x86_64-linux /root/.ssh/id_ed25519 4 2 big-parallel,kvm,nixos-test,benchmark' -L
|
||||||
nix path-info -r nixpkgs#hello
|
nix path-info -r nixpkgs#hello
|
||||||
curl -I "http://nix-cache/$(basename "$(nix path-info nixpkgs#hello)").narinfo"
|
curl -I "http://nix-cache/$(basename "$(nix path-info nixpkgs#hello)").narinfo"
|
||||||
```
|
```
|
||||||
|
|||||||
@@ -0,0 +1,116 @@
|
|||||||
|
# Proxmox VM disk images
|
||||||
|
|
||||||
|
`proxmox-*` hosts (VM platform, not `lxc-*`) can be built as standalone,
|
||||||
|
ready-to-attach `.raw` disk images via disko's own image-builder — no
|
||||||
|
`nixos-install`, no live installer boot. This uses the same `disko.devices`
|
||||||
|
config (`modules/disko/proxmox.nix`) already used to format a real disk on
|
||||||
|
install, so there's nothing host-specific to write; it's available for every
|
||||||
|
`proxmox-*` target automatically.
|
||||||
|
|
||||||
|
`scripts/proxmox/create-proxmox-resource.sh --type vm --host <name>` automates the
|
||||||
|
whole walkthrough below (and the equivalent LXC one) end to end, including
|
||||||
|
host-key handling and building the image directly on the Proxmox node
|
||||||
|
itself (no local build, no image transfer) — see its `--help`. The steps
|
||||||
|
here are what it runs under the hood, useful for doing any of it by hand
|
||||||
|
or understanding what it does before you trust it against real
|
||||||
|
infrastructure.
|
||||||
|
|
||||||
|
## Building
|
||||||
|
|
||||||
|
```sh
|
||||||
|
nix build .#nixosConfigurations.proxmox-server.config.system.build.diskoImagesScript
|
||||||
|
sudo ./result --build-memory 2048
|
||||||
|
```
|
||||||
|
|
||||||
|
This produces `<hostname>.raw` in the current directory (e.g. `server.raw`
|
||||||
|
for `proxmox-server`, matching `networking.hostName`, not the flake attribute
|
||||||
|
name — every `proxmox-*` host gets a distinctly named image instead of all
|
||||||
|
of them producing an identical `main.raw`). The script builds inside a
|
||||||
|
temporary QEMU VM and moves the finished image out to the working directory
|
||||||
|
when done; `--build-memory` controls how much RAM that build VM gets.
|
||||||
|
|
||||||
|
`disko.devices.disk.main.imageSize` (currently `20G`, in
|
||||||
|
`modules/disko/proxmox.nix`) sets the image's total size — disko doesn't
|
||||||
|
support auto-resizing, so this needs to comfortably fit ESP + swap + root at
|
||||||
|
build time. Grow the virtual disk (and resize the filesystem) in Proxmox
|
||||||
|
after attaching if a host needs more than that; this is the normal way to
|
||||||
|
size these images, not a one-time decision to get exactly right up front.
|
||||||
|
|
||||||
|
## Host keys
|
||||||
|
|
||||||
|
The disko image script runs a real activation pass inside its temporary
|
||||||
|
build VM while constructing the image — the same sops-nix
|
||||||
|
activation-before-first-boot problem the installer and LXC tarball workflows
|
||||||
|
have (see `docs/auto-installer.md`) applies here too, unmodified. Disko has
|
||||||
|
a native mechanism for it:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
sudo ./result \
|
||||||
|
--pre-format-files host-keys/server_ssh_host_ed25519_key /etc/ssh/ssh_host_ed25519_key \
|
||||||
|
--pre-format-files host-keys/server_ssh_host_ed25519_key.pub /etc/ssh/ssh_host_ed25519_key.pub \
|
||||||
|
--build-memory 2048
|
||||||
|
```
|
||||||
|
|
||||||
|
Generate the key first with `scripts/secrets/sync-host-keys.sh <hostname>`, same
|
||||||
|
as any other host — see `docs/auto-installer.md` for the full walkthrough
|
||||||
|
(it registers the new key in `.sops.yaml` and re-encrypts the affected
|
||||||
|
`secrets/*.yaml` files too, no manual editing needed).
|
||||||
|
|
||||||
|
## Deploying to Proxmox
|
||||||
|
|
||||||
|
The image needs **UEFI (OVMF)**, not Proxmox's default SeaBIOS —
|
||||||
|
`modules/boot/efi.nix` uses `systemd-boot`, which only works with UEFI
|
||||||
|
firmware. `virtio-scsi` is safe to use as the disk bus:
|
||||||
|
`hardware-configuration/vm/proxmox.nix` already includes `virtio_scsi` in
|
||||||
|
its initrd kernel modules.
|
||||||
|
|
||||||
|
1. Copy the image to the Proxmox host:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
scp server.raw root@<proxmox-host>:/var/lib/vz/import/
|
||||||
|
```
|
||||||
|
|
||||||
|
2. Create an empty VM shell (no disk yet) — replace `<vmid>` with a free ID
|
||||||
|
and `<storage>` with your storage pool's name (`pvesm status` or
|
||||||
|
Datacenter → Storage in the web UI):
|
||||||
|
|
||||||
|
```sh
|
||||||
|
qm create <vmid> --name proxmox-server --memory 2048 --cores 2 \
|
||||||
|
--net0 virtio,bridge=vmbr0 \
|
||||||
|
--bios ovmf --machine q35 \
|
||||||
|
--scsihw virtio-scsi-pci \
|
||||||
|
--efidisk0 <storage>:1,efitype=4m,pre-enrolled-keys=0
|
||||||
|
```
|
||||||
|
|
||||||
|
(`--efidisk0` is required for UEFI — it's where OVMF persists boot-entry
|
||||||
|
NVRAM; without it, systemd-boot's boot entry may not survive a reboot.)
|
||||||
|
|
||||||
|
3. Import the raw disk into storage:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
qm importdisk <vmid> /var/lib/vz/import/server.raw <storage>
|
||||||
|
```
|
||||||
|
|
||||||
|
This prints the resulting disk identifier (e.g. `vm-<vmid>-disk-1`).
|
||||||
|
|
||||||
|
4. Attach it and set it as the boot disk:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
qm set <vmid> --scsi0 <storage>:vm-<vmid>-disk-1
|
||||||
|
qm set <vmid> --boot order=scsi0
|
||||||
|
```
|
||||||
|
|
||||||
|
5. Boot it:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
qm start <vmid>
|
||||||
|
```
|
||||||
|
|
||||||
|
No install step — it boots straight into the already-activated system.
|
||||||
|
|
||||||
|
## Why not `nix build .#nixosConfigurations.<host>.config.system.build.vm`?
|
||||||
|
|
||||||
|
That's a different, unrelated feature — `system.build.vm` (`nixos-rebuild
|
||||||
|
build-vm`) produces an ephemeral QEMU script for locally testing a
|
||||||
|
configuration, not a distributable disk image. It's not part of this
|
||||||
|
workflow.
|
||||||
+132
-13
@@ -1,6 +1,10 @@
|
|||||||
# pxe-boot
|
# pxe-boot
|
||||||
|
|
||||||
The `pxe-boot` host serves HTTP boot assets for iPXE clients.
|
The `pxe-boot` host serves HTTP boot assets for iPXE clients — including
|
||||||
|
self-staged copies of both this flake's own auto-installer netboot image
|
||||||
|
(see `docs/auto-installer.md` for what that image actually is and does once
|
||||||
|
booted) and a vanilla, unmodified NixOS minimal netboot image for plain
|
||||||
|
rescue/inspection use.
|
||||||
|
|
||||||
## Host Role
|
## Host Role
|
||||||
|
|
||||||
@@ -11,6 +15,7 @@ The `pxe-boot` host serves HTTP boot assets for iPXE clients.
|
|||||||
- TFTP root for first-stage bootloaders: `/srv/pxe/tftp`
|
- TFTP root for first-stage bootloaders: `/srv/pxe/tftp`
|
||||||
- iPXE entry script: `/srv/pxe/http/boot.ipxe`
|
- iPXE entry script: `/srv/pxe/http/boot.ipxe`
|
||||||
- Generated iPXE menu: `/srv/pxe/http/menu.ipxe`
|
- Generated iPXE menu: `/srv/pxe/http/menu.ipxe`
|
||||||
|
- Debian Minimal iPXE script: `/srv/pxe/http/debian.ipxe`
|
||||||
- SystemRescue iPXE script: `/srv/pxe/http/systemrescue.ipxe`
|
- SystemRescue iPXE script: `/srv/pxe/http/systemrescue.ipxe`
|
||||||
- TFTP fallback script: `/srv/pxe/tftp/autoexec.ipxe`
|
- TFTP fallback script: `/srv/pxe/tftp/autoexec.ipxe`
|
||||||
- Boot binaries copied from the Nix `ipxe` package:
|
- Boot binaries copied from the Nix `ipxe` package:
|
||||||
@@ -24,36 +29,140 @@ The host creates these directories with systemd tmpfiles:
|
|||||||
```text
|
```text
|
||||||
/srv/pxe
|
/srv/pxe
|
||||||
/srv/pxe/http
|
/srv/pxe/http
|
||||||
/srv/pxe/http/images
|
/srv/pxe/http/images -> /mnt/pxe-images (symlink to NFS share)
|
||||||
/srv/pxe/http/nixos
|
/srv/pxe/http/auto-installer
|
||||||
|
/srv/pxe/http/nixos-minimal
|
||||||
|
/srv/pxe/http/debian
|
||||||
/srv/pxe/http/systemrescue
|
/srv/pxe/http/systemrescue
|
||||||
/srv/pxe/http/ubuntu
|
/srv/pxe/http/ubuntu
|
||||||
/srv/pxe/http/rescue
|
/srv/pxe/http/rescue
|
||||||
/srv/pxe/tftp
|
/srv/pxe/tftp
|
||||||
```
|
```
|
||||||
|
|
||||||
Mount shared image storage under `/srv/pxe/http`, preferably
|
`/srv/pxe/http/images` is a symlink to `/mnt/pxe-images`, which is an NFS
|
||||||
`/srv/pxe/http/images` unless a menu entry expects files in a specific
|
mount of `server.sweet.home:/tank/pxe-boot/images`
|
||||||
directory such as `/srv/pxe/http/nixos`.
|
(`modules/pxe-boot/mount-pxe-images.nix`). Place large images there (ISOs,
|
||||||
|
disk images) rather than on the pxe-boot host's own root disk. For an LXC
|
||||||
|
pxe-boot container the mount uses NFSv3+nolock with `nofail` (eager,
|
||||||
|
non-blocking on server unavailability); for a Proxmox VM it uses NFSv4.2
|
||||||
|
with `x-systemd.automount` (lazy, triggered on first access).
|
||||||
|
|
||||||
|
When running as `lxc-pxe-boot`, the Proxmox container must have
|
||||||
|
`features: nesting=1,mount=nfs` (at minimum) in its Proxmox config. `nesting=1`
|
||||||
|
is required by systemd 260+ for credential isolation (user namespace creation
|
||||||
|
and internal move-mounts); without it, AppArmor denies both, and every
|
||||||
|
systemd service that uses `PrivateUsers`, `PrivateDevices`, or credential
|
||||||
|
passing fails on boot. `mount=nfs` allows the NFSv3 mount. Both are set
|
||||||
|
automatically by `scripts/proxmox/create-proxmox-resource.sh` (via
|
||||||
|
`PROXMOX_DEFAULT_LXC_FEATURES` in `scripts/env.sh` which defaults to
|
||||||
|
`nesting=1,keyctl=1,mount=nfs;nfs4`). If you ever change these features
|
||||||
|
manually via `pct set`, be sure to include both — `pct set` replaces the
|
||||||
|
entire features string, it does not append to it.
|
||||||
|
|
||||||
The HTTP iPXE chain is:
|
The HTTP iPXE chain is:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
undionly.kpxe or ipxe.efi
|
undionly.kpxe or ipxe.efi
|
||||||
-> autoexec.ipxe from the TFTP root, when iPXE requests it
|
-> autoexec.ipxe from the TFTP root, when iPXE requests it
|
||||||
-> http://192.168.2.247/boot.ipxe
|
-> http://192.168.2.223/boot.ipxe
|
||||||
-> http://192.168.2.247/menu.ipxe
|
-> http://192.168.2.223/menu.ipxe
|
||||||
```
|
```
|
||||||
|
|
||||||
The generated menu currently exposes entries for:
|
The generated menu currently exposes entries for:
|
||||||
|
|
||||||
- NixOS installer
|
- NixOS Auto-Installer
|
||||||
|
- NixOS Minimal
|
||||||
|
- Debian Minimal
|
||||||
|
- FreeIPA Server (Rocky Linux 9)
|
||||||
- SystemRescue environment
|
- SystemRescue environment
|
||||||
- iPXE shell
|
- iPXE shell
|
||||||
- Reboot
|
- Reboot
|
||||||
|
|
||||||
Kernel and initrd artifacts for the NixOS installer entry must be placed under
|
Both NixOS entries chain-load a `netboot.ipxe` staged into their own
|
||||||
`/srv/pxe/http/nixos` by an operator or a separate build process.
|
directory (`/srv/pxe/http/auto-installer/netboot.ipxe` and
|
||||||
|
`/srv/pxe/http/nixos-minimal/netboot.ipxe`), each nixpkgs' own generated
|
||||||
|
netboot iPXE script (correct `init=`/`initrd=` kernel parameters included)
|
||||||
|
rather than a hand-rolled boot line — that script in turn expects its
|
||||||
|
kernel/initrd siblings in the same directory. Each directory's three files
|
||||||
|
(`bzImage`, `initrd`, `netboot.ipxe`) are built from source and staged
|
||||||
|
automatically by `modules/pxe-boot/stage-installer-artifacts.nix` via
|
||||||
|
`systemd.tmpfiles.rules` — no manual operator step required:
|
||||||
|
|
||||||
|
- `auto-installer` is this flake's own `netbootSystem` (`flake.nix`) — the
|
||||||
|
same auto-installer image `nix build .#pxe` produces. See
|
||||||
|
`docs/auto-installer.md`.
|
||||||
|
- `nixos-minimal` is `netbootMinimalSystem` (`flake.nix`) — nixpkgs'
|
||||||
|
`netboot-minimal.nix` composed on its own, with none of this flake's
|
||||||
|
auto-installer wiring (no `common.nix`, no `auto-install.sh`, no baked
|
||||||
|
host keys or custom users). Same `nix build .#pxe-minimal` mechanism as
|
||||||
|
the auto-installer image, just a different module composition. Useful
|
||||||
|
as a plain rescue/inspection shell that doesn't assume anything about
|
||||||
|
this flake.
|
||||||
|
|
||||||
|
Both images set `networking.hostName` to match their menu entry/staged
|
||||||
|
directory name (`auto-installer` / `nixos-minimal`), so each one's
|
||||||
|
generated system name (`nixos-system-<name>-*`) is self-describing rather
|
||||||
|
than the nixpkgs default of `nixos-system-nixos-*` for both.
|
||||||
|
|
||||||
|
The Debian Minimal entry chains `http://<pxeServerIp>/debian.ipxe`, which loads
|
||||||
|
the Debian bookworm netboot kernel and initrd from `/srv/pxe/http/debian/`. The
|
||||||
|
`fetch-debian-netboot.service` oneshot downloads these files from
|
||||||
|
`deb.debian.org` on first boot (idempotent — skips if files are already
|
||||||
|
present):
|
||||||
|
|
||||||
|
```text
|
||||||
|
/srv/pxe/http/debian/linux (Debian bookworm netboot kernel)
|
||||||
|
/srv/pxe/http/debian/initrd.gz (Debian bookworm netboot initrd)
|
||||||
|
```
|
||||||
|
|
||||||
|
The service requires outbound internet access on the pxe-boot host. To
|
||||||
|
re-download (e.g. after a Debian point release), delete the files and restart
|
||||||
|
the service:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
rm /srv/pxe/http/debian/linux /srv/pxe/http/debian/initrd.gz
|
||||||
|
systemctl restart fetch-debian-netboot.service
|
||||||
|
```
|
||||||
|
|
||||||
|
To update to a different Debian release, change `debianRelease` in
|
||||||
|
`modules/build-types/pxe-boot.nix` and redeploy.
|
||||||
|
|
||||||
|
The **FreeIPA Server (Rocky Linux 9)** entry chains
|
||||||
|
`http://<pxeServerIp>/rocky-freeipa.ipxe`, which boots the Rocky Linux 9
|
||||||
|
Anaconda installer with a Kickstart file (`rocky-freeipa.ks`) hosted on the
|
||||||
|
same server. The `fetch-rocky-pxeboot.service` oneshot downloads the pxeboot
|
||||||
|
kernel and initrd from the Rocky Linux mirror on first boot (idempotent):
|
||||||
|
|
||||||
|
```text
|
||||||
|
/srv/pxe/http/rocky/vmlinuz (Rocky Linux 9 Anaconda pxeboot kernel)
|
||||||
|
/srv/pxe/http/rocky/initrd.img (Rocky Linux 9 Anaconda pxeboot initrd)
|
||||||
|
```
|
||||||
|
|
||||||
|
The Kickstart file is generated from the NixOS module and staged at
|
||||||
|
`/srv/pxe/http/rocky-freeipa.ks`. It performs a fully unattended install:
|
||||||
|
|
||||||
|
1. Installs Rocky Linux 9 with `ipa-server` + `ipa-server-dns` packages
|
||||||
|
2. Configures static IP `192.168.2.138`, hostname `domain-controller.sweet.home`
|
||||||
|
3. Creates user `wayne` with the `adminSshKey` from `variables.nix`
|
||||||
|
4. Generates random IPA passwords and writes them to `/root/ipa-credentials.txt`
|
||||||
|
5. Creates a `freeipa-first-boot.service` oneshot that runs `ipa-server-install`
|
||||||
|
on first reboot (~20 minutes)
|
||||||
|
|
||||||
|
After the install completes:
|
||||||
|
- SSH in as `wayne@domain-controller` using the admin key
|
||||||
|
- Monitor FreeIPA install progress: `sudo tail -f /root/freeipa-install.log`
|
||||||
|
- Retrieve credentials: `sudo cat /root/ipa-credentials.txt` (save to password manager)
|
||||||
|
- Configure Pi-hole: `server=/sweet.home/192.168.2.138` in dnsmasq
|
||||||
|
|
||||||
|
To refresh the pxeboot files (e.g. after a Rocky point release):
|
||||||
|
|
||||||
|
```bash
|
||||||
|
rm /srv/pxe/http/rocky/vmlinuz /srv/pxe/http/rocky/initrd.img
|
||||||
|
systemctl restart fetch-rocky-pxeboot.service
|
||||||
|
```
|
||||||
|
|
||||||
|
To update to a different Rocky release, change `rockyRelease` in
|
||||||
|
`modules/build-types/pxe-boot.nix` and redeploy.
|
||||||
|
|
||||||
The SystemRescue entry expects the source ISO at:
|
The SystemRescue entry expects the source ISO at:
|
||||||
|
|
||||||
@@ -61,13 +170,16 @@ The SystemRescue entry expects the source ISO at:
|
|||||||
/srv/pxe/http/images/systemrescue.iso
|
/srv/pxe/http/images/systemrescue.iso
|
||||||
```
|
```
|
||||||
|
|
||||||
|
Since `/srv/pxe/http/images` is the NFS-backed symlink, place the ISO on the
|
||||||
|
NFS share at `server.sweet.home:/tank/pxe-boot/images/systemrescue.iso`.
|
||||||
|
|
||||||
The `stage-systemrescue.service` oneshot extracts that ISO into:
|
The `stage-systemrescue.service` oneshot extracts that ISO into:
|
||||||
|
|
||||||
```text
|
```text
|
||||||
/srv/pxe/http/systemrescue
|
/srv/pxe/http/systemrescue
|
||||||
```
|
```
|
||||||
|
|
||||||
The rescue menu entry then chains `http://192.168.2.247/systemrescue.ipxe`,
|
The rescue menu entry then chains `http://192.168.2.223/systemrescue.ipxe`,
|
||||||
which loads the SystemRescue kernel and initramfs from the extracted tree and
|
which loads the SystemRescue kernel and initramfs from the extracted tree and
|
||||||
uses `archiso_http_srv` to fetch the squashfs payload over HTTP.
|
uses `archiso_http_srv` to fetch the squashfs payload over HTTP.
|
||||||
|
|
||||||
@@ -76,7 +188,7 @@ uses `archiso_http_srv` to fetch the squashfs payload over HTTP.
|
|||||||
Safe evaluation check:
|
Safe evaluation check:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
nix eval .#nixosConfigurations.pxe-boot.config.system.build.toplevel.drvPath --raw
|
nix eval .#nixosConfigurations.proxmox-pxe-boot.config.system.build.toplevel.drvPath --raw
|
||||||
```
|
```
|
||||||
|
|
||||||
After deployment by an operator, basic service checks are:
|
After deployment by an operator, basic service checks are:
|
||||||
@@ -84,6 +196,13 @@ After deployment by an operator, basic service checks are:
|
|||||||
```bash
|
```bash
|
||||||
curl http://pxe-boot/boot.ipxe
|
curl http://pxe-boot/boot.ipxe
|
||||||
curl http://pxe-boot/menu.ipxe
|
curl http://pxe-boot/menu.ipxe
|
||||||
|
curl http://pxe-boot/debian.ipxe
|
||||||
|
curl -I http://pxe-boot/debian/linux
|
||||||
|
curl -I http://pxe-boot/debian/initrd.gz
|
||||||
|
curl http://pxe-boot/rocky-freeipa.ipxe
|
||||||
|
curl http://pxe-boot/rocky-freeipa.ks
|
||||||
|
curl -I http://pxe-boot/rocky/vmlinuz
|
||||||
|
curl -I http://pxe-boot/rocky/initrd.img
|
||||||
curl http://pxe-boot/systemrescue.ipxe
|
curl http://pxe-boot/systemrescue.ipxe
|
||||||
curl -I http://pxe-boot/systemrescue/sysresccd/boot/x86_64/vmlinuz
|
curl -I http://pxe-boot/systemrescue/sysresccd/boot/x86_64/vmlinuz
|
||||||
curl -I http://pxe-boot/systemrescue/sysresccd/boot/x86_64/sysresccd.img
|
curl -I http://pxe-boot/systemrescue/sysresccd/boot/x86_64/sysresccd.img
|
||||||
|
|||||||
@@ -1,143 +0,0 @@
|
|||||||
# Spec: Refactor Flake Targets into Platform × Build-Type Matrix
|
|
||||||
|
|
||||||
## Context
|
|
||||||
|
|
||||||
The flake at `~/nixos` currently defines these output targets (flat, ad-hoc naming):
|
|
||||||
|
|
||||||
- `docker`
|
|
||||||
- `linode-minimal`
|
|
||||||
- `nix-cache`
|
|
||||||
- `nix-minimal`
|
|
||||||
- `nixos`
|
|
||||||
- `server`
|
|
||||||
- `pxe-boot`
|
|
||||||
|
|
||||||
Some already follow a `platform-buildtype` convention (`linode-minimal`), most don't.
|
|
||||||
`~/nix-auto-installer` is a related repo and should be checked for any coupling to
|
|
||||||
these target names (scripts, docs, CI, or install automation that reference them by
|
|
||||||
name) before renaming anything.
|
|
||||||
|
|
||||||
## Goal
|
|
||||||
|
|
||||||
Restructure the flake so targets are generated from two orthogonal concepts:
|
|
||||||
|
|
||||||
**Build types** (what the system is for):
|
|
||||||
- `minimal`
|
|
||||||
- `nix-cache`
|
|
||||||
- `server`
|
|
||||||
- `docker`
|
|
||||||
- `pxe-boot`
|
|
||||||
- `gui`
|
|
||||||
|
|
||||||
**Platforms** (what it's deployed on):
|
|
||||||
- `linode` (Linode VM)
|
|
||||||
- `proxmox` (Proxmox VM)
|
|
||||||
- `lxc` (Proxmox LXC container)
|
|
||||||
|
|
||||||
Final targets should be named consistently as `<platform>-<buildtype>`, e.g.:
|
|
||||||
|
|
||||||
```
|
|
||||||
linode-minimal proxmox-minimal lxc-minimal
|
|
||||||
linode-nix-cache proxmox-nix-cache lxc-nix-cache
|
|
||||||
linode-server proxmox-server lxc-server
|
|
||||||
linode-docker proxmox-docker lxc-docker
|
|
||||||
linode-pxe-boot proxmox-pxe-boot lxc-pxe-boot
|
|
||||||
linode-gui proxmox-gui lxc-gui
|
|
||||||
```
|
|
||||||
|
|
||||||
That's the full matrix (18 targets) if every build type applies to every platform.
|
|
||||||
See **Open Questions** below — some combinations may not make sense and should be
|
|
||||||
confirmed with me before being built out, not silently included or dropped.
|
|
||||||
|
|
||||||
## Migration mapping (old → new)
|
|
||||||
|
|
||||||
| Old target | New target | Notes |
|
|
||||||
|--------------------|------------------------------------------------------|-------|
|
|
||||||
| `linode-minimal` | `linode-minimal` | Already correct, keep as-is |
|
|
||||||
| `nix-minimal` | likely `proxmox-minimal` or a platform-less base module | Ambiguous — see Open Questions |
|
|
||||||
| `nix-cache` | base module consumed by `linode-nix-cache`, `proxmox-nix-cache`, `lxc-nix-cache` | Currently platform-less; needs to become a build-type module, not a standalone target |
|
|
||||||
| `server` | base module consumed by `linode-server`, `proxmox-server`, `lxc-server` | Same as above |
|
|
||||||
| `docker` | base module consumed by `linode-docker`, `proxmox-docker`, `lxc-docker` | Confirm docker actually makes sense as an LXC/VM guest build vs. a standalone container image — see Open Questions |
|
|
||||||
| `pxe-boot` | TBD — may stay a single target rather than a per-platform one | See Open Questions |
|
|
||||||
| `nixos` | TBD — unclear what this maps to in the new scheme | See Open Questions |
|
|
||||||
|
|
||||||
## Open Questions (Claude Code: raise these with me before implementing, don't guess)
|
|
||||||
|
|
||||||
1. **`nixos` target** — what is this currently used for (bare metal install, dev
|
|
||||||
shell, template)? It doesn't obviously map to any of the six build types.
|
|
||||||
2. **`nix-minimal` vs `linode-minimal`** — are these two different things, or is
|
|
||||||
`nix-minimal` a leftover/duplicate?
|
|
||||||
3. **`pxe-boot` and `gui` across all three platforms** — does PXE boot make sense
|
|
||||||
for an LXC container or a cloud VM (Linode), or is it inherently bare-metal/
|
|
||||||
network-boot only and should remain a single non-platform target? Does `gui`
|
|
||||||
make sense inside an LXC container?
|
|
||||||
4. **`docker` as a build type** — is this "a NixOS host configured to run Docker"
|
|
||||||
(which would sensibly have linode/proxmox/lxc variants), or "a Docker container
|
|
||||||
image built by the flake" (which wouldn't take a platform prefix at all, since
|
|
||||||
it doesn't run on Linode/Proxmox/LXC as a guest OS)? These are structurally
|
|
||||||
different and change how it should be wired in.
|
|
||||||
5. Confirm whether all 18 combinations should actually exist, or whether this is
|
|
||||||
meant to produce only the combinations that are genuinely useful (e.g. maybe no
|
|
||||||
one needs `lxc-pxe-boot`).
|
|
||||||
|
|
||||||
## Implementation approach
|
|
||||||
|
|
||||||
1. **Inventory first.** Read the current `flake.nix` and any `nixosConfigurations`/
|
|
||||||
`modules` structure. Map every existing target to what module(s) it actually
|
|
||||||
pulls in. Don't assume — confirm against the real file contents.
|
|
||||||
2. **Separate build-type and platform into their own module directories**, e.g.:
|
|
||||||
```
|
|
||||||
modules/build-types/minimal.nix
|
|
||||||
modules/build-types/nix-cache.nix
|
|
||||||
modules/build-types/server.nix
|
|
||||||
modules/build-types/docker.nix
|
|
||||||
modules/build-types/pxe-boot.nix
|
|
||||||
modules/build-types/gui.nix
|
|
||||||
|
|
||||||
modules/platforms/linode.nix
|
|
||||||
modules/platforms/proxmox.nix
|
|
||||||
modules/platforms/lxc.nix
|
|
||||||
```
|
|
||||||
Build-type modules should contain only what makes a system "minimal" vs
|
|
||||||
"server" vs "gui", etc. Platform modules should contain only what's specific
|
|
||||||
to running as a Linode VM vs Proxmox VM vs LXC container (virtualisation
|
|
||||||
guest tools, boot method, filesystem/image format, LXC-specific constraints
|
|
||||||
like no kernel modules, etc).
|
|
||||||
3. **Generate the target matrix programmatically** in `flake.nix` rather than
|
|
||||||
hand-writing 18 near-identical `nixosConfigurations` entries — e.g. a small
|
|
||||||
function that takes a platform name and build-type name, composes the two
|
|
||||||
modules plus any shared base module, and produces the named output. This
|
|
||||||
keeps future build types/platforms a one-line addition rather than a copy-paste
|
|
||||||
job.
|
|
||||||
4. **Only build combinations we've confirmed make sense** (see Open Questions) —
|
|
||||||
don't emit all 18 by default if some are structurally invalid.
|
|
||||||
5. **Preserve existing working configs during the transition.** Don't delete the
|
|
||||||
old target names until their replacements build successfully — rename/alias
|
|
||||||
at the end, not the start, so there's no window where the flake is broken.
|
|
||||||
|
|
||||||
## Verification
|
|
||||||
|
|
||||||
For every new target produced:
|
|
||||||
```bash
|
|
||||||
nix flake check
|
|
||||||
nix build .#nixosConfigurations.<target>.config.system.build.toplevel
|
|
||||||
```
|
|
||||||
Confirm each builds without evaluation errors before considering it done. If a
|
|
||||||
target fails to build, report which one and why rather than silently skipping it.
|
|
||||||
|
|
||||||
## Deliverables
|
|
||||||
|
|
||||||
- Refactored `flake.nix` using the composed module + generated-matrix approach.
|
|
||||||
- New `modules/build-types/*.nix` and `modules/platforms/*.nix` files.
|
|
||||||
- Old flat target names removed only after their replacements are verified.
|
|
||||||
- A short `README.md` (or section in existing docs) listing the final target
|
|
||||||
names and what each one is for.
|
|
||||||
- A summary at the end of what changed, what was removed, and any of the Open
|
|
||||||
Questions above that got resolved differently than expected.
|
|
||||||
|
|
||||||
## Out of scope
|
|
||||||
|
|
||||||
- Don't touch `~/nix-auto-installer` contents beyond checking it for references
|
|
||||||
to the old target names — if changes there are needed, flag them, don't make
|
|
||||||
them without confirming.
|
|
||||||
- Don't add new build types or platforms beyond the ones listed here.
|
|
||||||
Generated
+157
-7
@@ -1,5 +1,62 @@
|
|||||||
{
|
{
|
||||||
"nodes": {
|
"nodes": {
|
||||||
|
"clan-core": {
|
||||||
|
"inputs": {
|
||||||
|
"data-mesher": "data-mesher",
|
||||||
|
"disko": [
|
||||||
|
"disko"
|
||||||
|
],
|
||||||
|
"flake-parts": "flake-parts",
|
||||||
|
"nix-darwin": "nix-darwin",
|
||||||
|
"nix-select": "nix-select",
|
||||||
|
"nixpkgs": [
|
||||||
|
"nixpkgs"
|
||||||
|
],
|
||||||
|
"sops-nix": [
|
||||||
|
"sops-nix"
|
||||||
|
],
|
||||||
|
"systems": "systems",
|
||||||
|
"treefmt-nix": "treefmt-nix"
|
||||||
|
},
|
||||||
|
"locked": {
|
||||||
|
"lastModified": 1783497933,
|
||||||
|
"narHash": "sha256-TxmwEews6URFPqOWEHNychtXbFDgLZjbOfEXtvtOm6U=",
|
||||||
|
"rev": "3dc0221ca09033599fe98055e9bbc81bdf32732a",
|
||||||
|
"type": "tarball",
|
||||||
|
"url": "https://git.clan.lol/api/v1/repos/clan/clan-core/archive/3dc0221ca09033599fe98055e9bbc81bdf32732a.tar.gz"
|
||||||
|
},
|
||||||
|
"original": {
|
||||||
|
"type": "tarball",
|
||||||
|
"url": "https://git.clan.lol/clan/clan-core/archive/26.05.tar.gz"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"data-mesher": {
|
||||||
|
"inputs": {
|
||||||
|
"flake-parts": [
|
||||||
|
"clan-core",
|
||||||
|
"flake-parts"
|
||||||
|
],
|
||||||
|
"nixpkgs": [
|
||||||
|
"clan-core",
|
||||||
|
"nixpkgs"
|
||||||
|
],
|
||||||
|
"treefmt-nix": [
|
||||||
|
"clan-core",
|
||||||
|
"treefmt-nix"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"locked": {
|
||||||
|
"lastModified": 1778718524,
|
||||||
|
"narHash": "sha256-pXLoI6Ax0EnUK6r34UM1vibVC7CfTu6j72R2692ZzPs=",
|
||||||
|
"rev": "12c552ad547d87254f33f33bddd1a2cdbeac754d",
|
||||||
|
"type": "tarball",
|
||||||
|
"url": "https://git.clan.lol/api/v1/repos/clan/data-mesher/archive/12c552ad547d87254f33f33bddd1a2cdbeac754d.tar.gz"
|
||||||
|
},
|
||||||
|
"original": {
|
||||||
|
"type": "tarball",
|
||||||
|
"url": "https://git.clan.lol/clan/data-mesher/archive/main.tar.gz"
|
||||||
|
}
|
||||||
|
},
|
||||||
"disko": {
|
"disko": {
|
||||||
"inputs": {
|
"inputs": {
|
||||||
"nixpkgs": [
|
"nixpkgs": [
|
||||||
@@ -51,9 +108,30 @@
|
|||||||
"type": "github"
|
"type": "github"
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
|
"flake-parts": {
|
||||||
|
"inputs": {
|
||||||
|
"nixpkgs-lib": [
|
||||||
|
"clan-core",
|
||||||
|
"nixpkgs"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"locked": {
|
||||||
|
"lastModified": 1778716662,
|
||||||
|
"narHash": "sha256-m1Yf0wZ8j1OHjTc2UwHwyQRSnNeSgLJOd7q5Y45hzi4=",
|
||||||
|
"owner": "hercules-ci",
|
||||||
|
"repo": "flake-parts",
|
||||||
|
"rev": "f7c1a2d347e4c52d5fb8d10cb4d94b5884e546fb",
|
||||||
|
"type": "github"
|
||||||
|
},
|
||||||
|
"original": {
|
||||||
|
"owner": "hercules-ci",
|
||||||
|
"repo": "flake-parts",
|
||||||
|
"type": "github"
|
||||||
|
}
|
||||||
|
},
|
||||||
"flake-utils": {
|
"flake-utils": {
|
||||||
"inputs": {
|
"inputs": {
|
||||||
"systems": "systems"
|
"systems": "systems_2"
|
||||||
},
|
},
|
||||||
"locked": {
|
"locked": {
|
||||||
"lastModified": 1694529238,
|
"lastModified": 1694529238,
|
||||||
@@ -95,11 +173,11 @@
|
|||||||
]
|
]
|
||||||
},
|
},
|
||||||
"locked": {
|
"locked": {
|
||||||
"lastModified": 1783740085,
|
"lastModified": 1785119570,
|
||||||
"narHash": "sha256-qajyHfZY29G2oEQk+uHxmsJcRoBUBXP9maTpFlwP/dI=",
|
"narHash": "sha256-Rgs2xKnGLFWQscxUaXX07oyZeuMDOHEbqDOsgliLFGM=",
|
||||||
"owner": "nix-community",
|
"owner": "nix-community",
|
||||||
"repo": "home-manager",
|
"repo": "home-manager",
|
||||||
"rev": "3cd22efe6471dc7365c822bd9ad73a21e55f38fb",
|
"rev": "d4fd24667c8cbef124bb70a20380cab75ec8474d",
|
||||||
"type": "github"
|
"type": "github"
|
||||||
},
|
},
|
||||||
"original": {
|
"original": {
|
||||||
@@ -109,6 +187,40 @@
|
|||||||
"type": "github"
|
"type": "github"
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
|
"nix-darwin": {
|
||||||
|
"inputs": {
|
||||||
|
"nixpkgs": [
|
||||||
|
"clan-core",
|
||||||
|
"nixpkgs"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"locked": {
|
||||||
|
"lastModified": 1779036909,
|
||||||
|
"narHash": "sha256-zXcwYQGCT6pzinK+1dBB2ekTVtfxGZAapb3Evdcu4fY=",
|
||||||
|
"owner": "nix-darwin",
|
||||||
|
"repo": "nix-darwin",
|
||||||
|
"rev": "56c666e108467d87d13508936aade6d567f2a501",
|
||||||
|
"type": "github"
|
||||||
|
},
|
||||||
|
"original": {
|
||||||
|
"owner": "nix-darwin",
|
||||||
|
"repo": "nix-darwin",
|
||||||
|
"type": "github"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"nix-select": {
|
||||||
|
"locked": {
|
||||||
|
"lastModified": 1763303120,
|
||||||
|
"narHash": "sha256-yxcNOha7Cfv2nhVpz9ZXSNKk0R7wt4AiBklJ8D24rVg=",
|
||||||
|
"rev": "3d1e3860bef36857a01a2ddecba7cdb0a14c35a9",
|
||||||
|
"type": "tarball",
|
||||||
|
"url": "https://git.clan.lol/api/v1/repos/clan/nix-select/archive/3d1e3860bef36857a01a2ddecba7cdb0a14c35a9.tar.gz"
|
||||||
|
},
|
||||||
|
"original": {
|
||||||
|
"type": "tarball",
|
||||||
|
"url": "https://git.clan.lol/clan/nix-select/archive/main.tar.gz"
|
||||||
|
}
|
||||||
|
},
|
||||||
"nixos-conf-editor": {
|
"nixos-conf-editor": {
|
||||||
"inputs": {
|
"inputs": {
|
||||||
"flake-compat": "flake-compat",
|
"flake-compat": "flake-compat",
|
||||||
@@ -147,11 +259,11 @@
|
|||||||
},
|
},
|
||||||
"nixpkgs_2": {
|
"nixpkgs_2": {
|
||||||
"locked": {
|
"locked": {
|
||||||
"lastModified": 1784011430,
|
"lastModified": 1785133411,
|
||||||
"narHash": "sha256-lDebytrYdd47IBLwvNOD+6AGeoqZ78CIKlp70hzW280=",
|
"narHash": "sha256-Yjv0WEg39KRYS0rBdTbu6Fc/or/ihAKk13W9sQ6VWd0=",
|
||||||
"owner": "NixOS",
|
"owner": "NixOS",
|
||||||
"repo": "nixpkgs",
|
"repo": "nixpkgs",
|
||||||
"rev": "8eeec934ae0dbeca3d7868c059568a65c08b2fc3",
|
"rev": "2f5a153c270b70cb0f8c11f46d96d6d3bc39f4e3",
|
||||||
"type": "github"
|
"type": "github"
|
||||||
},
|
},
|
||||||
"original": {
|
"original": {
|
||||||
@@ -163,6 +275,7 @@
|
|||||||
},
|
},
|
||||||
"root": {
|
"root": {
|
||||||
"inputs": {
|
"inputs": {
|
||||||
|
"clan-core": "clan-core",
|
||||||
"disko": "disko",
|
"disko": "disko",
|
||||||
"home-manager": "home-manager",
|
"home-manager": "home-manager",
|
||||||
"nixos-conf-editor": "nixos-conf-editor",
|
"nixos-conf-editor": "nixos-conf-editor",
|
||||||
@@ -214,6 +327,22 @@
|
|||||||
}
|
}
|
||||||
},
|
},
|
||||||
"systems": {
|
"systems": {
|
||||||
|
"locked": {
|
||||||
|
"lastModified": 1774449309,
|
||||||
|
"narHash": "sha256-brhZ8DmuGtzkCYHJg4HEd602amKm89Y9ytsFZ5uWD1w=",
|
||||||
|
"owner": "nix-systems",
|
||||||
|
"repo": "default",
|
||||||
|
"rev": "c29398b59d2048c4ab79345812849c9bd15e9150",
|
||||||
|
"type": "github"
|
||||||
|
},
|
||||||
|
"original": {
|
||||||
|
"owner": "nix-systems",
|
||||||
|
"ref": "future-26.11",
|
||||||
|
"repo": "default",
|
||||||
|
"type": "github"
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"systems_2": {
|
||||||
"locked": {
|
"locked": {
|
||||||
"lastModified": 1681028828,
|
"lastModified": 1681028828,
|
||||||
"narHash": "sha256-Vy1rq5AaRuLzOxct8nz4T6wlgyUR7zLU309k9mBC768=",
|
"narHash": "sha256-Vy1rq5AaRuLzOxct8nz4T6wlgyUR7zLU309k9mBC768=",
|
||||||
@@ -227,6 +356,27 @@
|
|||||||
"repo": "default",
|
"repo": "default",
|
||||||
"type": "github"
|
"type": "github"
|
||||||
}
|
}
|
||||||
|
},
|
||||||
|
"treefmt-nix": {
|
||||||
|
"inputs": {
|
||||||
|
"nixpkgs": [
|
||||||
|
"clan-core",
|
||||||
|
"nixpkgs"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"locked": {
|
||||||
|
"lastModified": 1780220602,
|
||||||
|
"narHash": "sha256-eynAfOmbmxJnkp7YewvCEbShNnnYJ9gLLqkzsYtBPeM=",
|
||||||
|
"owner": "numtide",
|
||||||
|
"repo": "treefmt-nix",
|
||||||
|
"rev": "db947814a175b7ca6ded66e21383d938df01c227",
|
||||||
|
"type": "github"
|
||||||
|
},
|
||||||
|
"original": {
|
||||||
|
"owner": "numtide",
|
||||||
|
"repo": "treefmt-nix",
|
||||||
|
"type": "github"
|
||||||
|
}
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
"root": "root",
|
"root": "root",
|
||||||
|
|||||||
@@ -16,6 +16,19 @@
|
|||||||
url = "github:Mic92/sops-nix";
|
url = "github:Mic92/sops-nix";
|
||||||
inputs.nixpkgs.follows = "nixpkgs";
|
inputs.nixpkgs.follows = "nixpkgs";
|
||||||
};
|
};
|
||||||
|
clan-core = {
|
||||||
|
url = "https://git.clan.lol/clan/clan-core/archive/26.05.tar.gz";
|
||||||
|
# Deduplicate modules: clan-core bundles its own disko and sops-nix
|
||||||
|
# (both imported by nixosModules.clanCore). Without follows, we'd get
|
||||||
|
# two different versions of each, and disko's _module.args.diskoLib
|
||||||
|
# unique option would conflict. With follows, clan-core uses the same
|
||||||
|
# store paths as us, so NixOS deduplicates the imports.
|
||||||
|
inputs = {
|
||||||
|
nixpkgs.follows = "nixpkgs";
|
||||||
|
disko.follows = "disko";
|
||||||
|
sops-nix.follows = "sops-nix";
|
||||||
|
};
|
||||||
|
};
|
||||||
};
|
};
|
||||||
|
|
||||||
outputs = { self, nixpkgs, nixos-conf-editor, home-manager, sops-nix, ... } @ inputs:
|
outputs = { self, nixpkgs, nixos-conf-editor, home-manager, sops-nix, ... } @ inputs:
|
||||||
@@ -23,6 +36,8 @@
|
|||||||
let
|
let
|
||||||
system = "x86_64-linux";
|
system = "x86_64-linux";
|
||||||
inherit (nixpkgs) lib;
|
inherit (nixpkgs) lib;
|
||||||
|
pkgs = nixpkgs.legacyPackages.${system};
|
||||||
|
vars = import ./variables.nix;
|
||||||
|
|
||||||
# Generates a nixosConfiguration from a platform (what it runs on) and
|
# Generates a nixosConfiguration from a platform (what it runs on) and
|
||||||
# a build type (what it's for), plus the per-identity host.nix that
|
# a build type (what it's for), plus the per-identity host.nix that
|
||||||
@@ -30,47 +45,70 @@
|
|||||||
# (hostName, hostId, per-machine secrets). Every build type except
|
# (hostName, hostId, per-machine secrets). Every build type except
|
||||||
# nix-cache itself consumes the nix-cache substituter and remote
|
# nix-cache itself consumes the nix-cache substituter and remote
|
||||||
# builder.
|
# builder.
|
||||||
mkTarget = { platform, buildType, hostPath, homeFile ? ./modules/common/home.nix }:
|
mkTarget = { platform, buildType, hostPath, homeFile ? ./modules/common/home.nix, nameSuffix ? "" }:
|
||||||
|
let
|
||||||
|
flakeTarget = "${platform}-${buildType}${nameSuffix}";
|
||||||
|
in
|
||||||
nixpkgs.lib.nixosSystem {
|
nixpkgs.lib.nixosSystem {
|
||||||
inherit system;
|
inherit system;
|
||||||
modules = [
|
modules = [
|
||||||
inputs.disko.nixosModules.disko
|
inputs.disko.nixosModules.disko
|
||||||
sops-nix.nixosModules.sops
|
sops-nix.nixosModules.sops
|
||||||
|
inputs.clan-core.nixosModules.clanCore
|
||||||
|
{
|
||||||
|
# Required clan settings. directory is the flake root (where
|
||||||
|
# vars/ and sops/ directories live); machine.name is the flake
|
||||||
|
# target name (matches what clan vars generate uses as the key
|
||||||
|
# under vars/per-machine/). enableRecommendedDefaults = false
|
||||||
|
# is mandatory: without it, clan unconditionally enables
|
||||||
|
# networking.useNetworkd, adds packages, and tweaks nix settings
|
||||||
|
# -- none of which belong here.
|
||||||
|
clan.core = {
|
||||||
|
settings.directory = self;
|
||||||
|
settings.machine.name = flakeTarget;
|
||||||
|
enableRecommendedDefaults = false;
|
||||||
|
};
|
||||||
|
}
|
||||||
|
./modules/clan/ssh-host-key.nix
|
||||||
./modules/common/configuration.nix
|
./modules/common/configuration.nix
|
||||||
./modules/platforms/${platform}.nix
|
./modules/platforms/${platform}.nix
|
||||||
./modules/build-types/${buildType}.nix
|
./modules/build-types/${buildType}.nix
|
||||||
hostPath
|
hostPath
|
||||||
{ environment.etc."flake-target".text = "${platform}-${buildType}"; }
|
{ environment.etc."flake-target".text = flakeTarget; }
|
||||||
home-manager.nixosModules.home-manager
|
home-manager.nixosModules.home-manager
|
||||||
{
|
{
|
||||||
home-manager = {
|
home-manager = {
|
||||||
useGlobalPkgs = true;
|
useGlobalPkgs = true;
|
||||||
useUserPackages = true;
|
useUserPackages = true;
|
||||||
|
extraSpecialArgs = { inherit vars; };
|
||||||
users.nixos = import homeFile;
|
users.nixos = import homeFile;
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
] ++ lib.optionals (buildType != "nix-cache") [
|
] ++ lib.optionals (buildType != "nix-cache") [
|
||||||
./modules/nix-cache/client.nix
|
./modules/nix-cache/client.nix
|
||||||
./modules/remote-builder-client.nix
|
./modules/nix-cache/remote-builder-client.nix
|
||||||
];
|
];
|
||||||
specialArgs = { inherit inputs; };
|
# flakeTarget is passed via specialArgs (not read back from
|
||||||
|
# config.environment.etc."flake-target" above) specifically so
|
||||||
|
# modules/platforms/lxc.nix can use it to select its own host key
|
||||||
|
# file without a same-option circular dependency (a module
|
||||||
|
# contributing to environment.etc can't read the merged
|
||||||
|
# environment.etc it's itself contributing to).
|
||||||
|
specialArgs = { inherit inputs vars netbootSystem netbootMinimalSystem flakeTarget; };
|
||||||
};
|
};
|
||||||
|
|
||||||
# Generated platform x build-type matrix. pxe-boot has no linode
|
# Generated platform x build-type matrix. pxe-boot has no linode
|
||||||
# variant (PXE/DHCP/TFTP need LAN L2 adjacency, which a Linode VPS
|
# variant (PXE/DHCP/TFTP need LAN L2 adjacency, which a Linode VPS
|
||||||
# doesn't have).
|
# doesn't have).
|
||||||
generatedTargets = {
|
generatedTargets = {
|
||||||
linode-minimal = mkTarget { platform = "linode"; buildType = "minimal"; hostPath = ./hosts/linode-minimal/host.nix; };
|
linode-minimal = mkTarget { platform = "linode"; buildType = "minimal"; hostPath = ./hosts/nix-minimal/host.nix; };
|
||||||
proxmox-minimal = mkTarget { platform = "proxmox"; buildType = "minimal"; hostPath = ./hosts/proxmox-minimal/host.nix; };
|
proxmox-minimal = mkTarget { platform = "proxmox"; buildType = "minimal"; hostPath = ./hosts/nix-minimal/host.nix; };
|
||||||
lxc-minimal = mkTarget { platform = "lxc"; buildType = "minimal"; hostPath = ./hosts/lxc-minimal/host.nix; };
|
lxc-minimal = mkTarget { platform = "lxc"; buildType = "minimal"; hostPath = ./hosts/nix-minimal/host.nix; };
|
||||||
|
|
||||||
linode-nix-cache = mkTarget { platform = "linode"; buildType = "nix-cache"; hostPath = ./hosts/nix-cache/host.nix; };
|
linode-nix-cache = mkTarget { platform = "linode"; buildType = "nix-cache"; hostPath = ./hosts/nix-cache/host.nix; };
|
||||||
proxmox-nix-cache = mkTarget { platform = "proxmox"; buildType = "nix-cache"; hostPath = ./hosts/nix-cache/host.nix; };
|
proxmox-nix-cache = mkTarget { platform = "proxmox"; buildType = "nix-cache"; hostPath = ./hosts/nix-cache/host.nix; };
|
||||||
lxc-nix-cache = mkTarget { platform = "lxc"; buildType = "nix-cache"; hostPath = ./hosts/nix-cache/host.nix; };
|
lxc-nix-cache = mkTarget { platform = "lxc"; buildType = "nix-cache"; hostPath = ./hosts/nix-cache/host.nix; };
|
||||||
|
|
||||||
linode-server = mkTarget { platform = "linode"; buildType = "server"; hostPath = ./hosts/server/host.nix; };
|
|
||||||
proxmox-server = mkTarget { platform = "proxmox"; buildType = "server"; hostPath = ./hosts/server/host.nix; };
|
|
||||||
lxc-server = mkTarget { platform = "lxc"; buildType = "server"; hostPath = ./hosts/server/host.nix; };
|
|
||||||
|
|
||||||
linode-docker = mkTarget { platform = "linode"; buildType = "docker"; hostPath = ./hosts/docker/host.nix; };
|
linode-docker = mkTarget { platform = "linode"; buildType = "docker"; hostPath = ./hosts/docker/host.nix; };
|
||||||
proxmox-docker = mkTarget { platform = "proxmox"; buildType = "docker"; hostPath = ./hosts/docker/host.nix; };
|
proxmox-docker = mkTarget { platform = "proxmox"; buildType = "docker"; hostPath = ./hosts/docker/host.nix; };
|
||||||
@@ -79,14 +117,130 @@
|
|||||||
linode-gui = mkTarget { platform = "linode"; buildType = "gui"; hostPath = ./hosts/nixos/host.nix; homeFile = ./hosts/nixos/home.nix; };
|
linode-gui = mkTarget { platform = "linode"; buildType = "gui"; hostPath = ./hosts/nixos/host.nix; homeFile = ./hosts/nixos/home.nix; };
|
||||||
proxmox-gui = mkTarget { platform = "proxmox"; buildType = "gui"; hostPath = ./hosts/nixos/host.nix; homeFile = ./hosts/nixos/home.nix; };
|
proxmox-gui = mkTarget { platform = "proxmox"; buildType = "gui"; hostPath = ./hosts/nixos/host.nix; homeFile = ./hosts/nixos/home.nix; };
|
||||||
lxc-gui = mkTarget { platform = "lxc"; buildType = "gui"; hostPath = ./hosts/nixos/host.nix; homeFile = ./hosts/nixos/home.nix; };
|
lxc-gui = mkTarget { platform = "lxc"; buildType = "gui"; hostPath = ./hosts/nixos/host.nix; homeFile = ./hosts/nixos/home.nix; };
|
||||||
|
baremetal-gui = mkTarget { platform = "baremetal"; buildType = "gui"; hostPath = ./hosts/nixos/host.nix; homeFile = ./hosts/nixos/home.nix; };
|
||||||
|
|
||||||
proxmox-pxe-boot = mkTarget { platform = "proxmox"; buildType = "pxe-boot"; hostPath = ./hosts/pxe-boot/host.nix; };
|
proxmox-pxe-boot = mkTarget { platform = "proxmox"; buildType = "pxe-boot"; hostPath = ./hosts/pxe-boot/host.nix; };
|
||||||
lxc-pxe-boot = mkTarget { platform = "lxc"; buildType = "pxe-boot"; hostPath = ./hosts/pxe-boot/host.nix; };
|
lxc-pxe-boot = mkTarget { platform = "lxc"; buildType = "pxe-boot"; hostPath = ./hosts/pxe-boot/host.nix; };
|
||||||
|
|
||||||
|
linode-tailscale-router = mkTarget { platform = "linode"; buildType = "tailscale-router"; hostPath = ./hosts/tailscale-router/host.nix; };
|
||||||
|
proxmox-tailscale-router = mkTarget { platform = "proxmox"; buildType = "tailscale-router"; hostPath = ./hosts/tailscale-router/host.nix; };
|
||||||
|
lxc-tailscale-router = mkTarget { platform = "lxc"; buildType = "tailscale-router"; hostPath = ./hosts/tailscale-router/host.nix; };
|
||||||
|
|
||||||
|
lxc-tor-relay = mkTarget { platform = "lxc"; buildType = "tor-relay"; hostPath = ./hosts/tor-relay/host.nix; };
|
||||||
|
|
||||||
|
proxmox-ha-server-1 = mkTarget { platform = "proxmox"; buildType = "ha-server"; hostPath = ./hosts/ha-server-1/host.nix; nameSuffix = "-1"; };
|
||||||
|
proxmox-ha-server-2 = mkTarget { platform = "proxmox"; buildType = "ha-server"; hostPath = ./hosts/ha-server-2/host.nix; nameSuffix = "-2"; };
|
||||||
|
|
||||||
|
proxmox-ha-docker-1 = mkTarget { platform = "proxmox"; buildType = "ha-docker"; hostPath = ./hosts/ha-docker-1/host.nix; nameSuffix = "-1"; };
|
||||||
|
proxmox-ha-docker-2 = mkTarget { platform = "proxmox"; buildType = "ha-docker"; hostPath = ./hosts/ha-docker-2/host.nix; nameSuffix = "-2"; };
|
||||||
|
};
|
||||||
|
|
||||||
|
# Auto-install environments (migrated from the former nix-auto-installer
|
||||||
|
# flake): a self-contained NixOS installer that boots, discovers this
|
||||||
|
# flake's own nixosConfigurations over the network, and runs
|
||||||
|
# nixos-install against whichever one the operator picks. These are
|
||||||
|
# deliberately not part of the platform x build-type matrix above —
|
||||||
|
# they're throwaway boot media, not persistent hosts, so they skip
|
||||||
|
# disko/sops-nix/home-manager and just need `vars`.
|
||||||
|
installerTargets = {
|
||||||
|
installer = nixpkgs.lib.nixosSystem {
|
||||||
|
inherit system;
|
||||||
|
modules = [ ./modules/installer/iso.nix ];
|
||||||
|
specialArgs = { inherit vars; };
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
# Same installer environment, built as netboot (kernel + initrd +
|
||||||
|
# iPXE script) instead of an ISO — this is what packages.pxe bundles.
|
||||||
|
#
|
||||||
|
# Deliberately imports common.nix directly, NOT ./modules/installer/iso.nix
|
||||||
|
# (which pulls in nixpkgs' installation-cd-minimal.nix) -- confirmed live
|
||||||
|
# that composing the ISO module together with netboot-minimal.nix hangs
|
||||||
|
# every boot waiting for a device that can never exist on a netboot
|
||||||
|
# client ("A start job is running for /dev/disk/by-label/nixos-minimal-...").
|
||||||
|
# Both installation-cd-base.nix and netboot.nix set fileSystems."/" via
|
||||||
|
# the identical lib.mkImageMediaOverride (mkOverride 60) priority --
|
||||||
|
# genuinely conflicting root-filesystem strategies (ISO-by-label vs.
|
||||||
|
# netboot-tmpfs) at the same priority, and the ISO one was winning.
|
||||||
|
# netboot-minimal.nix's own chain (netboot-base.nix) already imports
|
||||||
|
# profiles/installation-device.nix independently, so common.nix's
|
||||||
|
# initialHashedPassword override (which assumes that profile is
|
||||||
|
# present) still applies correctly without iso.nix in the mix.
|
||||||
|
#
|
||||||
|
# networking.hostName is set explicitly (rather than left at nixpkgs'
|
||||||
|
# own "nixos" default) so this image's generated system name
|
||||||
|
# (nixos-system-auto-installer-*) matches its iPXE menu entry —
|
||||||
|
# see modules/build-types/pxe-boot.nix's :auto-installer item — and
|
||||||
|
# its staged directory, /srv/pxe/http/auto-installer.
|
||||||
|
netbootSystem = nixpkgs.lib.nixosSystem {
|
||||||
|
inherit system;
|
||||||
|
modules = [
|
||||||
|
./modules/installer/common.nix
|
||||||
|
({ modulesPath, ... }: {
|
||||||
|
imports = [
|
||||||
|
(modulesPath + "/installer/netboot/netboot-minimal.nix")
|
||||||
|
];
|
||||||
|
})
|
||||||
|
{ networking.hostName = "auto-installer"; }
|
||||||
|
];
|
||||||
|
specialArgs = { inherit vars; };
|
||||||
|
};
|
||||||
|
|
||||||
|
# A genuinely vanilla NixOS minimal netboot image: nixpkgs'
|
||||||
|
# netboot-minimal.nix on its own, with none of this flake's
|
||||||
|
# auto-installer wiring (no common.nix — no auto-install.sh, no
|
||||||
|
# baked host keys, no custom users/passwords). Built from source via
|
||||||
|
# the same nixosSystem + netboot-minimal.nix path as netbootSystem
|
||||||
|
# above, so both go through an identical build mechanism; the only
|
||||||
|
# difference is what's composed in. hostName again matches this
|
||||||
|
# image's iPXE menu entry (:nixos-minimal) and staged directory
|
||||||
|
# (/srv/pxe/http/nixos-minimal).
|
||||||
|
netbootMinimalSystem = nixpkgs.lib.nixosSystem {
|
||||||
|
inherit system;
|
||||||
|
modules = [
|
||||||
|
({ modulesPath, ... }: {
|
||||||
|
imports = [
|
||||||
|
(modulesPath + "/installer/netboot/netboot-minimal.nix")
|
||||||
|
];
|
||||||
|
})
|
||||||
|
{
|
||||||
|
networking.hostName = "nixos-minimal";
|
||||||
|
system.stateVersion = "26.05";
|
||||||
|
boot.zfs.forceImportRoot = false;
|
||||||
|
}
|
||||||
|
];
|
||||||
};
|
};
|
||||||
|
|
||||||
in
|
in
|
||||||
{
|
{
|
||||||
|
|
||||||
nixosConfigurations = generatedTargets;
|
nixosConfigurations = generatedTargets // installerTargets;
|
||||||
|
|
||||||
|
# Buildable auto-installer artifacts (`nix build .#<name>`). No `lxc`
|
||||||
|
# variant (installer-boots-as-an-LXC-container) or `all` bundle
|
||||||
|
# anymore — lxc-* and proxmox-* hosts deploy via their own tarball/
|
||||||
|
# disk-image outputs instead (see docs/auto-installer.md and
|
||||||
|
# docs/proxmox-images.md), which left the installer's own LXC form
|
||||||
|
# with no real use case: it's excluded from the install menu (same
|
||||||
|
# bind-mount problem as any LXC nixos-install target) and nothing
|
||||||
|
# else needed booting the installer itself as a container.
|
||||||
|
packages.${system} = {
|
||||||
|
iso = installerTargets.installer.config.system.build.isoImage;
|
||||||
|
|
||||||
|
pxe = pkgs.linkFarm "pxe" [
|
||||||
|
{ name = "netboot.ipxe"; path = netbootSystem.config.system.build.netbootIpxeScript; }
|
||||||
|
{ name = "initrd"; path = netbootSystem.config.system.build.netbootRamdisk; }
|
||||||
|
{ name = "kernel"; path = netbootSystem.config.system.build.kernel; }
|
||||||
|
];
|
||||||
|
|
||||||
|
# Vanilla NixOS minimal netboot bundle — see netbootMinimalSystem
|
||||||
|
# above. Staged onto the pxe-boot host alongside packages.pxe by
|
||||||
|
# modules/pxe-boot/stage-installer-artifacts.nix.
|
||||||
|
pxe-minimal = pkgs.linkFarm "pxe-minimal" [
|
||||||
|
{ name = "netboot.ipxe"; path = netbootMinimalSystem.config.system.build.netbootIpxeScript; }
|
||||||
|
{ name = "initrd"; path = netbootMinimalSystem.config.system.build.netbootRamdisk; }
|
||||||
|
{ name = "kernel"; path = netbootMinimalSystem.config.system.build.kernel; }
|
||||||
|
];
|
||||||
|
};
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|||||||
+17
-3
@@ -1,10 +1,24 @@
|
|||||||
{ ... }:
|
{ vars, ... }:
|
||||||
|
|
||||||
{
|
{
|
||||||
networking.hostName = "docker";
|
networking = {
|
||||||
networking.hostId = "007f0200";
|
hostName = "docker";
|
||||||
|
hostId = "007f0200";
|
||||||
|
useDHCP = false;
|
||||||
|
interfaces = {
|
||||||
|
${vars.vmLanInterface}.ipv4.addresses = [{ address = vars.dockerIp; prefixLength = vars.lanPrefixLength; }];
|
||||||
|
${vars.lxcStorageInterface}.ipv4.addresses = [{ address = vars.dockerStorageIp; prefixLength = vars.haClientPrefixLength; }];
|
||||||
|
};
|
||||||
|
defaultGateway = { address = vars.lanGateway; interface = vars.vmLanInterface; };
|
||||||
|
nameservers = [ vars.domainControllerIp ];
|
||||||
|
};
|
||||||
boot.zfs.forceImportRoot = false;
|
boot.zfs.forceImportRoot = false;
|
||||||
|
|
||||||
|
# Only advertise the LAN interface to IPA DNS. Without this, SSSD registers
|
||||||
|
# every Docker bridge (172.x.x.x) as an A record for docker.sweet.home —
|
||||||
|
# the default dyndns.interface = "*" catches them all.
|
||||||
|
security.ipa.dyndns.interface = vars.lxcLanInterface; # eth0
|
||||||
|
|
||||||
# Preserved from the pre-refactor `docker` target — stateVersion must never
|
# Preserved from the pre-refactor `docker` target — stateVersion must never
|
||||||
# be bumped on an already-installed machine.
|
# be bumped on an already-installed machine.
|
||||||
system.stateVersion = "25.05";
|
system.stateVersion = "25.05";
|
||||||
|
|||||||
@@ -0,0 +1,34 @@
|
|||||||
|
{ vars, ... }:
|
||||||
|
{
|
||||||
|
networking = {
|
||||||
|
hostName = vars.haDocker1Host;
|
||||||
|
hostId = "a1d0c4e1";
|
||||||
|
useDHCP = false;
|
||||||
|
interfaces = {
|
||||||
|
# ens18 — LAN management NIC (vmbr0, 192.168.2.0/24)
|
||||||
|
${vars.vmLanInterface}.ipv4.addresses = [{
|
||||||
|
address = vars.haDocker1Ip;
|
||||||
|
prefixLength = vars.lanPrefixLength;
|
||||||
|
}];
|
||||||
|
# ens19 — storage-client NIC (vmbr2, 192.168.20.0/24) — NFS from HA cluster
|
||||||
|
${vars.haDockerStorageInterface}.ipv4.addresses = [{
|
||||||
|
address = vars.haDocker1StorageIp;
|
||||||
|
prefixLength = vars.haClientPrefixLength;
|
||||||
|
}];
|
||||||
|
# ens20 — swarm cluster NIC (vmbr3, 192.168.30.0/24) — Docker gossip + VXLAN
|
||||||
|
${vars.haDockerSwarmInterface}.ipv4.addresses = [{
|
||||||
|
address = vars.haDocker1SwarmIp;
|
||||||
|
prefixLength = vars.haDockerSwarmPrefixLength;
|
||||||
|
}];
|
||||||
|
};
|
||||||
|
defaultGateway = { address = vars.lanGateway; interface = vars.vmLanInterface; };
|
||||||
|
nameservers = [ vars.domainControllerIp ];
|
||||||
|
};
|
||||||
|
|
||||||
|
# Only register the LAN IP with IPA DNS. Without this, sssd dyndns
|
||||||
|
# would also register Docker bridge IPs (172.x.x.x) and the storage/swarm
|
||||||
|
# NIC IPs as A records for ha-docker-1.sweet.home.
|
||||||
|
security.ipa.dyndns.interface = vars.vmLanInterface;
|
||||||
|
|
||||||
|
system.stateVersion = "26.05";
|
||||||
|
}
|
||||||
@@ -0,0 +1,32 @@
|
|||||||
|
{ vars, ... }:
|
||||||
|
{
|
||||||
|
networking = {
|
||||||
|
hostName = vars.haDocker2Host;
|
||||||
|
hostId = "a2d0c4e2";
|
||||||
|
useDHCP = false;
|
||||||
|
interfaces = {
|
||||||
|
# ens18 — LAN management NIC (vmbr0, 192.168.2.0/24)
|
||||||
|
${vars.vmLanInterface}.ipv4.addresses = [{
|
||||||
|
address = vars.haDocker2Ip;
|
||||||
|
prefixLength = vars.lanPrefixLength;
|
||||||
|
}];
|
||||||
|
# ens19 — storage-client NIC (vmbr2, 192.168.20.0/24) — NFS from HA cluster
|
||||||
|
${vars.haDockerStorageInterface}.ipv4.addresses = [{
|
||||||
|
address = vars.haDocker2StorageIp;
|
||||||
|
prefixLength = vars.haClientPrefixLength;
|
||||||
|
}];
|
||||||
|
# ens20 — swarm cluster NIC (vmbr3, 192.168.30.0/24) — Docker gossip + VXLAN
|
||||||
|
${vars.haDockerSwarmInterface}.ipv4.addresses = [{
|
||||||
|
address = vars.haDocker2SwarmIp;
|
||||||
|
prefixLength = vars.haDockerSwarmPrefixLength;
|
||||||
|
}];
|
||||||
|
};
|
||||||
|
defaultGateway = { address = vars.lanGateway; interface = vars.vmLanInterface; };
|
||||||
|
nameservers = [ vars.domainControllerIp ];
|
||||||
|
};
|
||||||
|
|
||||||
|
# Only register the LAN IP with IPA DNS — same reasoning as ha-docker-1.
|
||||||
|
security.ipa.dyndns.interface = vars.vmLanInterface;
|
||||||
|
|
||||||
|
system.stateVersion = "26.05";
|
||||||
|
}
|
||||||
@@ -0,0 +1,17 @@
|
|||||||
|
{ vars, ... }:
|
||||||
|
{
|
||||||
|
networking = {
|
||||||
|
hostName = vars.haServer1Host;
|
||||||
|
hostId = "3a4b5c6d";
|
||||||
|
useDHCP = false;
|
||||||
|
interfaces = {
|
||||||
|
${vars.vmLanInterface}.ipv4.addresses = [{ address = vars.haServer1Ip; prefixLength = vars.lanPrefixLength; }];
|
||||||
|
${vars.vmStorageInterface}.ipv4.addresses = [{ address = vars.haServer1StorageIp; prefixLength = vars.haStoragePrefixLength; }];
|
||||||
|
${vars.vmStorageClientInterface}.ipv4.addresses = [{ address = vars.haServer1ClientIp; prefixLength = vars.haClientPrefixLength; }];
|
||||||
|
};
|
||||||
|
defaultGateway = { address = vars.lanGateway; interface = vars.vmLanInterface; };
|
||||||
|
nameservers = [ vars.domainControllerIp ];
|
||||||
|
};
|
||||||
|
|
||||||
|
system.stateVersion = "26.05";
|
||||||
|
}
|
||||||
@@ -0,0 +1,17 @@
|
|||||||
|
{ vars, ... }:
|
||||||
|
{
|
||||||
|
networking = {
|
||||||
|
hostName = vars.haServer2Host;
|
||||||
|
hostId = "7e8f9a0b";
|
||||||
|
useDHCP = false;
|
||||||
|
interfaces = {
|
||||||
|
${vars.vmLanInterface}.ipv4.addresses = [{ address = vars.haServer2Ip; prefixLength = vars.lanPrefixLength; }];
|
||||||
|
${vars.vmStorageInterface}.ipv4.addresses = [{ address = vars.haServer2StorageIp; prefixLength = vars.haStoragePrefixLength; }];
|
||||||
|
${vars.vmStorageClientInterface}.ipv4.addresses = [{ address = vars.haServer2ClientIp; prefixLength = vars.haClientPrefixLength; }];
|
||||||
|
};
|
||||||
|
defaultGateway = { address = vars.lanGateway; interface = vars.vmLanInterface; };
|
||||||
|
nameservers = [ vars.domainControllerIp ];
|
||||||
|
};
|
||||||
|
|
||||||
|
system.stateVersion = "26.05";
|
||||||
|
}
|
||||||
@@ -1,9 +0,0 @@
|
|||||||
{ ... }:
|
|
||||||
|
|
||||||
{
|
|
||||||
networking.hostName = "linode-minimal";
|
|
||||||
|
|
||||||
# Preserved from the pre-refactor `linode-minimal` target — stateVersion
|
|
||||||
# must never be bumped on an already-installed machine.
|
|
||||||
system.stateVersion = "26.05";
|
|
||||||
}
|
|
||||||
@@ -1,9 +0,0 @@
|
|||||||
{ ... }:
|
|
||||||
|
|
||||||
{
|
|
||||||
networking.hostName = "lxc-minimal";
|
|
||||||
|
|
||||||
# No pre-existing deployed machine to preserve — pin explicitly to the
|
|
||||||
# current release rather than let it silently default.
|
|
||||||
system.stateVersion = "26.05";
|
|
||||||
}
|
|
||||||
+10
-13
@@ -1,19 +1,16 @@
|
|||||||
{ config, ... }:
|
{ vars, ... }:
|
||||||
|
|
||||||
{
|
{
|
||||||
networking.hostName = "nix-cache";
|
networking = {
|
||||||
|
hostName = vars.nixCacheHost;
|
||||||
sops.secrets."beszel-token".sopsFile = ../../secrets/nix-cache.yaml;
|
useDHCP = false;
|
||||||
sops.templates."nix-cache-beszel.env".content = ''
|
interfaces.${vars.lxcLanInterface}.ipv4.addresses = [{
|
||||||
TOKEN=${config.sops.placeholder."beszel-token"}
|
address = vars.nixCacheIp;
|
||||||
'';
|
prefixLength = vars.lanPrefixLength;
|
||||||
|
}];
|
||||||
services.beszel.agent.environment = {
|
defaultGateway = { address = vars.lanGateway; interface = vars.lxcLanInterface; };
|
||||||
#DOCKER_HOST = "tcp://docker-socket-proxy:2375";
|
nameservers = [ vars.domainControllerIp ];
|
||||||
#HUB_URL = "http://docker.sweet.home:8090";
|
|
||||||
KEY = "ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIFPR9kwtC4TAeTRu46A7+opZsYpxqkRJ+x/ZyB2GWCeG";
|
|
||||||
};
|
};
|
||||||
services.beszel.agent.environmentFile = config.sops.templates."nix-cache-beszel.env".path;
|
|
||||||
|
|
||||||
# Preserved from the pre-refactor `nix-cache` target — stateVersion must
|
# Preserved from the pre-refactor `nix-cache` target — stateVersion must
|
||||||
# never be bumped on an already-installed machine.
|
# never be bumped on an already-installed machine.
|
||||||
|
|||||||
@@ -1,4 +1,4 @@
|
|||||||
{ ... }:
|
_:
|
||||||
|
|
||||||
{
|
{
|
||||||
# Preserves the hostname of the existing, already-deployed machine
|
# Preserves the hostname of the existing, already-deployed machine
|
||||||
+32
-25
@@ -1,4 +1,4 @@
|
|||||||
{ config, pkgs, lib, ... }:
|
{ config, pkgs, lib, vars, ... }:
|
||||||
|
|
||||||
{
|
{
|
||||||
|
|
||||||
@@ -6,41 +6,37 @@
|
|||||||
../../modules/common/aliases.nix
|
../../modules/common/aliases.nix
|
||||||
];
|
];
|
||||||
|
|
||||||
home.username = "nixos"; # your actual username
|
home = {
|
||||||
home.homeDirectory = "/home/nixos";
|
username = vars.primaryUser;
|
||||||
home.stateVersion = "25.05"; # match your NixOS stateVersion
|
homeDirectory = "/home/${vars.primaryUser}";
|
||||||
|
stateVersion = "25.05"; # match your NixOS stateVersion
|
||||||
programs.home-manager.enable = true; # mandatory to activate HM
|
|
||||||
|
|
||||||
# Optional: packages
|
# Optional: packages
|
||||||
home.packages = with pkgs; [
|
packages = with pkgs; [
|
||||||
git
|
git
|
||||||
vim
|
vim
|
||||||
tmux
|
tmux
|
||||||
nextcloud-client
|
nextcloud-client
|
||||||
# vscode
|
# vscode
|
||||||
chromium
|
chromium
|
||||||
|
claude-code
|
||||||
|
fish
|
||||||
|
sops
|
||||||
];
|
];
|
||||||
|
|
||||||
# Optional: set environment vars
|
# Optional: set environment vars
|
||||||
home.sessionVariables = {
|
sessionVariables = {
|
||||||
EDITOR = "vim";
|
EDITOR = "nano";
|
||||||
|
SOPS_AGE_KEY_FILE = "${config.home.homeDirectory}/.config/sops/age/keys.txt";
|
||||||
};
|
};
|
||||||
|
|
||||||
# Optional: enable bash (or zsh, fish...)
|
file = {
|
||||||
programs.bash.enable = true;
|
|
||||||
services.nextcloud-client = {
|
|
||||||
enable = true;
|
|
||||||
# Optionally start in background directly
|
|
||||||
startInBackground = true;
|
|
||||||
};
|
|
||||||
home.file = {
|
|
||||||
".local/share/applications/proxmox-chromium-app.desktop".text = ''
|
".local/share/applications/proxmox-chromium-app.desktop".text = ''
|
||||||
[Desktop Entry]
|
[Desktop Entry]
|
||||||
Type=Application
|
Type=Application
|
||||||
Name=Proxmox (Chromium)
|
Name=Proxmox (Chromium)
|
||||||
Exec=chromium --app=https://pve.sweet.home:8006 --window-size=1920,1080 --window-position=0,0
|
Exec=chromium --app=https://pve.${vars.homeDomain}:${toString vars.ports.pveWeb} --window-size=1920,1080 --window-position=0,0
|
||||||
Icon=/home/nixos/.local/share/icons/proxmox.png
|
Icon=${config.home.homeDirectory}/.local/share/icons/proxmox.png
|
||||||
Terminal=false
|
Terminal=false
|
||||||
Categories=Hypervisor;
|
Categories=Hypervisor;
|
||||||
StartupWMClass=PVE
|
StartupWMClass=PVE
|
||||||
@@ -49,8 +45,8 @@ home.file = {
|
|||||||
[Desktop Entry]
|
[Desktop Entry]
|
||||||
Type=Application
|
Type=Application
|
||||||
Name=Proxmox Backup Server (Chromium)
|
Name=Proxmox Backup Server (Chromium)
|
||||||
Exec=chromium --app=https://192.168.2.108:8007 --window-size=1920,1080 --window-position=0,0
|
Exec=chromium --app=https://${vars.pbsIp}:${toString vars.ports.pbsWeb} --window-size=1920,1080 --window-position=0,0
|
||||||
Icon=/home/nixos/.local/share/icons/proxmox.png
|
Icon=${config.home.homeDirectory}/.local/share/icons/proxmox.png
|
||||||
Terminal=false
|
Terminal=false
|
||||||
Categories=backup;
|
Categories=backup;
|
||||||
|
|
||||||
@@ -59,8 +55,8 @@ home.file = {
|
|||||||
[Desktop Entry]
|
[Desktop Entry]
|
||||||
Type=Application
|
Type=Application
|
||||||
Name=Proxmox (Firefox)
|
Name=Proxmox (Firefox)
|
||||||
Exec=firefox --new-instance https://pve.sweet.home:8006 --profile ProxmoxWebApp --window-size=1920,1080 --class ProxmoxWebApp
|
Exec=firefox --new-instance https://pve.${vars.homeDomain}:${toString vars.ports.pveWeb} --profile ProxmoxWebApp --window-size=1920,1080 --class ProxmoxWebApp
|
||||||
Icon=/home/nixos/.local/share/icons/proxmox.png
|
Icon=${config.home.homeDirectory}/.local/share/icons/proxmox.png
|
||||||
Terminal=false
|
Terminal=false
|
||||||
Categories=Hypervisor;
|
Categories=Hypervisor;
|
||||||
StartupWMClass=PVE
|
StartupWMClass=PVE
|
||||||
@@ -69,11 +65,22 @@ home.file = {
|
|||||||
[Desktop Entry]
|
[Desktop Entry]
|
||||||
Type=Application
|
Type=Application
|
||||||
Name=Proxmox Backup Server (Firefox)
|
Name=Proxmox Backup Server (Firefox)
|
||||||
Exec=firefox --new-window https://192.168.2.108:8007 --profile PbsWebApp --window-size=1920,1080 --class PbsWebApp
|
Exec=firefox --new-window https://${vars.pbsIp}:${toString vars.ports.pbsWeb} --profile PbsWebApp --window-size=1920,1080 --class PbsWebApp
|
||||||
Icon=/home/nixos/.local/share/icons/proxmox.png
|
Icon=${config.home.homeDirectory}/.local/share/icons/proxmox.png
|
||||||
Terminal=false
|
Terminal=false
|
||||||
Categories=backup;
|
Categories=backup;
|
||||||
StartupWMClass=PBS
|
StartupWMClass=PBS
|
||||||
'';
|
'';
|
||||||
};
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
programs.home-manager.enable = true; # mandatory to activate HM
|
||||||
|
|
||||||
|
# Optional: enable bash (or zsh, fish...)
|
||||||
|
programs.bash.enable = true;
|
||||||
|
services.nextcloud-client = {
|
||||||
|
enable = true;
|
||||||
|
# Optionally start in background directly
|
||||||
|
startInBackground = true;
|
||||||
|
};
|
||||||
}
|
}
|
||||||
+10
-1
@@ -1,8 +1,17 @@
|
|||||||
{ ... }:
|
_:
|
||||||
|
|
||||||
{
|
{
|
||||||
|
imports = [
|
||||||
|
../../modules/networking/wifi.nix
|
||||||
|
];
|
||||||
|
|
||||||
networking.hostName = "nixos";
|
networking.hostName = "nixos";
|
||||||
|
|
||||||
|
# Only needed now that baremetal-gui exists (ZFS root) -- harmless on the
|
||||||
|
# ext4-rooted linode/proxmox/lxc-gui variants, so set unconditionally
|
||||||
|
# rather than only on the baremetal platform.
|
||||||
|
networking.hostId = "de6a9ffc";
|
||||||
|
|
||||||
# Preserved from the pre-refactor `nixos` target — stateVersion must never
|
# Preserved from the pre-refactor `nixos` target — stateVersion must never
|
||||||
# be bumped on an already-installed machine.
|
# be bumped on an already-installed machine.
|
||||||
system.stateVersion = "25.05";
|
system.stateVersion = "25.05";
|
||||||
|
|||||||
+12
-3
@@ -1,8 +1,17 @@
|
|||||||
{ ... }:
|
{ vars, ... }:
|
||||||
|
|
||||||
{
|
{
|
||||||
networking.hostName = "pxe-boot";
|
networking = {
|
||||||
|
hostName = "pxe-boot";
|
||||||
|
useDHCP = false;
|
||||||
|
interfaces.${vars.lxcLanInterface}.ipv4.addresses = [{
|
||||||
|
address = vars.pxeServerIp;
|
||||||
|
prefixLength = vars.lanPrefixLength;
|
||||||
|
}];
|
||||||
|
defaultGateway = { address = vars.lanGateway; interface = vars.lxcLanInterface; };
|
||||||
|
nameservers = [ vars.domainControllerIp ];
|
||||||
|
};
|
||||||
|
services.beszel.agent.environment = { };
|
||||||
# Preserved from the pre-refactor `pxe-boot` target — stateVersion must
|
# Preserved from the pre-refactor `pxe-boot` target — stateVersion must
|
||||||
# never be bumped on an already-installed machine.
|
# never be bumped on an already-installed machine.
|
||||||
system.stateVersion = "25.05";
|
system.stateVersion = "25.05";
|
||||||
|
|||||||
@@ -1,24 +0,0 @@
|
|||||||
{ config, ... }:
|
|
||||||
|
|
||||||
{
|
|
||||||
networking.hostName = "server";
|
|
||||||
networking.hostId = "6689f93e";
|
|
||||||
|
|
||||||
sops.secrets."beszel-token".sopsFile = ../../secrets/server.yaml;
|
|
||||||
sops.templates."server-beszel.env".content = ''
|
|
||||||
TOKEN=${config.sops.placeholder."beszel-token"}
|
|
||||||
'';
|
|
||||||
|
|
||||||
services.beszel.agent.environment = {
|
|
||||||
#DOCKER_HOST = "tcp://docker-socket-proxy:2375";
|
|
||||||
#HUB_URL = "http://docker.sweet.home:8090";
|
|
||||||
KEY = "ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIFPR9kwtC4TAeTRu46A7+opZsYpxqkRJ+x/ZyB2GWCeG";
|
|
||||||
EXTRA_FILESYSTEMS = "/tank/docker/volumes";
|
|
||||||
LOG_LEVEL = "debug";
|
|
||||||
};
|
|
||||||
services.beszel.agent.environmentFile = config.sops.templates."server-beszel.env".path;
|
|
||||||
|
|
||||||
# Preserved from the pre-refactor `server` target — stateVersion must never
|
|
||||||
# be bumped on an already-installed machine.
|
|
||||||
system.stateVersion = "25.05";
|
|
||||||
}
|
|
||||||
@@ -0,0 +1,19 @@
|
|||||||
|
{ vars, ... }:
|
||||||
|
|
||||||
|
{
|
||||||
|
networking = {
|
||||||
|
hostName = "tailscale-router";
|
||||||
|
useDHCP = false;
|
||||||
|
interfaces.${vars.lxcLanInterface}.ipv4.addresses = [{
|
||||||
|
address = vars.tailscaleRouterIp;
|
||||||
|
prefixLength = vars.lanPrefixLength;
|
||||||
|
}];
|
||||||
|
defaultGateway = { address = vars.lanGateway; interface = vars.lxcLanInterface; };
|
||||||
|
nameservers = [ vars.domainControllerIp ];
|
||||||
|
};
|
||||||
|
|
||||||
|
# No networking.hostId: only ZFS-touching hosts (server, docker) need one
|
||||||
|
# for pool-import safety, and this host does neither.
|
||||||
|
|
||||||
|
system.stateVersion = "26.05";
|
||||||
|
}
|
||||||
@@ -0,0 +1,21 @@
|
|||||||
|
{ vars, ... }:
|
||||||
|
|
||||||
|
{
|
||||||
|
networking = {
|
||||||
|
hostName = "tor-relay";
|
||||||
|
useDHCP = false;
|
||||||
|
interfaces.${vars.lxcLanInterface}.ipv4.addresses = [{
|
||||||
|
address = vars.torRelayIp;
|
||||||
|
prefixLength = vars.lanPrefixLength;
|
||||||
|
}];
|
||||||
|
defaultGateway = { address = vars.lanGateway; interface = vars.lxcLanInterface; };
|
||||||
|
nameservers = [ vars.domainControllerIp ];
|
||||||
|
};
|
||||||
|
|
||||||
|
# No networking.hostId: only ZFS-touching hosts need one for pool-import
|
||||||
|
# safety, and this host does neither.
|
||||||
|
|
||||||
|
# A genuinely new host (not a pre-refactor carry-over), so it tracks the
|
||||||
|
# flake's current nixpkgs release rather than being pinned to an older one.
|
||||||
|
system.stateVersion = "26.05";
|
||||||
|
}
|
||||||
@@ -1,9 +1,31 @@
|
|||||||
{ ... }:
|
{ config, vars, ... }:
|
||||||
|
|
||||||
{
|
{
|
||||||
services.beszel.agent.enable = true;
|
# Universal token shared by all beszel agents. Add to secrets/common.yaml:
|
||||||
services.beszel.agent.environment = {
|
# sops secrets/common.yaml
|
||||||
|
# beszel-token: <value from the beszel hub UI>
|
||||||
|
sops.secrets."beszel-token" = { };
|
||||||
|
|
||||||
|
sops.templates."beszel.env".content = ''
|
||||||
|
TOKEN=${config.sops.placeholder."beszel-token"}
|
||||||
|
'';
|
||||||
|
|
||||||
|
services.beszel.agent = {
|
||||||
|
enable = true;
|
||||||
|
environmentFile = config.sops.templates."beszel.env".path;
|
||||||
|
environment = {
|
||||||
#DOCKER_HOST = "tcp://docker-socket-proxy:2375";
|
#DOCKER_HOST = "tcp://docker-socket-proxy:2375";
|
||||||
HUB_URL = "http://docker.sweet.home:8090";
|
HUB_URL = "http://${vars.dockerHost}.${vars.homeDomain}:${toString vars.ports.beszelHub}";
|
||||||
|
KEY = vars.beszelHubKey;
|
||||||
};
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
# The upstream module runs beszel-agent under DynamicUser with
|
||||||
|
# ProtectSystem = "strict" and no StateDirectory, so /var/lib/beszel-agent
|
||||||
|
# (where the agent persists its hub-pairing fingerprint, per
|
||||||
|
# https://github.com/henrygd/beszel/discussions/1542) isn't writable --
|
||||||
|
# every restart silently fails to save it and regenerates a fresh one in
|
||||||
|
# memory, permanently desyncing from whatever the hub has on record after
|
||||||
|
# the very first successful pairing. Give it real persistent storage.
|
||||||
|
systemd.services.beszel-agent.serviceConfig.StateDirectory = "beszel-agent";
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,4 +1,4 @@
|
|||||||
{ ... }:
|
_:
|
||||||
|
|
||||||
{
|
{
|
||||||
boot.loader.systemd-boot.enable = true;
|
boot.loader.systemd-boot.enable = true;
|
||||||
|
|||||||
@@ -1,4 +1,4 @@
|
|||||||
{ pkgs, ... }:
|
{ pkgs, vars, ... }:
|
||||||
|
|
||||||
{
|
{
|
||||||
# Pins the Docker Engine version, carried forward from the pre-refactor
|
# Pins the Docker Engine version, carried forward from the pre-refactor
|
||||||
@@ -13,11 +13,10 @@
|
|||||||
imports = [
|
imports = [
|
||||||
../docker/mount-data.nix
|
../docker/mount-data.nix
|
||||||
../docker/enable-service.nix
|
../docker/enable-service.nix
|
||||||
../tailscale/enable-service.nix
|
../docker/nextcloud-cron-job.nix
|
||||||
../rotate-traefik-logs.nix
|
../docker/docker-health-to-gotify.nix
|
||||||
|
../traefik/rotate-logs.nix
|
||||||
../raspi/mount-data.nix
|
../raspi/mount-data.nix
|
||||||
../services/nextcloud-cron-job.nix
|
|
||||||
../services/docker-health-to-gotify.nix
|
|
||||||
../services/enable-rpcbind.nix
|
../services/enable-rpcbind.nix
|
||||||
];
|
];
|
||||||
|
|
||||||
@@ -28,13 +27,18 @@
|
|||||||
boot.supportedFilesystems = [ "nfs" ];
|
boot.supportedFilesystems = [ "nfs" ];
|
||||||
|
|
||||||
systemd.tmpfiles.rules = [
|
systemd.tmpfiles.rules = [
|
||||||
"L+ /home/nixos/docker - - - - /mnt/docker/config"
|
"L+ /home/${vars.primaryUser}/docker - - - - ${vars.nfsShares.dockerConfig.mountpoint}"
|
||||||
"d /mnt/docker 0755 nixos users -"
|
"d /mnt/docker 0755 ${vars.primaryUser} users -"
|
||||||
"d /mnt/raspi-backup 0755 nixos users -"
|
"d ${vars.nfsShares.raspiVolumes.mountpoint} 0755 ${vars.primaryUser} users -"
|
||||||
];
|
];
|
||||||
|
|
||||||
users.users.nixos.extraGroups = [ "docker" ];
|
users.users.${vars.primaryUser}.extraGroups = [ "docker" ];
|
||||||
services.openssh.settings.PermitRootLogin = "yes";
|
services.openssh.settings.PermitRootLogin = "yes";
|
||||||
|
|
||||||
networking.firewall.allowedTCPPorts = [ 80 8080 443 8090 ];
|
networking.firewall.allowedTCPPorts = [
|
||||||
|
vars.ports.dockerHttp
|
||||||
|
vars.ports.dockerExtra
|
||||||
|
vars.ports.dockerHttps
|
||||||
|
vars.ports.beszelHub
|
||||||
|
];
|
||||||
}
|
}
|
||||||
|
|||||||
+112
-19
@@ -1,8 +1,19 @@
|
|||||||
{ config, pkgs, lib, inputs, ... }:
|
{ config, pkgs, lib, inputs, vars, ... }:
|
||||||
|
|
||||||
{
|
{
|
||||||
|
imports = [
|
||||||
|
../docker/enable-service.nix
|
||||||
|
];
|
||||||
|
|
||||||
|
nixpkgs.overlays = [
|
||||||
|
(final: prev: {
|
||||||
|
docker = prev.docker_29;
|
||||||
|
docker_cli = prev.docker_29;
|
||||||
|
})
|
||||||
|
];
|
||||||
|
|
||||||
environment.systemPackages = with pkgs; [
|
environment.systemPackages = with pkgs; [
|
||||||
inputs.nixos-conf-editor.packages.${pkgs.system}.nixos-conf-editor
|
inputs.nixos-conf-editor.packages.${pkgs.stdenv.hostPlatform.system}.nixos-conf-editor
|
||||||
nodejs
|
nodejs
|
||||||
appimage-run
|
appimage-run
|
||||||
seahorse
|
seahorse
|
||||||
@@ -18,40 +29,122 @@
|
|||||||
];
|
];
|
||||||
|
|
||||||
boot.loader.grub.useOSProber = true;
|
boot.loader.grub.useOSProber = true;
|
||||||
|
programs.direnv.enable = true;
|
||||||
|
services = {
|
||||||
|
xserver = {
|
||||||
|
enable = true;
|
||||||
|
|
||||||
services.xserver.enable = true;
|
displayManager = {
|
||||||
services.xserver.displayManager.lightdm.enable = true;
|
lightdm.enable = true;
|
||||||
services.xserver.desktopManager.cinnamon.enable = true;
|
sessionCommands = ''
|
||||||
|
eval $(gnome-keyring-daemon --start --components=secrets,ssh)
|
||||||
|
export SSH_AUTH_SOCK
|
||||||
|
'';
|
||||||
|
};
|
||||||
|
|
||||||
services.xserver.xkb = {
|
desktopManager.cinnamon.enable = true;
|
||||||
|
|
||||||
|
xkb = {
|
||||||
layout = "au";
|
layout = "au";
|
||||||
variant = "";
|
variant = "";
|
||||||
};
|
};
|
||||||
|
};
|
||||||
|
|
||||||
services.printing.enable = true;
|
printing.enable = true;
|
||||||
|
|
||||||
security.rtkit.enable = true;
|
pipewire = {
|
||||||
services.pipewire = {
|
|
||||||
enable = true;
|
enable = true;
|
||||||
alsa.enable = true;
|
alsa.enable = true;
|
||||||
alsa.support32Bit = true;
|
alsa.support32Bit = true;
|
||||||
pulse.enable = true;
|
pulse.enable = true;
|
||||||
};
|
};
|
||||||
|
|
||||||
users.users.nixos.extraGroups = [ "networkmanager" ];
|
xrdp = {
|
||||||
|
enable = true;
|
||||||
|
defaultWindowManager = "cinnamon-session";
|
||||||
|
openFirewall = true;
|
||||||
|
};
|
||||||
|
|
||||||
|
gnome.gnome-keyring.enable = true;
|
||||||
|
};
|
||||||
|
|
||||||
|
security = {
|
||||||
|
rtkit.enable = true;
|
||||||
|
pam.services.login.enableGnomeKeyring = true;
|
||||||
|
};
|
||||||
|
|
||||||
|
# The networkmanager group only exists when NM is actually enabled — the
|
||||||
|
# lxc platform module force-disables it, so don't add the user to a group
|
||||||
|
# that won't exist there.
|
||||||
|
users.users.${vars.primaryUser}.extraGroups = lib.mkIf config.networking.networkmanager.enable [ "networkmanager" ];
|
||||||
|
|
||||||
programs.firefox.enable = true;
|
programs.firefox.enable = true;
|
||||||
|
|
||||||
services.xrdp.enable = true;
|
|
||||||
services.xrdp.defaultWindowManager = "cinnamon-session";
|
|
||||||
services.xrdp.openFirewall = true;
|
|
||||||
nixpkgs.config.allowUnfree = true;
|
nixpkgs.config.allowUnfree = true;
|
||||||
|
|
||||||
services.gnome.gnome-keyring.enable = true;
|
# GUI-specific Home Manager additions for the IPA primary user, extending
|
||||||
security.pam.services.login.enableGnomeKeyring = true;
|
# the baseline in modules/ipa/client.nix with desktop apps and services
|
||||||
|
# that only make sense on a graphical workstation.
|
||||||
services.xserver.displayManager.sessionCommands = ''
|
home-manager.users.${vars.ipaUser} = { pkgs, ... }: {
|
||||||
eval $(gnome-keyring-daemon --start --components=secrets,ssh)
|
home = {
|
||||||
export SSH_AUTH_SOCK
|
packages = with pkgs; [
|
||||||
|
git
|
||||||
|
vim
|
||||||
|
nextcloud-client
|
||||||
|
chromium
|
||||||
|
claude-code
|
||||||
|
fish
|
||||||
|
sops
|
||||||
|
];
|
||||||
|
sessionVariables = {
|
||||||
|
EDITOR = "nano";
|
||||||
|
SOPS_AGE_KEY_FILE = "/home/${vars.ipaUser}/.config/sops/age/keys.txt";
|
||||||
|
};
|
||||||
|
file = {
|
||||||
|
".local/share/applications/proxmox-chromium-app.desktop".text = ''
|
||||||
|
[Desktop Entry]
|
||||||
|
Type=Application
|
||||||
|
Name=Proxmox (Chromium)
|
||||||
|
Exec=chromium --app=https://pve.${vars.homeDomain}:${toString vars.ports.pveWeb} --window-size=1920,1080 --window-position=0,0
|
||||||
|
Icon=/home/${vars.ipaUser}/.local/share/icons/proxmox.png
|
||||||
|
Terminal=false
|
||||||
|
Categories=Hypervisor;
|
||||||
|
StartupWMClass=PVE
|
||||||
'';
|
'';
|
||||||
|
".local/share/applications/pbs-chromium-app.desktop".text = ''
|
||||||
|
[Desktop Entry]
|
||||||
|
Type=Application
|
||||||
|
Name=Proxmox Backup Server (Chromium)
|
||||||
|
Exec=chromium --app=https://${vars.pbsIp}:${toString vars.ports.pbsWeb} --window-size=1920,1080 --window-position=0,0
|
||||||
|
Icon=/home/${vars.ipaUser}/.local/share/icons/proxmox.png
|
||||||
|
Terminal=false
|
||||||
|
Categories=backup;
|
||||||
|
'';
|
||||||
|
".local/share/applications/proxmox-firefox-app.desktop".text = ''
|
||||||
|
[Desktop Entry]
|
||||||
|
Type=Application
|
||||||
|
Name=Proxmox (Firefox)
|
||||||
|
Exec=firefox --new-instance https://pve.${vars.homeDomain}:${toString vars.ports.pveWeb} --profile ProxmoxWebApp --window-size=1920,1080 --class ProxmoxWebApp
|
||||||
|
Icon=/home/${vars.ipaUser}/.local/share/icons/proxmox.png
|
||||||
|
Terminal=false
|
||||||
|
Categories=Hypervisor;
|
||||||
|
StartupWMClass=PVE
|
||||||
|
'';
|
||||||
|
".local/share/applications/pbs-firefox-app.desktop".text = ''
|
||||||
|
[Desktop Entry]
|
||||||
|
Type=Application
|
||||||
|
Name=Proxmox Backup Server (Firefox)
|
||||||
|
Exec=firefox --new-window https://${vars.pbsIp}:${toString vars.ports.pbsWeb} --profile PbsWebApp --window-size=1920,1080 --class PbsWebApp
|
||||||
|
Icon=/home/${vars.ipaUser}/.local/share/icons/proxmox.png
|
||||||
|
Terminal=false
|
||||||
|
Categories=backup;
|
||||||
|
StartupWMClass=PBS
|
||||||
|
'';
|
||||||
|
};
|
||||||
|
};
|
||||||
|
services.nextcloud-client = {
|
||||||
|
enable = true;
|
||||||
|
startInBackground = true;
|
||||||
|
};
|
||||||
|
};
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,84 @@
|
|||||||
|
# Docker Swarm node build type.
|
||||||
|
#
|
||||||
|
# Produces NixOS hosts that form a Docker Swarm manager cluster. Two nodes
|
||||||
|
# (ha-docker-1, ha-docker-2) are both managers so either can accept Docker
|
||||||
|
# API and `docker stack` commands.
|
||||||
|
#
|
||||||
|
# Key differences from the existing `docker` build type (used by CT 105):
|
||||||
|
# - nextcloud-cron-job.nix is EXCLUDED — `docker exec` breaks in swarm
|
||||||
|
# because the target container may be on the other node. The cron job
|
||||||
|
# is replaced by a nextcloud-cron sidecar in the Nextcloud stack.
|
||||||
|
# See docs/internal/docker-swarm-cutover.md.
|
||||||
|
# - traefik/rotate-logs.nix is EXCLUDED — log rotation moves to Docker's
|
||||||
|
# json-file log driver (max-size/max-file on the Traefik service
|
||||||
|
# definition). See docs/internal/docker-swarm-cutover.md.
|
||||||
|
# - raspi/mount-data.nix is EXCLUDED — specific to CT 105's backup role.
|
||||||
|
# - Swarm firewall ports (2377/tcp, 7946/tcp+udp, 4789/udp) are opened
|
||||||
|
# on the swarm NIC (ens20/vmbr3) only.
|
||||||
|
# - checkReversePath = "loose" is required for the Swarm ingress routing
|
||||||
|
# mesh: VXLAN return traffic is asymmetric (arrives ens20, exits ens18).
|
||||||
|
# - beszel-agent is enabled for host-level monitoring.
|
||||||
|
{ pkgs, vars, ... }:
|
||||||
|
|
||||||
|
{
|
||||||
|
# Pin Docker Engine to version 29, matching CT 105, so image layers cached
|
||||||
|
# on NFS volumes remain compatible across old and new hosts.
|
||||||
|
nixpkgs.overlays = [
|
||||||
|
(final: prev: {
|
||||||
|
docker = prev.docker_29;
|
||||||
|
docker_cli = prev.docker_29;
|
||||||
|
})
|
||||||
|
];
|
||||||
|
|
||||||
|
imports = [
|
||||||
|
../docker/enable-service.nix
|
||||||
|
../docker/mount-data.nix
|
||||||
|
../docker/docker-health-to-gotify.nix
|
||||||
|
../beszel/enable-agent.nix
|
||||||
|
../services/enable-rpcbind.nix
|
||||||
|
];
|
||||||
|
|
||||||
|
environment.systemPackages = with pkgs; [
|
||||||
|
nfs-utils
|
||||||
|
];
|
||||||
|
|
||||||
|
boot.supportedFilesystems = [ "nfs" ];
|
||||||
|
|
||||||
|
systemd.tmpfiles.rules = [
|
||||||
|
# Symlink ~/docker → NFS config mount so the docker-health-to-gotify
|
||||||
|
# script (and operator convenience) resolves ~/docker/... correctly.
|
||||||
|
"L+ /home/${vars.primaryUser}/docker - - - - ${vars.nfsShares.dockerConfig.mountpoint}"
|
||||||
|
"d /mnt/docker 0755 ${vars.primaryUser} users -"
|
||||||
|
];
|
||||||
|
|
||||||
|
users.users.${vars.primaryUser}.extraGroups = [ "docker" ];
|
||||||
|
|
||||||
|
networking.firewall = {
|
||||||
|
# LAN-facing service ports — same as the existing docker build type.
|
||||||
|
allowedTCPPorts = [
|
||||||
|
vars.ports.dockerHttp
|
||||||
|
vars.ports.dockerHttps
|
||||||
|
vars.ports.dockerExtra
|
||||||
|
vars.ports.beszelHub
|
||||||
|
];
|
||||||
|
|
||||||
|
# Swarm inter-node ports restricted to the swarm NIC (ens20/vmbr3).
|
||||||
|
# vmbr3 is an isolated internal bridge — no LAN reachability.
|
||||||
|
interfaces.${vars.haDockerSwarmInterface} = {
|
||||||
|
allowedTCPPorts = [
|
||||||
|
vars.ports.dockerSwarmMgmt # 2377 — Raft + cluster management
|
||||||
|
vars.ports.dockerSwarmDisc # 7946 — Serf gossip (TCP half)
|
||||||
|
];
|
||||||
|
allowedUDPPorts = [
|
||||||
|
vars.ports.dockerSwarmDisc # 7946 — Serf gossip (UDP half)
|
||||||
|
vars.ports.dockerSwarmVxlan # 4789 — VXLAN overlay data path
|
||||||
|
];
|
||||||
|
};
|
||||||
|
|
||||||
|
# Docker Swarm ingress routing mesh creates asymmetric routes: a request
|
||||||
|
# arrives on ens18 (LAN) for a container that lives on ens20's VXLAN
|
||||||
|
# overlay; the return path differs from the incoming interface. Strict
|
||||||
|
# rp_filter drops these packets. "loose" allows them.
|
||||||
|
checkReversePath = "loose";
|
||||||
|
};
|
||||||
|
}
|
||||||
@@ -0,0 +1,54 @@
|
|||||||
|
# HA file server build type: DRBD + XFS + LIO iSCSI + NFS, managed by
|
||||||
|
# Corosync + Pacemaker. Both ha-server-1 and ha-server-2 use this type.
|
||||||
|
#
|
||||||
|
# NFS start/stop:
|
||||||
|
# services.nfs.server.enable = true configures /etc/exports, wires up
|
||||||
|
# rpcbind, and loads kernel modules — but nfs-server.service.wantedBy is
|
||||||
|
# force-cleared so systemd does NOT auto-start it at boot. Pacemaker's
|
||||||
|
# ha-group resource group (configured by scripts/ha/cluster-init.sh)
|
||||||
|
# starts and stops nfs-server as part of the failover sequence after the
|
||||||
|
# XFS mount and iSCSI target are brought up on the new Active node.
|
||||||
|
#
|
||||||
|
# Beszel agent:
|
||||||
|
# Enabled here via enable-agent.nix. The agent KEY (used to pair with
|
||||||
|
# the Beszel hub) is not set yet — add it to hosts/ha-server-{1,2}/host.nix
|
||||||
|
# under services.beszel.agent.environment.KEY once the hub accepts the
|
||||||
|
# new agents, following the pattern in hosts/server/host.nix.
|
||||||
|
{ lib, pkgs, vars, ... }:
|
||||||
|
|
||||||
|
let
|
||||||
|
# Generates /etc/exports lines for all nfsShares data entries.
|
||||||
|
# LAN (VLAN 2): NFS via vip-lan (192.168.2.229) for pxe-boot and other LAN clients.
|
||||||
|
# Storage-client (VLAN 20): NFS via vip-storage (192.168.20.229) for docker and
|
||||||
|
# future swarm nodes; firewall restricts these ports to haClientCidr only.
|
||||||
|
mkNfsExports = storageRoot:
|
||||||
|
lib.concatMapStrings
|
||||||
|
(share:
|
||||||
|
" ${storageRoot}/${share.subpath} ${vars.lanCidr}${vars.nfsShares.options}\n" +
|
||||||
|
" ${storageRoot}/${share.subpath} ${vars.haClientCidr}${vars.nfsShares.options}\n")
|
||||||
|
(lib.filter builtins.isAttrs (lib.attrValues vars.nfsShares));
|
||||||
|
in
|
||||||
|
{
|
||||||
|
imports = [
|
||||||
|
../ha/pacemaker-stack.nix
|
||||||
|
../ha/iscsi-target.nix
|
||||||
|
../ha/cluster-config.nix
|
||||||
|
../beszel/enable-agent.nix
|
||||||
|
];
|
||||||
|
|
||||||
|
# xfsprogs: mkfs.xfs/xfs_info needed by cluster-init.sh.
|
||||||
|
# openiscsi: iscsiadm needed by acceptance-tests.sh T4 (iSCSI discovery check).
|
||||||
|
environment.systemPackages = [ pkgs.xfsprogs pkgs.openiscsi ];
|
||||||
|
|
||||||
|
services.nfs.server = {
|
||||||
|
enable = true;
|
||||||
|
exports = mkNfsExports vars.haStorageRoot;
|
||||||
|
};
|
||||||
|
|
||||||
|
# Pacemaker controls nfs-server — prevent systemd from starting it at boot
|
||||||
|
# on both nodes (only the Active node should be serving NFS).
|
||||||
|
systemd.services.nfs-server.wantedBy = lib.mkForce [ ];
|
||||||
|
|
||||||
|
# Same reason as server.nix: exports use standard auth, not Kerberos.
|
||||||
|
systemd.services.rpc-svcgssd.enable = false;
|
||||||
|
}
|
||||||
@@ -1,9 +1,12 @@
|
|||||||
{ pkgs, ... }:
|
{ lib, pkgs, config, vars, ... }:
|
||||||
|
|
||||||
{
|
{
|
||||||
networking.networkmanager.enable = true;
|
networking.networkmanager.enable = true;
|
||||||
|
|
||||||
users.users.nixos.extraGroups = [ "networkmanager" ];
|
# The networkmanager group only exists when NM is actually enabled — the
|
||||||
|
# lxc platform module force-disables it, so don't add the user to a group
|
||||||
|
# that won't exist there.
|
||||||
|
users.users.${vars.primaryUser}.extraGroups = lib.mkIf config.networking.networkmanager.enable [ "networkmanager" ];
|
||||||
|
|
||||||
environment.systemPackages = with pkgs; [
|
environment.systemPackages = with pkgs; [
|
||||||
inetutils
|
inetutils
|
||||||
|
|||||||
@@ -1,10 +1,14 @@
|
|||||||
{ config, lib, pkgs, inputs, ... }:
|
{ config, lib, pkgs, inputs, vars, ... }:
|
||||||
|
|
||||||
let
|
let
|
||||||
pxeRoot = "/srv/pxe";
|
pxeRoot = "/srv/pxe";
|
||||||
httpRoot = "${pxeRoot}/http";
|
httpRoot = "${pxeRoot}/http";
|
||||||
tftpRoot = "${pxeRoot}/tftp";
|
tftpRoot = "${pxeRoot}/tftp";
|
||||||
pxeBaseUrl = "http://192.168.2.247";
|
pxeBaseUrl = "http://${vars.pxeServerIp}";
|
||||||
|
|
||||||
|
# Base network address extracted from lanCidr (e.g. "192.168.2.0" from
|
||||||
|
# "192.168.2.0/24") — used by dnsmasq's proxy DHCP range directive.
|
||||||
|
lanBaseAddr = lib.head (lib.splitString "/" vars.lanCidr);
|
||||||
|
|
||||||
bootIpxe = pkgs.writeText "boot.ipxe" ''
|
bootIpxe = pkgs.writeText "boot.ipxe" ''
|
||||||
#!ipxe
|
#!ipxe
|
||||||
@@ -21,6 +25,216 @@ let
|
|||||||
chain ${pxeBaseUrl}/boot.ipxe
|
chain ${pxeBaseUrl}/boot.ipxe
|
||||||
'';
|
'';
|
||||||
|
|
||||||
|
debianRelease = "bookworm";
|
||||||
|
debianMirror = "https://deb.debian.org/debian";
|
||||||
|
debianNetbootBase = "${debianMirror}/dists/${debianRelease}/main/installer-amd64/current/images/netboot/debian-installer/amd64";
|
||||||
|
|
||||||
|
rockyRelease = "9";
|
||||||
|
rockyArch = "x86_64";
|
||||||
|
rockyMirror = "https://dl.rockylinux.org/pub/rocky/${rockyRelease}";
|
||||||
|
rockyPxebootBase = "${rockyMirror}/BaseOS/${rockyArch}/os/images/pxeboot";
|
||||||
|
|
||||||
|
debianIpxe = pkgs.writeText "debian.ipxe" ''
|
||||||
|
#!ipxe
|
||||||
|
|
||||||
|
set base ${pxeBaseUrl}
|
||||||
|
|
||||||
|
kernel ''${base}/debian/linux
|
||||||
|
initrd ''${base}/debian/initrd.gz
|
||||||
|
boot
|
||||||
|
'';
|
||||||
|
|
||||||
|
fetchDebianNetboot = pkgs.writeShellScript "fetch-debian-netboot" ''
|
||||||
|
set -eu
|
||||||
|
|
||||||
|
dir="${httpRoot}/debian"
|
||||||
|
mirror="${debianNetbootBase}"
|
||||||
|
|
||||||
|
if [ -f "$dir/linux" ] && [ -f "$dir/initrd.gz" ]; then
|
||||||
|
echo "Debian ${debianRelease} netboot files already present; skipping download."
|
||||||
|
exit 0
|
||||||
|
fi
|
||||||
|
|
||||||
|
echo "Downloading Debian ${debianRelease} netboot kernel and initrd from $mirror ..."
|
||||||
|
${pkgs.curl}/bin/curl -fsSL -o "$dir/linux.tmp" "$mirror/linux"
|
||||||
|
${pkgs.curl}/bin/curl -fsSL -o "$dir/initrd.gz.tmp" "$mirror/initrd.gz"
|
||||||
|
mv "$dir/linux.tmp" "$dir/linux"
|
||||||
|
mv "$dir/initrd.gz.tmp" "$dir/initrd.gz"
|
||||||
|
echo "Debian ${debianRelease} netboot files staged."
|
||||||
|
'';
|
||||||
|
|
||||||
|
# Rocky Linux 9 iPXE script — boots vmlinuz+initrd.img from the staged
|
||||||
|
# /rocky/ directory and hands Anaconda the hosted Kickstart URL.
|
||||||
|
# net.ifnames=0 biosdevname=0 ensures the NIC is eth0 in both the
|
||||||
|
# installer and the installed system (matches the Kickstart NM config).
|
||||||
|
rockyFreeIpaIpxe = pkgs.writeText "rocky-freeipa.ipxe" ''
|
||||||
|
#!ipxe
|
||||||
|
|
||||||
|
set base ${pxeBaseUrl}
|
||||||
|
|
||||||
|
kernel ''${base}/rocky/vmlinuz inst.ks=''${base}/rocky-freeipa.ks inst.repo=${rockyMirror}/BaseOS/${rockyArch}/os/ net.ifnames=0 biosdevname=0 ip=dhcp quiet
|
||||||
|
initrd ''${base}/rocky/initrd.img
|
||||||
|
boot
|
||||||
|
'';
|
||||||
|
|
||||||
|
# Kickstart file for ${vars.ipaServer}.
|
||||||
|
# Installs Rocky Linux 9, sets a static IP, creates ${vars.ipaUser} with
|
||||||
|
# the admin SSH key, then on first reboot runs ipa-server-install via a
|
||||||
|
# systemd oneshot service. Passwords are generated at %post time, written
|
||||||
|
# to /root/ipa-credentials.txt (chmod 600), and read back by the
|
||||||
|
# first-boot script — never hardcoded here or in the repo.
|
||||||
|
rockyFreeIpaKs = pkgs.writeText "rocky-freeipa.ks" ''
|
||||||
|
#version=RHEL9
|
||||||
|
# Unattended Rocky Linux 9 + FreeIPA install
|
||||||
|
# Target: ${vars.ipaServer} ${vars.domainControllerIp}
|
||||||
|
|
||||||
|
url --url=${rockyMirror}/BaseOS/${rockyArch}/os/
|
||||||
|
repo --name=appstream --baseurl=${rockyMirror}/AppStream/${rockyArch}/os/
|
||||||
|
|
||||||
|
lang en_US.UTF-8
|
||||||
|
keyboard us
|
||||||
|
timezone UTC --utc
|
||||||
|
|
||||||
|
# DHCP during install; static IP configured in %post via NM config file
|
||||||
|
network --bootproto=dhcp --device=link --activate
|
||||||
|
network --hostname=${vars.ipaServer}
|
||||||
|
|
||||||
|
selinux --enforcing
|
||||||
|
firewall --enabled --service=ssh
|
||||||
|
|
||||||
|
rootpw --lock
|
||||||
|
user --name=${vars.ipaUser} --groups=wheel --shell=/bin/bash
|
||||||
|
sshkey --username=${vars.ipaUser} "${vars.adminSshKey}"
|
||||||
|
|
||||||
|
zerombr
|
||||||
|
clearpart --all --initlabel --drives=sda
|
||||||
|
# Keep net.ifnames=0 biosdevname=0 in the installed GRUB so the NIC
|
||||||
|
# stays eth0 after reboot (matches the NM connection file below).
|
||||||
|
bootloader --location=mbr --boot-drive=sda --append="net.ifnames=0 biosdevname=0"
|
||||||
|
|
||||||
|
part /boot --fstype=xfs --size=1024 --ondisk=sda
|
||||||
|
part swap --fstype=swap --size=2048 --ondisk=sda
|
||||||
|
part / --fstype=xfs --grow --size=1 --ondisk=sda --asprimary
|
||||||
|
|
||||||
|
%packages
|
||||||
|
@^minimal-environment
|
||||||
|
ipa-server
|
||||||
|
ipa-server-dns
|
||||||
|
%end
|
||||||
|
|
||||||
|
reboot
|
||||||
|
|
||||||
|
%post --log=/root/ks-post.log
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
# -- Static IP: write NM connection file directly (NM not running in chroot) --
|
||||||
|
mkdir -p /etc/NetworkManager/system-connections
|
||||||
|
cat > /etc/NetworkManager/system-connections/eth0.nmconnection << 'NMCONN'
|
||||||
|
[connection]
|
||||||
|
id=eth0
|
||||||
|
type=ethernet
|
||||||
|
interface-name=eth0
|
||||||
|
autoconnect=true
|
||||||
|
|
||||||
|
[ethernet]
|
||||||
|
|
||||||
|
[ipv4]
|
||||||
|
method=manual
|
||||||
|
addresses=${vars.domainControllerIp}/${toString vars.lanPrefixLength}
|
||||||
|
gateway=${vars.lanGateway}
|
||||||
|
dns=${vars.domainControllerIp};
|
||||||
|
dns-search=${vars.homeDomain};
|
||||||
|
|
||||||
|
[ipv6]
|
||||||
|
method=auto
|
||||||
|
NMCONN
|
||||||
|
chmod 600 /etc/NetworkManager/system-connections/eth0.nmconnection
|
||||||
|
|
||||||
|
# -- /etc/hosts: FQDN must resolve to the real IP (not loopback) for IPA --
|
||||||
|
sed -i '/domain-controller/d' /etc/hosts
|
||||||
|
echo '${vars.domainControllerIp} ${vars.ipaServer} domain-controller' >> /etc/hosts
|
||||||
|
|
||||||
|
# -- Generate IPA passwords and store securely --
|
||||||
|
DM_PASS=$(openssl rand -base64 24 | tr -dc 'A-Za-z0-9' | head -c 24)
|
||||||
|
ADMIN_PASS=$(openssl rand -base64 24 | tr -dc 'A-Za-z0-9' | head -c 24)
|
||||||
|
printf 'Directory Manager: %s\nIPA Admin: %s\n' "$DM_PASS" "$ADMIN_PASS" \
|
||||||
|
> /root/ipa-credentials.txt
|
||||||
|
chmod 600 /root/ipa-credentials.txt
|
||||||
|
|
||||||
|
# -- First-boot script: reads passwords back, runs ipa-server-install --
|
||||||
|
cat > /usr/local/sbin/freeipa-first-boot.sh << 'FIRSTBOOT'
|
||||||
|
#!/bin/bash
|
||||||
|
set -euo pipefail
|
||||||
|
exec >> /root/freeipa-install.log 2>&1
|
||||||
|
echo "=== FreeIPA first-boot install started at $(date) ==="
|
||||||
|
|
||||||
|
DM_PASS=$(grep '^Directory Manager:' /root/ipa-credentials.txt | awk '{print $NF}')
|
||||||
|
ADMIN_PASS=$(grep '^IPA Admin:' /root/ipa-credentials.txt | awk '{print $NF}')
|
||||||
|
|
||||||
|
ipa-server-install \
|
||||||
|
--realm=${lib.strings.toUpper vars.homeDomain} \
|
||||||
|
--domain=${vars.homeDomain} \
|
||||||
|
--hostname=${vars.ipaServer} \
|
||||||
|
--ds-password="$DM_PASS" \
|
||||||
|
--admin-password="$ADMIN_PASS" \
|
||||||
|
--setup-dns \
|
||||||
|
--forwarder=${vars.domainControllerIp} \
|
||||||
|
--no-dnssec-validation \
|
||||||
|
--no-ntp \
|
||||||
|
--unattended
|
||||||
|
|
||||||
|
echo "=== FreeIPA install complete at $(date) ==="
|
||||||
|
echo "Credentials: /root/ipa-credentials.txt (save to password manager)"
|
||||||
|
echo "CA backup: /root/cacert.p12 (encrypted with Directory Manager password)"
|
||||||
|
systemctl disable freeipa-first-boot.service
|
||||||
|
FIRSTBOOT
|
||||||
|
chmod 700 /usr/local/sbin/freeipa-first-boot.sh
|
||||||
|
|
||||||
|
# -- Systemd oneshot service: runs freeipa-first-boot.sh on first real boot --
|
||||||
|
cat > /etc/systemd/system/freeipa-first-boot.service << 'UNIT'
|
||||||
|
[Unit]
|
||||||
|
Description=FreeIPA first-boot installation
|
||||||
|
After=network-online.target
|
||||||
|
Wants=network-online.target
|
||||||
|
ConditionPathExists=/root/ipa-credentials.txt
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
Type=oneshot
|
||||||
|
ExecStart=/usr/local/sbin/freeipa-first-boot.sh
|
||||||
|
TimeoutStartSec=1800
|
||||||
|
RemainAfterExit=yes
|
||||||
|
|
||||||
|
[Install]
|
||||||
|
WantedBy=multi-user.target
|
||||||
|
UNIT
|
||||||
|
|
||||||
|
mkdir -p /etc/systemd/system/multi-user.target.wants
|
||||||
|
ln -sf /etc/systemd/system/freeipa-first-boot.service \
|
||||||
|
/etc/systemd/system/multi-user.target.wants/freeipa-first-boot.service
|
||||||
|
|
||||||
|
echo "Kickstart %post complete. FreeIPA installs on first reboot (~20 min)."
|
||||||
|
%end
|
||||||
|
'';
|
||||||
|
|
||||||
|
fetchRockyPxeboot = pkgs.writeShellScript "fetch-rocky-pxeboot" ''
|
||||||
|
set -eu
|
||||||
|
|
||||||
|
dir="${httpRoot}/rocky"
|
||||||
|
base="${rockyPxebootBase}"
|
||||||
|
|
||||||
|
if [ -f "$dir/vmlinuz" ] && [ -f "$dir/initrd.img" ]; then
|
||||||
|
echo "Rocky Linux ${rockyRelease} pxeboot files already present; skipping download."
|
||||||
|
exit 0
|
||||||
|
fi
|
||||||
|
|
||||||
|
echo "Downloading Rocky Linux ${rockyRelease} pxeboot kernel and initrd from $base ..."
|
||||||
|
${pkgs.curl}/bin/curl -fsSL -o "$dir/vmlinuz.tmp" "$base/vmlinuz"
|
||||||
|
${pkgs.curl}/bin/curl -fsSL -o "$dir/initrd.img.tmp" "$base/initrd.img"
|
||||||
|
mv "$dir/vmlinuz.tmp" "$dir/vmlinuz"
|
||||||
|
mv "$dir/initrd.img.tmp" "$dir/initrd.img"
|
||||||
|
echo "Rocky Linux ${rockyRelease} pxeboot files staged."
|
||||||
|
'';
|
||||||
|
|
||||||
systemRescueIpxe = pkgs.writeText "systemrescue.ipxe" ''
|
systemRescueIpxe = pkgs.writeText "systemrescue.ipxe" ''
|
||||||
#!ipxe
|
#!ipxe
|
||||||
|
|
||||||
@@ -68,17 +282,27 @@ let
|
|||||||
set base ${pxeBaseUrl}
|
set base ${pxeBaseUrl}
|
||||||
|
|
||||||
menu PXE Boot Menu
|
menu PXE Boot Menu
|
||||||
item nixos NixOS Installer
|
item auto-installer NixOS Auto-Installer
|
||||||
|
item nixos-minimal NixOS Minimal
|
||||||
|
item debian Debian Minimal
|
||||||
|
item rocky-freeipa FreeIPA Server (Rocky Linux 9)
|
||||||
item rescue Rescue Environment
|
item rescue Rescue Environment
|
||||||
item shell iPXE Shell
|
item shell iPXE Shell
|
||||||
item reboot Reboot
|
item reboot Reboot
|
||||||
|
|
||||||
choose target && goto ''${target}
|
choose target && goto ''${target}
|
||||||
|
|
||||||
:nixos
|
:auto-installer
|
||||||
kernel ''${base}/nixos/bzImage ip=dhcp
|
chain ''${base}/auto-installer/netboot.ipxe
|
||||||
initrd ''${base}/nixos/initrd
|
|
||||||
boot
|
:nixos-minimal
|
||||||
|
chain ''${base}/nixos-minimal/netboot.ipxe
|
||||||
|
|
||||||
|
:debian
|
||||||
|
chain ''${base}/debian.ipxe
|
||||||
|
|
||||||
|
:rocky-freeipa
|
||||||
|
chain ''${base}/rocky-freeipa.ipxe
|
||||||
|
|
||||||
:rescue
|
:rescue
|
||||||
chain ''${base}/systemrescue.ipxe
|
chain ''${base}/systemrescue.ipxe
|
||||||
@@ -91,11 +315,18 @@ let
|
|||||||
'';
|
'';
|
||||||
in
|
in
|
||||||
{
|
{
|
||||||
|
imports = [
|
||||||
|
../pxe-boot/stage-installer-artifacts.nix
|
||||||
|
../pxe-boot/mount-pxe-images.nix
|
||||||
|
../beszel/enable-agent.nix
|
||||||
|
];
|
||||||
|
|
||||||
environment.systemPackages = with pkgs; [
|
environment.systemPackages = with pkgs; [
|
||||||
ipxe
|
ipxe
|
||||||
];
|
];
|
||||||
|
|
||||||
services.nginx = {
|
services = {
|
||||||
|
nginx = {
|
||||||
enable = true;
|
enable = true;
|
||||||
|
|
||||||
virtualHosts."pxe-boot" = {
|
virtualHosts."pxe-boot" = {
|
||||||
@@ -111,32 +342,73 @@ in
|
|||||||
|
|
||||||
# TFTP is only used to deliver the initial iPXE bootloader. After iPXE
|
# TFTP is only used to deliver the initial iPXE bootloader. After iPXE
|
||||||
# starts, all further assets are fetched via nginx over HTTP.
|
# starts, all further assets are fetched via nginx over HTTP.
|
||||||
services.atftpd = {
|
atftpd = {
|
||||||
enable = true;
|
enable = true;
|
||||||
root = tftpRoot;
|
root = tftpRoot;
|
||||||
extraOptions = [
|
extraOptions = [ "--verbose=5" ];
|
||||||
"--verbose=5"
|
|
||||||
];
|
|
||||||
};
|
};
|
||||||
|
|
||||||
systemd.tmpfiles.rules = [
|
openssh.settings.PermitRootLogin = "yes";
|
||||||
|
};
|
||||||
|
|
||||||
|
systemd = {
|
||||||
|
tmpfiles.rules = [
|
||||||
"d ${pxeRoot} 0755 root root -"
|
"d ${pxeRoot} 0755 root root -"
|
||||||
"d ${httpRoot} 0755 root root -"
|
"d ${httpRoot} 0755 root root -"
|
||||||
"d ${httpRoot}/images 0755 root root -"
|
"L+ ${httpRoot}/images - - - - ${vars.nfsShares.pxebootImages.mountpoint}"
|
||||||
"d ${httpRoot}/nixos 0755 root root -"
|
"d ${httpRoot}/auto-installer 0755 root root -"
|
||||||
|
"d ${httpRoot}/nixos-minimal 0755 root root -"
|
||||||
"d ${httpRoot}/systemrescue 0755 root root -"
|
"d ${httpRoot}/systemrescue 0755 root root -"
|
||||||
|
"d ${httpRoot}/debian 0755 root root -"
|
||||||
"d ${httpRoot}/ubuntu 0755 root root -"
|
"d ${httpRoot}/ubuntu 0755 root root -"
|
||||||
"d ${httpRoot}/rescue 0755 root root -"
|
"d ${httpRoot}/rescue 0755 root root -"
|
||||||
|
"d ${httpRoot}/rocky 0755 root root -"
|
||||||
"d ${tftpRoot} 0755 root root -"
|
"d ${tftpRoot} 0755 root root -"
|
||||||
"C+ ${httpRoot}/boot.ipxe 0644 root root - ${bootIpxe}"
|
"C+ ${httpRoot}/boot.ipxe 0644 root root - ${bootIpxe}"
|
||||||
"C+ ${httpRoot}/menu.ipxe 0644 root root - ${menuIpxe}"
|
"C+ ${httpRoot}/menu.ipxe 0644 root root - ${menuIpxe}"
|
||||||
|
"C+ ${httpRoot}/debian.ipxe 0644 root root - ${debianIpxe}"
|
||||||
|
"C+ ${httpRoot}/rocky-freeipa.ipxe 0644 root root - ${rockyFreeIpaIpxe}"
|
||||||
|
"C+ ${httpRoot}/rocky-freeipa.ks 0644 root root - ${rockyFreeIpaKs}"
|
||||||
"C+ ${httpRoot}/systemrescue.ipxe 0644 root root - ${systemRescueIpxe}"
|
"C+ ${httpRoot}/systemrescue.ipxe 0644 root root - ${systemRescueIpxe}"
|
||||||
"C+ ${tftpRoot}/autoexec.ipxe 0644 root root - ${autoexecIpxe}"
|
"C+ ${tftpRoot}/autoexec.ipxe 0644 root root - ${autoexecIpxe}"
|
||||||
"C+ ${tftpRoot}/ipxe.efi 0644 root root - ${pkgs.ipxe}/ipxe.efi"
|
"C+ ${tftpRoot}/ipxe.efi 0644 root root - ${pkgs.ipxe}/ipxe.efi"
|
||||||
"C+ ${tftpRoot}/undionly.kpxe 0644 root root - ${pkgs.ipxe}/undionly.kpxe"
|
"C+ ${tftpRoot}/undionly.kpxe 0644 root root - ${pkgs.ipxe}/undionly.kpxe"
|
||||||
];
|
];
|
||||||
|
|
||||||
systemd.services.stage-systemrescue = {
|
services = {
|
||||||
|
fetch-debian-netboot = {
|
||||||
|
description = "Download Debian ${debianRelease} netboot kernel and initrd for HTTP PXE boot";
|
||||||
|
after = [
|
||||||
|
"local-fs.target"
|
||||||
|
"systemd-tmpfiles-setup.service"
|
||||||
|
"network-online.target"
|
||||||
|
];
|
||||||
|
wants = [ "network-online.target" ];
|
||||||
|
wantedBy = [ "multi-user.target" ];
|
||||||
|
serviceConfig = {
|
||||||
|
Type = "oneshot";
|
||||||
|
ExecStart = fetchDebianNetboot;
|
||||||
|
RemainAfterExit = true;
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
fetch-rocky-pxeboot = {
|
||||||
|
description = "Download Rocky Linux ${rockyRelease} pxeboot kernel and initrd for HTTP PXE boot";
|
||||||
|
after = [
|
||||||
|
"local-fs.target"
|
||||||
|
"systemd-tmpfiles-setup.service"
|
||||||
|
"network-online.target"
|
||||||
|
];
|
||||||
|
wants = [ "network-online.target" ];
|
||||||
|
wantedBy = [ "multi-user.target" ];
|
||||||
|
serviceConfig = {
|
||||||
|
Type = "oneshot";
|
||||||
|
ExecStart = fetchRockyPxeboot;
|
||||||
|
RemainAfterExit = true;
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
stage-systemrescue = {
|
||||||
description = "Stage SystemRescue ISO contents for HTTP PXE boot";
|
description = "Stage SystemRescue ISO contents for HTTP PXE boot";
|
||||||
after = [
|
after = [
|
||||||
"local-fs.target"
|
"local-fs.target"
|
||||||
@@ -148,9 +420,32 @@ in
|
|||||||
ExecStart = stageSystemRescue;
|
ExecStart = stageSystemRescue;
|
||||||
};
|
};
|
||||||
};
|
};
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
services.openssh.settings.PermitRootLogin = "yes";
|
services.dnsmasq = {
|
||||||
|
enable = true;
|
||||||
|
settings = {
|
||||||
|
# Disable DNS listener — only proxy DHCP is needed here.
|
||||||
|
# Without this dnsmasq tries to bind port 53 which systemd-resolved
|
||||||
|
# already owns, causing startup failure.
|
||||||
|
port = 0;
|
||||||
|
dhcp-range = [ "${lanBaseAddr},proxy" ];
|
||||||
|
dhcp-match = [
|
||||||
|
"set:ipxe,175"
|
||||||
|
"set:efi64,option:client-arch,7"
|
||||||
|
"set:efi64,option:client-arch,9"
|
||||||
|
];
|
||||||
|
dhcp-userclass = "set:ipxe,iPXE";
|
||||||
|
dhcp-boot = [
|
||||||
|
"tag:ipxe,tag:efi64,http://${vars.pxeServerIp}/boot.ipxe"
|
||||||
|
"tag:ipxe,http://${vars.pxeServerIp}/boot.ipxe"
|
||||||
|
"tag:efi64,ipxe.efi,,${vars.pxeServerIp}"
|
||||||
|
"undionly.kpxe,,${vars.pxeServerIp}"
|
||||||
|
];
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
networking.firewall.allowedTCPPorts = [ 80 ];
|
networking.firewall.allowedTCPPorts = [ vars.ports.pxeBootHttp ];
|
||||||
networking.firewall.allowedUDPPorts = [ 69 ];
|
networking.firewall.allowedUDPPorts = [ vars.ports.pxeBootTftp vars.ports.dhcp ];
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,28 +0,0 @@
|
|||||||
{ ... }:
|
|
||||||
|
|
||||||
{
|
|
||||||
imports = [
|
|
||||||
../beszel/enable-agent.nix
|
|
||||||
../services/zfs/enable-service.nix
|
|
||||||
];
|
|
||||||
|
|
||||||
boot.zfs.extraPools = [ "tank" ];
|
|
||||||
|
|
||||||
systemd.services.nfs-server = {
|
|
||||||
after = [ "zfs-mount.service" ];
|
|
||||||
requires = [ "zfs-mount.service" ];
|
|
||||||
};
|
|
||||||
|
|
||||||
services.nfs.server = {
|
|
||||||
enable = true;
|
|
||||||
exports = ''
|
|
||||||
/tank/docker/config 192.168.2.0/24(rw,sync,no_subtree_check,no_root_squash)
|
|
||||||
/tank/docker/volumes 192.168.2.0/24(rw,sync,no_subtree_check,no_root_squash)
|
|
||||||
/tank/docker/databases 192.168.2.0/24(rw,sync,no_subtree_check,no_root_squash)
|
|
||||||
/tank/docker/nextcloud-data 192.168.2.0/24(rw,sync,no_subtree_check,no_root_squash)
|
|
||||||
/tank/raspi/volumes 192.168.2.0/24(rw,sync,no_subtree_check,no_root_squash)
|
|
||||||
'';
|
|
||||||
};
|
|
||||||
|
|
||||||
networking.firewall.allowedTCPPorts = [ 111 2049 ];
|
|
||||||
}
|
|
||||||
@@ -0,0 +1,44 @@
|
|||||||
|
{ vars, ... }:
|
||||||
|
|
||||||
|
{
|
||||||
|
imports = [
|
||||||
|
../tailscale/subnet-router.nix
|
||||||
|
../tailscale/ts-dns-forwarder.nix
|
||||||
|
../beszel/enable-agent.nix
|
||||||
|
];
|
||||||
|
|
||||||
|
# "server", not "both": this build type advertises LAN subnet routes but
|
||||||
|
# doesn't use another tailscale exit node itself, so it doesn't need the
|
||||||
|
# "client"-side loose reverse-path filtering that "both" would also enable.
|
||||||
|
# Deliberately kept explicit here (not just relying on subnet-router.nix's
|
||||||
|
# own setting) so the intent is clear at the build-type level.
|
||||||
|
services.tailscale.useRoutingFeatures = "server";
|
||||||
|
|
||||||
|
# Advertise the LAN subnet so Tailscale peers can route back to LAN machines.
|
||||||
|
# Must also be approved in the Tailscale admin console (Machines → Edit route settings).
|
||||||
|
services.tailscale.extraUpFlags = [ "--advertise-routes=${vars.lanCidr}" ];
|
||||||
|
|
||||||
|
networking.firewall = {
|
||||||
|
# Forwarded subnet-router traffic arrives on tailscale0 already
|
||||||
|
# tailscale-authenticated -- the firewall's normal per-port allow-list
|
||||||
|
# would otherwise drop it. Standard NixOS/Tailscale subnet-router guidance.
|
||||||
|
trustedInterfaces = [ "tailscale0" ];
|
||||||
|
|
||||||
|
# SNAT LAN traffic going into Tailscale so the remote peer sees it as
|
||||||
|
# coming from this router's Tailscale IP rather than a raw LAN IP.
|
||||||
|
# Without this, Tailscale drops forwarded packets whose source is not a
|
||||||
|
# recognised Tailscale address.
|
||||||
|
#
|
||||||
|
# We target POSTROUTING directly (always-existing built-in chain) rather
|
||||||
|
# than nixos-nat-post: extraCommands runs after the old nixos-nat-post is
|
||||||
|
# deleted but before the new one is created, so -A nixos-nat-post silently
|
||||||
|
# fails. The -C check makes the rule idempotent across firewall reloads.
|
||||||
|
extraCommands = ''
|
||||||
|
iptables -t nat -C POSTROUTING -s ${vars.lanCidr} -o tailscale0 -j MASQUERADE 2>/dev/null || \
|
||||||
|
iptables -t nat -A POSTROUTING -s ${vars.lanCidr} -o tailscale0 -j MASQUERADE
|
||||||
|
'';
|
||||||
|
extraStopCommands = ''
|
||||||
|
iptables -t nat -D POSTROUTING -s ${vars.lanCidr} -o tailscale0 -j MASQUERADE 2>/dev/null || true
|
||||||
|
'';
|
||||||
|
};
|
||||||
|
}
|
||||||
@@ -0,0 +1,8 @@
|
|||||||
|
{ ... }:
|
||||||
|
|
||||||
|
{
|
||||||
|
imports = [
|
||||||
|
../tor/enable-relay.nix
|
||||||
|
../beszel/enable-agent.nix
|
||||||
|
];
|
||||||
|
}
|
||||||
@@ -0,0 +1,30 @@
|
|||||||
|
{ pkgs, ... }: {
|
||||||
|
# Defines the SSH host key as a clan vars generator so that:
|
||||||
|
# - `clan vars generate <target>` creates and encrypts the key pair
|
||||||
|
# - The private key lives at vars/per-machine/<target>/openssh/ssh_host_ed25519_key/secret
|
||||||
|
# (sops binary-encrypted, admin-key-only; decrypted by the build script)
|
||||||
|
# - The public key lives at vars/per-machine/<target>/openssh/ssh_host_ed25519_key.pub/value
|
||||||
|
# (plaintext; used by sync-host-keys.sh to derive the sops age fingerprint)
|
||||||
|
#
|
||||||
|
# neededFor = "activation" means clan's deployment tool would upload this
|
||||||
|
# before running nixos-rebuild/nixos-install (for VM/baremetal via
|
||||||
|
# nixos-anywhere). For lxc-* hosts, the build script bakes it into the
|
||||||
|
# tarball directly via NIXOS_HOST_KEYS_DIR -- the neededFor value here
|
||||||
|
# simply ensures it is NOT mapped to sops.secrets (which would try to
|
||||||
|
# decrypt it at runtime as a regular service secret, which is wrong: the
|
||||||
|
# SSH host key reaches the container via the tarball, not sops).
|
||||||
|
clan.core.vars.generators.openssh = {
|
||||||
|
files."ssh_host_ed25519_key" = {
|
||||||
|
secret = true;
|
||||||
|
neededFor = "activation";
|
||||||
|
};
|
||||||
|
files."ssh_host_ed25519_key.pub" = {
|
||||||
|
secret = false;
|
||||||
|
neededFor = "activation";
|
||||||
|
};
|
||||||
|
runtimeInputs = [ pkgs.openssh ];
|
||||||
|
script = ''
|
||||||
|
ssh-keygen -t ed25519 -N "" -C "" -f "$out/ssh_host_ed25519_key"
|
||||||
|
'';
|
||||||
|
};
|
||||||
|
}
|
||||||
@@ -1,29 +1,7 @@
|
|||||||
{ config, pkgs, lib, ... }:
|
_:
|
||||||
|
|
||||||
let
|
|
||||||
# Flake attribute names are now <platform>-<buildtype> (e.g. proxmox-docker)
|
|
||||||
# and no longer match networking.hostName, since a host's hostname stays
|
|
||||||
# fixed while the platform backing it can change. Each nixosConfiguration
|
|
||||||
# stamps its own active target name into /etc/flake-target at build time.
|
|
||||||
mySwitchCmd = ''
|
|
||||||
sudo nixos-rebuild switch \
|
|
||||||
--no-write-lock-file \
|
|
||||||
--refresh \
|
|
||||||
--flake git+https://gitea.lan.ddnsgeek.com/beatzaplenty/nixos.git#$(cat /etc/flake-target)
|
|
||||||
'';
|
|
||||||
myTestCmd = ''
|
|
||||||
sudo nixos-rebuild test \
|
|
||||||
--no-write-lock-file \
|
|
||||||
--refresh \
|
|
||||||
--flake git+https://gitea.lan.ddnsgeek.com/beatzaplenty/nixos.git#$(cat /etc/flake-target)
|
|
||||||
'';
|
|
||||||
in
|
|
||||||
{
|
{
|
||||||
programs.bash = {
|
# Switch-nix, Test-nix, and buildImage are defined system-wide in
|
||||||
enable = true;
|
# modules/common/configuration.nix so all users (including IPA accounts)
|
||||||
shellAliases = {
|
# get them. Add any Home-Manager-only per-user shell config here.
|
||||||
"Switch-nix" = mySwitchCmd;
|
|
||||||
"Test-nix" = myTestCmd;
|
|
||||||
};
|
|
||||||
};
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,91 +1,122 @@
|
|||||||
{ config, lib, pkgs, ... }:
|
{ config, lib, pkgs, vars, ... }:
|
||||||
|
|
||||||
|
let
|
||||||
|
switchCmd = ''
|
||||||
|
sudo nixos-rebuild switch \
|
||||||
|
--no-write-lock-file \
|
||||||
|
--refresh \
|
||||||
|
--flake git+https://${vars.giteaDomain}/${vars.giteaRepoPath}.git?dir=${vars.giteaRepoFlakePath}#$(cat /etc/flake-target)
|
||||||
|
'';
|
||||||
|
testCmd = ''
|
||||||
|
sudo nixos-rebuild test \
|
||||||
|
--no-write-lock-file \
|
||||||
|
--refresh \
|
||||||
|
--flake git+https://${vars.giteaDomain}/${vars.giteaRepoPath}.git?dir=${vars.giteaRepoFlakePath}#$(cat /etc/flake-target)
|
||||||
|
'';
|
||||||
|
buildImageFn = ''
|
||||||
|
buildImage() {
|
||||||
|
if [ -z "$1" ]; then
|
||||||
|
echo "usage: buildImage <flake-target> (e.g. lxc-docker)" >&2
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
NIXOS_HOST_KEYS_DIR="$(pwd)/host-keys" nix build --impure \
|
||||||
|
".#nixosConfigurations.$1.config.system.build.tarball"
|
||||||
|
}
|
||||||
|
'';
|
||||||
|
in
|
||||||
{
|
{
|
||||||
imports =
|
imports = [
|
||||||
[ # Include the results of the hardware scan.
|
./set-locale.nix
|
||||||
# ./hardware-configuration.nix
|
../ipa/client.nix
|
||||||
../set-locale.nix
|
|
||||||
];
|
];
|
||||||
# Use the GRUB 2 boot loader.
|
|
||||||
# boot.loader.grub.enable = true;
|
|
||||||
#boot.loader.grub.device = "/dev/sda"; # or "nodev" for efi only
|
|
||||||
|
|
||||||
networking.networkmanager.enable = true; # Easiest to use and most distros use this by default.
|
# System-wide shell config so all users (including IPA accounts) get the
|
||||||
|
# same management aliases as the local nixos user's Home Manager provides.
|
||||||
|
programs.bash = {
|
||||||
|
shellAliases = {
|
||||||
|
"Switch-nix" = switchCmd;
|
||||||
|
"Test-nix" = testCmd;
|
||||||
|
};
|
||||||
|
interactiveShellInit = buildImageFn;
|
||||||
|
};
|
||||||
|
|
||||||
# Set your time zone.
|
networking.networkmanager.enable = true;
|
||||||
time.timeZone = "Australia/Brisbane";
|
|
||||||
|
# Recommended over the true default (bypasses ZFS's own import safeguards)
|
||||||
|
# per the option's own docs; matches hosts/docker/host.nix and
|
||||||
|
# modules/services/zfs/enable-service.nix. Harmless no-op on hosts without ZFS.
|
||||||
|
boot.zfs.forceImportRoot = false;
|
||||||
|
|
||||||
|
time.timeZone = vars.timeZone;
|
||||||
|
|
||||||
# Enable QEMU agent
|
|
||||||
services.qemuGuest.enable = true;
|
services.qemuGuest.enable = true;
|
||||||
|
|
||||||
# Enable docker-compose
|
|
||||||
environment.systemPackages = with pkgs; [
|
environment.systemPackages = with pkgs; [
|
||||||
vim
|
vim
|
||||||
btop
|
btop
|
||||||
git
|
git
|
||||||
gcr
|
gcr
|
||||||
|
jq
|
||||||
];
|
];
|
||||||
|
|
||||||
# Secrets shared by every host, decrypted at activation via each host's
|
# Secrets shared by every host, decrypted at activation via each host's
|
||||||
# existing SSH host key (sops-nix derives the age key from
|
# SSH host key (sops-nix derives the age key from
|
||||||
# /etc/ssh/ssh_host_ed25519_key automatically — see modules/common/README
|
# /etc/ssh/ssh_host_ed25519_key automatically). hashedPassword secrets need
|
||||||
# or docs/ for the sops workflow). hashedPassword/hashedPasswordFile need
|
|
||||||
# neededForUsers so they're available before the normal secret-activation
|
# neededForUsers so they're available before the normal secret-activation
|
||||||
# step, since user creation happens very early in boot.
|
# step — user creation happens very early in boot.
|
||||||
sops.defaultSopsFile = ../../secrets/common.yaml;
|
sops = {
|
||||||
sops.secrets."root-hashedPassword".neededForUsers = true;
|
defaultSopsFile = ../../secrets/common.yaml;
|
||||||
sops.secrets."nixos-hashedPassword".neededForUsers = true;
|
|
||||||
sops.secrets."nix-github-token" = { };
|
|
||||||
|
|
||||||
# nix.conf doesn't support a *File-style option for access-tokens, so the
|
secrets = {
|
||||||
# token is rendered into a runtime-only file (never touches the Nix store)
|
"root-hashedPassword".neededForUsers = true;
|
||||||
# and pulled in via nix.conf's native !include directive.
|
"nixos-hashedPassword".neededForUsers = true;
|
||||||
sops.templates."nix-github-token.conf".content = ''
|
"nix-github-token" = { };
|
||||||
access-tokens = github.com=${config.sops.placeholder."nix-github-token"}
|
"nix-gitea-token" = { };
|
||||||
|
};
|
||||||
|
|
||||||
|
# nix.conf has no *File-style option for access-tokens, so tokens are
|
||||||
|
# rendered into a runtime-only file (never touches the Nix store) and
|
||||||
|
# pulled in via nix.conf's native !include directive.
|
||||||
|
templates."nix-access-tokens.conf".content = ''
|
||||||
|
access-tokens = github.com=${config.sops.placeholder."nix-github-token"} ${vars.giteaDomain}=${config.sops.placeholder."nix-gitea-token"}
|
||||||
'';
|
'';
|
||||||
|
};
|
||||||
|
|
||||||
nix.extraOptions = ''
|
nix.extraOptions = ''
|
||||||
!include ${config.sops.templates."nix-github-token.conf".path}
|
!include ${config.sops.templates."nix-access-tokens.conf".path}
|
||||||
'';
|
'';
|
||||||
|
|
||||||
#Set root password
|
users = {
|
||||||
users.users.root = {
|
# mutableUsers = false makes update-users-groups.pl enforce hashedPasswordFile
|
||||||
|
# on every activation, not just on newly-created accounts. Without this, a
|
||||||
|
# freshly-built proxmox disk image (activation runs without a usable sops key,
|
||||||
|
# so both accounts land in shadow with '!') will never have its passwords fixed
|
||||||
|
# by subsequent boots.
|
||||||
|
mutableUsers = false;
|
||||||
|
|
||||||
|
users.root = {
|
||||||
hashedPasswordFile = config.sops.secrets."root-hashedPassword".path;
|
hashedPasswordFile = config.sops.secrets."root-hashedPassword".path;
|
||||||
};
|
};
|
||||||
|
|
||||||
# Define a user account. Don't forget to set a password with ‘passwd’.
|
users.${vars.primaryUser} = {
|
||||||
users.users.nixos = {
|
|
||||||
isNormalUser = true;
|
isNormalUser = true;
|
||||||
extraGroups = [ "wheel" ]; # Enable ‘sudo’ for the user.
|
extraGroups = [ "wheel" ];
|
||||||
packages = with pkgs; [
|
packages = with pkgs; [ tree ];
|
||||||
tree
|
|
||||||
];
|
|
||||||
hashedPasswordFile = config.sops.secrets."nixos-hashedPassword".path;
|
hashedPasswordFile = config.sops.secrets."nixos-hashedPassword".path;
|
||||||
openssh.authorizedKeys.keys = [
|
openssh.authorizedKeys.keys = [ vars.adminSshKey ] ++ vars.extraAdminSshKeys;
|
||||||
"ssh-rsa AAAAB3NzaC1yc2EAAAADAQABAAABgQCq/Q5LvIXlZwO2kdeAN5nLGZ59nZB7JHYMEszHxmNtGMzv1lM31jiPNsr0z2EKVZhE7OOfa2IF9rhWYD7JUA9G0yzdZ4WTXFNGVVOJoOVH6vAF3XCxoVilOEwTc7h2Wiy+rzd0B28/3spffzQQWJhY6GRQVa8j+6xAGF60Fcvl1vLosYT9Bn2ZbK4TCWOwAn2jqXIieGpZdn/UNZbGOeKRiCvhktDfMAzuQzN/9jMu/oF4pkPn2X1UrsQdNlvp0Ci8md612MozIpncQJyAF1ADhunr3sMx0isUXiqD29R5DS4TftpekqLNLak+zcxFa8N7DcRNp3DcKfJvyTkwQrR4r+b7lFLYOLHLagSso9CzeW/paAS2q9I5SBm/2DtE1diLLg2jZikYcstsu/G5RgvbzbKqjiaMwTdXC3AMvDxQrs7U5pDRZFzoofG3cpODbTm+uy3m0kP70z0M1K45UbDG0p+itnTu9x40JbQEgefbx38AItNvAIx1A8HO4I1VX28= wayne@stream"
|
};
|
||||||
"ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAICMJhrfFayLBG+gWtO6oAvgambw5nWWgztiTFEaaaVRH debian@surface"
|
|
||||||
];
|
|
||||||
};
|
};
|
||||||
|
|
||||||
|
|
||||||
# Enable the OpenSSH daemon.
|
|
||||||
services.openssh.enable = true;
|
services.openssh.enable = true;
|
||||||
|
|
||||||
#Enable flakes
|
|
||||||
|
|
||||||
nix.settings = {
|
nix.settings = {
|
||||||
experimental-features = [ "nix-command" "flakes" ];
|
experimental-features = [ "nix-command" "flakes" ];
|
||||||
auto-optimise-store = true;
|
auto-optimise-store = true;
|
||||||
};
|
};
|
||||||
|
|
||||||
|
|
||||||
programs.git = {
|
programs.git = {
|
||||||
enable = true;
|
enable = true;
|
||||||
package = pkgs.git;
|
package = pkgs.git;
|
||||||
config = {
|
config.credential.helper = "store";
|
||||||
credential.helper = "store";
|
|
||||||
};
|
};
|
||||||
};
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
}
|
}
|
||||||
+22
-18
@@ -1,17 +1,34 @@
|
|||||||
{ config, pkgs, lib, ... }:
|
{ config, pkgs, lib, vars, ... }:
|
||||||
|
|
||||||
let
|
let
|
||||||
remote = "root@proxmox-ip:/var/lib/vz/template/iso";
|
remote = "root@proxmox-ip:/var/lib/vz/template/iso";
|
||||||
localMount = "${config.home.homeDirectory}/proxmox-iso";
|
localMount = "${config.home.homeDirectory}/proxmox-iso";
|
||||||
in {
|
in
|
||||||
|
{
|
||||||
|
|
||||||
imports = [
|
imports = [
|
||||||
./aliases.nix
|
./aliases.nix
|
||||||
];
|
];
|
||||||
|
|
||||||
home.username = "nixos"; # your actual username
|
home = {
|
||||||
home.homeDirectory = "/home/nixos";
|
username = vars.primaryUser;
|
||||||
home.stateVersion = "25.11"; # match your NixOS stateVersion
|
homeDirectory = "/home/${vars.primaryUser}";
|
||||||
|
stateVersion = "25.11"; # match your NixOS stateVersion
|
||||||
|
|
||||||
|
# Optional: packages
|
||||||
|
packages = with pkgs; [
|
||||||
|
git
|
||||||
|
vim
|
||||||
|
tmux
|
||||||
|
nano
|
||||||
|
sshfs
|
||||||
|
];
|
||||||
|
|
||||||
|
# Optional: set environment vars
|
||||||
|
sessionVariables = {
|
||||||
|
EDITOR = "nano";
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
programs.home-manager.enable = true; # mandatory to activate HM
|
programs.home-manager.enable = true; # mandatory to activate HM
|
||||||
|
|
||||||
@@ -22,19 +39,6 @@ programs.bash.enable = true;
|
|||||||
# modules/common/configuration.nix instead (covers the daemon for every
|
# modules/common/configuration.nix instead (covers the daemon for every
|
||||||
# user, not just this one).
|
# user, not just this one).
|
||||||
|
|
||||||
# Optional: packages
|
|
||||||
home.packages = with pkgs; [
|
|
||||||
git
|
|
||||||
vim
|
|
||||||
tmux
|
|
||||||
nano
|
|
||||||
sshfs
|
|
||||||
];
|
|
||||||
|
|
||||||
# Optional: set environment vars
|
|
||||||
home.sessionVariables = {
|
|
||||||
EDITOR = "nano";
|
|
||||||
};
|
|
||||||
# systemd.user.services.mount-proxmox-iso = {
|
# systemd.user.services.mount-proxmox-iso = {
|
||||||
# Unit = {
|
# Unit = {
|
||||||
# Description = "Mount Proxmox ISO dir via SSHFS";
|
# Description = "Mount Proxmox ISO dir via SSHFS";
|
||||||
|
|||||||
@@ -0,0 +1,35 @@
|
|||||||
|
# Shared activation-script logic to preserve the SSH host key across
|
||||||
|
# nixos-rebuild on platforms that embed the key via environment.etc (lxc and
|
||||||
|
# proxmox). When NIXOS_HOST_KEYS_DIR is not set the key is absent from
|
||||||
|
# environment.etc, and NixOS's etc activation removes any /etc file not in
|
||||||
|
# the new generation — which would destroy the live key and break sops-nix
|
||||||
|
# decryption permanently. These scripts save the key to /run before etc
|
||||||
|
# removes it, then restore it afterward.
|
||||||
|
#
|
||||||
|
# Explicit deps enforce the correct ordering: without them the topological
|
||||||
|
# sort places preserveSshHostKey after etc (confirmed live on lxc-tor-relay:
|
||||||
|
# position 7 vs etc's position 5), so the key is gone before it can be saved.
|
||||||
|
_: {
|
||||||
|
system.activationScripts = {
|
||||||
|
preserveSshHostKey = ''
|
||||||
|
if [ -f /etc/ssh/ssh_host_ed25519_key ]; then
|
||||||
|
cp /etc/ssh/ssh_host_ed25519_key /run/sshd-host-key-preserve.tmp
|
||||||
|
cp /etc/ssh/ssh_host_ed25519_key.pub /run/sshd-host-key-preserve.pub.tmp
|
||||||
|
fi
|
||||||
|
'';
|
||||||
|
|
||||||
|
restoreSshHostKey = {
|
||||||
|
deps = [ "etc" ];
|
||||||
|
text = ''
|
||||||
|
if [ ! -f /etc/ssh/ssh_host_ed25519_key ] && [ -f /run/sshd-host-key-preserve.tmp ]; then
|
||||||
|
install -m 0600 /run/sshd-host-key-preserve.tmp /etc/ssh/ssh_host_ed25519_key
|
||||||
|
install -m 0644 /run/sshd-host-key-preserve.pub.tmp /etc/ssh/ssh_host_ed25519_key.pub
|
||||||
|
fi
|
||||||
|
rm -f /run/sshd-host-key-preserve.tmp /run/sshd-host-key-preserve.pub.tmp
|
||||||
|
'';
|
||||||
|
};
|
||||||
|
|
||||||
|
etc = { deps = [ "preserveSshHostKey" ]; };
|
||||||
|
setupSecrets = { deps = [ "restoreSshHostKey" ]; };
|
||||||
|
};
|
||||||
|
}
|
||||||
@@ -1,4 +1,4 @@
|
|||||||
{ ... }:
|
_:
|
||||||
|
|
||||||
{
|
{
|
||||||
i18n.defaultLocale = "en_AU.UTF-8";
|
i18n.defaultLocale = "en_AU.UTF-8";
|
||||||
@@ -0,0 +1,87 @@
|
|||||||
|
{ vars, ... }:
|
||||||
|
|
||||||
|
{
|
||||||
|
# ZFS RAID0 (striped, no redundancy) root pool for the bare-metal gui
|
||||||
|
# host — two disks, each contributing its own top-level vdev. disko's
|
||||||
|
# zpool `mode` defaults to "" (plain stripe) when left unset, which is
|
||||||
|
# what gives RAID0 semantics here rather than mirror/raidz.
|
||||||
|
#
|
||||||
|
# Device paths are placeholders until the real hardware profile lands —
|
||||||
|
# fill in vars.guiRootDisk1/guiRootDisk2 (stable /dev/disk/by-id/...
|
||||||
|
# paths, not /dev/sdX) before running disko against real hardware. Swap
|
||||||
|
# is deliberately left out for now — sizing that sensibly needs the
|
||||||
|
# box's actual RAM size, which comes with the hardware profile too.
|
||||||
|
#
|
||||||
|
# Not yet imported anywhere: this awaits the new bare-metal platform
|
||||||
|
# module (alongside modules/boot/efi.nix for systemd-boot, matching
|
||||||
|
# modules/platforms/proxmox.nix's pattern) once the hardware config is
|
||||||
|
# in hand.
|
||||||
|
disko.devices = {
|
||||||
|
disk = {
|
||||||
|
disk1 = {
|
||||||
|
type = "disk";
|
||||||
|
device = vars.guiRootDisk1;
|
||||||
|
|
||||||
|
content = {
|
||||||
|
type = "gpt";
|
||||||
|
|
||||||
|
partitions = {
|
||||||
|
esp = {
|
||||||
|
priority = 1;
|
||||||
|
name = "ESP";
|
||||||
|
size = "512M";
|
||||||
|
type = "EF00";
|
||||||
|
|
||||||
|
content = {
|
||||||
|
type = "filesystem";
|
||||||
|
format = "vfat";
|
||||||
|
mountpoint = "/boot";
|
||||||
|
mountOptions = [ "umask=0077" ];
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
zfs = {
|
||||||
|
size = "100%";
|
||||||
|
|
||||||
|
content = {
|
||||||
|
type = "zfs";
|
||||||
|
pool = "rpool";
|
||||||
|
};
|
||||||
|
};
|
||||||
|
};
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
disk2 = {
|
||||||
|
type = "disk";
|
||||||
|
device = vars.guiRootDisk2;
|
||||||
|
|
||||||
|
content = {
|
||||||
|
type = "gpt";
|
||||||
|
|
||||||
|
partitions = {
|
||||||
|
zfs = {
|
||||||
|
size = "100%";
|
||||||
|
|
||||||
|
content = {
|
||||||
|
type = "zfs";
|
||||||
|
pool = "rpool";
|
||||||
|
};
|
||||||
|
};
|
||||||
|
};
|
||||||
|
};
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
zpool.rpool = {
|
||||||
|
type = "zpool";
|
||||||
|
|
||||||
|
rootFsOptions = {
|
||||||
|
compression = "zstd";
|
||||||
|
"com.sun:auto-snapshot" = "false";
|
||||||
|
};
|
||||||
|
mountpoint = "/";
|
||||||
|
options.ashift = "12";
|
||||||
|
};
|
||||||
|
};
|
||||||
|
}
|
||||||
@@ -0,0 +1,35 @@
|
|||||||
|
_:
|
||||||
|
|
||||||
|
{
|
||||||
|
# Linode provisions and sizes these disks itself (via the Linode
|
||||||
|
# dashboard/API) before the OS ever boots, and presents them as whole,
|
||||||
|
# unpartitioned block devices — /dev/sda is the root filesystem directly,
|
||||||
|
# /dev/sdb is swap directly, no partition table on either. Nothing here
|
||||||
|
# should ever repartition or resize them:
|
||||||
|
# - `destroy = false` skips each disk entirely during disko's destroy
|
||||||
|
# stage (see disko's disk.destroy option) — no wipefs, ever.
|
||||||
|
# - the filesystem content type's own create step only runs mkfs if the
|
||||||
|
# device isn't already formatted (checked via `blkid`), so re-running
|
||||||
|
# this against an already-provisioned Linode disk is a no-op.
|
||||||
|
disko.devices.disk = {
|
||||||
|
main = {
|
||||||
|
device = "/dev/sda";
|
||||||
|
destroy = false;
|
||||||
|
|
||||||
|
content = {
|
||||||
|
type = "filesystem";
|
||||||
|
format = "ext4";
|
||||||
|
mountpoint = "/";
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
swap = {
|
||||||
|
device = "/dev/sdb";
|
||||||
|
destroy = false;
|
||||||
|
|
||||||
|
content = {
|
||||||
|
type = "swap";
|
||||||
|
};
|
||||||
|
};
|
||||||
|
};
|
||||||
|
}
|
||||||
@@ -1,4 +1,4 @@
|
|||||||
{ ... }:
|
{ config, vars, ... }:
|
||||||
|
|
||||||
{
|
{
|
||||||
disko.devices = {
|
disko.devices = {
|
||||||
@@ -6,6 +6,16 @@
|
|||||||
type = "disk";
|
type = "disk";
|
||||||
device = "/dev/sda";
|
device = "/dev/sda";
|
||||||
|
|
||||||
|
# Only used when building a standalone disk image directly (`nix build
|
||||||
|
# .#nixosConfigurations.<host>.config.system.build.diskoImagesScript`)
|
||||||
|
# rather than formatting a real device — see docs/proxmox-images.md.
|
||||||
|
# imageSize sets the .raw file's total size (root's "100%" below fills
|
||||||
|
# whatever's left after ESP + swap within it); imageName keeps each
|
||||||
|
# host's image distinctly named instead of every proxmox-* host
|
||||||
|
# producing an identical "main.raw".
|
||||||
|
imageSize = vars.proxmoxImageSize;
|
||||||
|
imageName = config.networking.hostName;
|
||||||
|
|
||||||
content = {
|
content = {
|
||||||
type = "gpt";
|
type = "gpt";
|
||||||
|
|
||||||
|
|||||||
+2
-2
@@ -1,4 +1,4 @@
|
|||||||
{ pkgs, ... }:
|
{ pkgs, vars, ... }:
|
||||||
|
|
||||||
{
|
{
|
||||||
systemd.services.docker-health-to-gotify = {
|
systemd.services.docker-health-to-gotify = {
|
||||||
@@ -9,7 +9,7 @@
|
|||||||
# Run as root so it can read /etc/secrets and access docker socket
|
# Run as root so it can read /etc/secrets and access docker socket
|
||||||
# User = "root";
|
# User = "root";
|
||||||
#EnvironmentFile = "-/etc/secrets/docker-health-alert.env";
|
#EnvironmentFile = "-/etc/secrets/docker-health-alert.env";
|
||||||
ExecStart = "${pkgs.bash}/bin/bash /home/nixos/docker/monitoring/gotify/docker-health-to-gotify.sh";
|
ExecStart = "${pkgs.bash}/bin/bash /home/${vars.primaryUser}/docker/monitoring/gotify/docker-health-to-gotify.sh";
|
||||||
StandardOutput = "journal";
|
StandardOutput = "journal";
|
||||||
StandardError = "journal";
|
StandardError = "journal";
|
||||||
};
|
};
|
||||||
@@ -1,23 +1,45 @@
|
|||||||
{ pkgs, ... }:
|
{ lib, pkgs, vars, ... }:
|
||||||
|
|
||||||
|
let
|
||||||
|
gid = toString vars.dockerAccessGid;
|
||||||
|
in
|
||||||
{
|
{
|
||||||
# virtualisation.docker.enable = true;
|
|
||||||
virtualisation.docker = {
|
virtualisation.docker = {
|
||||||
enable = true;
|
enable = true;
|
||||||
package = pkgs.docker;
|
package = pkgs.docker;
|
||||||
# listenOptions = [
|
|
||||||
# "unix:///var/run/docker.sock"
|
|
||||||
# "tcp://0.0.0.0:2375"
|
|
||||||
#];
|
|
||||||
|
|
||||||
# daemon.settings = {
|
|
||||||
# metrics-addr = "0.0.0.0:9323";
|
|
||||||
# experimental = true;
|
|
||||||
# };
|
|
||||||
};
|
};
|
||||||
|
# Pin the docker group GID to match the IPA "docker-access" group so that
|
||||||
|
# IPA group membership alone grants access to the Docker socket. Any user
|
||||||
|
# whose supplementary groups (resolved by SSSD from IPA) include GID
|
||||||
|
# vars.dockerAccessGid will pass the socket group-permission check without
|
||||||
|
# any per-host users.groups.docker.members entry.
|
||||||
|
users.groups.docker.gid = lib.mkForce vars.dockerAccessGid;
|
||||||
|
users.users.${vars.primaryUser}.extraGroups = [ "docker" ];
|
||||||
environment.systemPackages = with pkgs; [
|
environment.systemPackages = with pkgs; [
|
||||||
docker-compose
|
docker-compose
|
||||||
docker-buildx
|
docker-buildx
|
||||||
];
|
];
|
||||||
|
|
||||||
|
# NixOS's group activation uses plain `groupmod` without --non-unique.
|
||||||
|
# When SSSD is active it exposes the IPA "docker-access" group at
|
||||||
|
# vars.dockerAccessGid via NSS, so groupmod sees that GID as already in
|
||||||
|
# use and silently skips the change (warning: "not applying GID change").
|
||||||
|
# This script runs after the normal "groups" step and applies the change
|
||||||
|
# with --non-unique (which lets the local docker group share the GID with
|
||||||
|
# the SSSD-provided IPA group). If the GID actually changed it also
|
||||||
|
# restarts docker.socket so the socket is recreated with the new GID.
|
||||||
|
system.activationScripts.docker-group-gid = {
|
||||||
|
deps = [ "groups" ];
|
||||||
|
text = ''
|
||||||
|
current=$(grep "^docker:" /etc/group | cut -d: -f3)
|
||||||
|
if [ "$current" != "${gid}" ]; then
|
||||||
|
${pkgs.shadow}/bin/groupmod --non-unique -g ${gid} docker
|
||||||
|
if ${pkgs.systemd}/bin/systemctl is-active --quiet docker.socket; then
|
||||||
|
${pkgs.systemd}/bin/systemctl stop docker.service docker.socket
|
||||||
|
rm -f /var/run/docker.sock
|
||||||
|
${pkgs.systemd}/bin/systemctl start docker.socket docker.service
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
'';
|
||||||
|
};
|
||||||
}
|
}
|
||||||
@@ -1,63 +1,79 @@
|
|||||||
{ config, lib, pkgs, ... }:
|
{ config, lib, pkgs, vars, ... }:
|
||||||
|
|
||||||
|
let
|
||||||
|
# `x-systemd.automount` never works inside a Linux container (LXC
|
||||||
|
# included, regardless of privilege) -- confirmed live on lxc-docker:
|
||||||
|
# systemd logs "Starting of <unit>.automount unsupported" for every
|
||||||
|
# share and never mounts them. Mount eagerly there instead, with
|
||||||
|
# `nofail` so a boot with the NFS server unreachable doesn't hang
|
||||||
|
# (the VM platforms rely on automount itself to get that same
|
||||||
|
# non-blocking behavior, so they don't need `nofail` too).
|
||||||
|
automountOpts = if config.boot.isContainer then [ "nofail" ] else [ "x-systemd.automount" ];
|
||||||
|
|
||||||
|
# FQDN in the storage.home zone — resolves to the Pacemaker vip-storage
|
||||||
|
# (192.168.20.229) on docker's eth1/vmbr2 interface. Using the DNS name
|
||||||
|
# rather than the raw IP means a future VIP renumber only requires a DNS
|
||||||
|
# update, not a NixOS rebuild. The storage.home zone is served by the same
|
||||||
|
# FreeIPA nameserver (domainControllerIp) that docker already uses, so
|
||||||
|
# resolution reaches it over eth0 without any extra routing.
|
||||||
|
nfsServer = vars.haStorageNfsFqdn;
|
||||||
|
storageRoot = vars.haStorageRoot;
|
||||||
|
in
|
||||||
{
|
{
|
||||||
fileSystems."/mnt/docker/config" = {
|
fileSystems = {
|
||||||
device = "server:/tank/docker/config";
|
${vars.nfsShares.dockerConfig.mountpoint} = {
|
||||||
|
device = "${nfsServer}:${storageRoot}/${vars.nfsShares.dockerConfig.subpath}";
|
||||||
fsType = "nfs";
|
fsType = "nfs";
|
||||||
|
|
||||||
options = [
|
options = [
|
||||||
"nfsvers=4.2"
|
"nfsvers=4.2"
|
||||||
"_netdev"
|
"_netdev"
|
||||||
"x-systemd.automount"
|
|
||||||
"noatime"
|
"noatime"
|
||||||
];
|
] ++ automountOpts;
|
||||||
};
|
};
|
||||||
|
|
||||||
fileSystems."/mnt/docker/databases" = {
|
${vars.nfsShares.dockerDatabases.mountpoint} = {
|
||||||
device = "server:/tank/docker/databases";
|
device = "${nfsServer}:${storageRoot}/${vars.nfsShares.dockerDatabases.subpath}";
|
||||||
fsType = "nfs";
|
fsType = "nfs";
|
||||||
|
|
||||||
options = [
|
options = [
|
||||||
"nfsvers=4.2"
|
"nfsvers=4.2"
|
||||||
"_netdev"
|
"_netdev"
|
||||||
"x-systemd.automount"
|
|
||||||
"noatime"
|
"noatime"
|
||||||
];
|
] ++ automountOpts;
|
||||||
};
|
};
|
||||||
|
|
||||||
fileSystems."/mnt/docker/volumes" = {
|
${vars.nfsShares.dockerVolumes.mountpoint} = {
|
||||||
device = "server:/tank/docker/volumes";
|
device = "${nfsServer}:${storageRoot}/${vars.nfsShares.dockerVolumes.subpath}";
|
||||||
fsType = "nfs";
|
fsType = "nfs";
|
||||||
|
|
||||||
options = [
|
options = [
|
||||||
"nfsvers=4.2"
|
"nfsvers=4.2"
|
||||||
"_netdev"
|
"_netdev"
|
||||||
"x-systemd.automount"
|
|
||||||
"noatime"
|
"noatime"
|
||||||
];
|
] ++ automountOpts;
|
||||||
};
|
};
|
||||||
|
|
||||||
fileSystems."/mnt/nextcloud-data" = {
|
${vars.nfsShares.nextcloudData.mountpoint} = {
|
||||||
device = "server:/tank/docker/nextcloud-data";
|
device = "${nfsServer}:${storageRoot}/${vars.nfsShares.nextcloudData.subpath}";
|
||||||
fsType = "nfs";
|
fsType = "nfs";
|
||||||
|
|
||||||
options = [
|
options = [
|
||||||
"nfsvers=4.2"
|
"nfsvers=4.2"
|
||||||
"_netdev"
|
"_netdev"
|
||||||
"x-systemd.automount"
|
|
||||||
"noatime"
|
"noatime"
|
||||||
];
|
] ++ automountOpts;
|
||||||
};
|
};
|
||||||
|
|
||||||
fileSystems."/mnt/raspi-backup" = {
|
${vars.nfsShares.raspiVolumes.mountpoint} = {
|
||||||
device = "server:/tank/raspi/volumes";
|
device = "${nfsServer}:${storageRoot}/${vars.nfsShares.raspiVolumes.subpath}";
|
||||||
fsType = "nfs";
|
fsType = "nfs";
|
||||||
|
|
||||||
options = [
|
options = [
|
||||||
"nfsvers=4.2"
|
"nfsvers=4.2"
|
||||||
"_netdev"
|
"_netdev"
|
||||||
"x-systemd.automount"
|
|
||||||
"noatime"
|
"noatime"
|
||||||
];
|
] ++ automountOpts;
|
||||||
|
};
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,22 @@
|
|||||||
|
{ pkgs, vars, ... }:
|
||||||
|
|
||||||
|
{
|
||||||
|
# Create nextcloud cron scheduled task
|
||||||
|
systemd.services.nextcloud = {
|
||||||
|
description = "Nextcloud scheduled task";
|
||||||
|
script = ''${pkgs.bash}/bin/bash ~/docker/services-up.sh --profile nextcloud exec -u 33 nextcloud-webapp php ./cron.php'';
|
||||||
|
serviceConfig = {
|
||||||
|
Type = "oneshot";
|
||||||
|
User = vars.primaryUser;
|
||||||
|
};
|
||||||
|
path = with pkgs; [ docker docker-compose ];
|
||||||
|
};
|
||||||
|
|
||||||
|
systemd.timers.nextcloud = {
|
||||||
|
wantedBy = [ "timers.target" ];
|
||||||
|
timerConfig = {
|
||||||
|
OnCalendar = "*:0/5";
|
||||||
|
Persistent = true;
|
||||||
|
};
|
||||||
|
};
|
||||||
|
}
|
||||||
@@ -0,0 +1,164 @@
|
|||||||
|
# Cluster-wide HA config shared by both ha-server nodes.
|
||||||
|
#
|
||||||
|
# Covers everything that is identical on both nodes and references cluster
|
||||||
|
# topology (node IPs, hostnames, DRBD resource). Per-node identity
|
||||||
|
# (hostname, static IP, stateVersion) lives in hosts/ha-server-{1,2}/host.nix.
|
||||||
|
#
|
||||||
|
# Corosync authkey:
|
||||||
|
# /etc/corosync/authkey (mode 0400) is managed by sops-nix below.
|
||||||
|
# Bootstrap: run scripts/ha/cluster-init.sh on node1 to generate the key,
|
||||||
|
# then encrypt it with: sops -e --input-type binary /etc/corosync/authkey > secrets/ha-corosync-authkey
|
||||||
|
# Both host keys must be registered via sync-host-keys.sh first so both nodes can decrypt it.
|
||||||
|
#
|
||||||
|
# DRBD fencing:
|
||||||
|
# resource-only with crm-fence-peer.sh: DRBD calls the Pacemaker-aware
|
||||||
|
# crm-fence-peer.sh handler before promoting. The handler checks the CIB
|
||||||
|
# to confirm the peer's DRBD resource is stopped and returns 7 (successfully
|
||||||
|
# fenced), allowing safe promotion without requiring power-fencing (STONITH).
|
||||||
|
# The unfence handler crm-unfence-peer.sh clears the outdate flag when the
|
||||||
|
# peer reconnects. This is the correct setting for Pacemaker+DRBD clusters
|
||||||
|
# with STONITH disabled; crm-fence-peer.sh replaces the need for a separate
|
||||||
|
# STONITH device during the testing phase. Switch to resource-and-stonith
|
||||||
|
# once the fence_pve_ssh STONITH resource is active (see
|
||||||
|
# scripts/ha/cluster-enable-stonith.sh).
|
||||||
|
#
|
||||||
|
# PATH wrapper: when the DRBD kernel module invokes the fence-peer handler
|
||||||
|
# via the UMH (User Mode Helper) mechanism it provides a minimal PATH that
|
||||||
|
# omits /run/current-system/sw/bin. crm-fence-peer.sh calls cibadmin,
|
||||||
|
# crm_mon etc.; if those aren't found a pipeline in the script breaks with
|
||||||
|
# SIGPIPE. A process killed by signal has WEXITSTATUS() == 0, so the kernel
|
||||||
|
# sees exit code 0 and logs "fence-peer helper broken, returned 0", looping
|
||||||
|
# forever. The writeShellScript wrappers below prepend the NixOS sw path
|
||||||
|
# before exec-ing the real handler, giving it a working Pacemaker toolchain.
|
||||||
|
{ lib, pkgs, vars, ... }:
|
||||||
|
let
|
||||||
|
fencePeerWrapper = pkgs.writeShellScript "drbd-fence-peer" ''
|
||||||
|
export PATH="/run/current-system/sw/bin:/run/current-system/sw/sbin:$PATH"
|
||||||
|
exec /run/current-system/sw/lib/drbd/crm-fence-peer.sh "$@"
|
||||||
|
'';
|
||||||
|
unfencePeerWrapper = pkgs.writeShellScript "drbd-unfence-peer" ''
|
||||||
|
export PATH="/run/current-system/sw/bin:/run/current-system/sw/sbin:$PATH"
|
||||||
|
exec /run/current-system/sw/lib/drbd/crm-unfence-peer.sh "$@"
|
||||||
|
'';
|
||||||
|
in
|
||||||
|
{
|
||||||
|
# Root SSH access — same key set as the nixos user so all admin keys can reach root.
|
||||||
|
users.users.root.openssh.authorizedKeys.keys = [ vars.adminSshKey ] ++ vars.extraAdminSshKeys;
|
||||||
|
|
||||||
|
# Passwordless sudo for wheel — operator SSHes as nixos and uses sudo for
|
||||||
|
# cluster management commands (drbdadm, crm*, pcs, etc.)
|
||||||
|
security.sudo.wheelNeedsPassword = lib.mkForce false;
|
||||||
|
|
||||||
|
# DRBD lock-file directory (drbd-utils checks for it; missing → harmless but noisy warnings).
|
||||||
|
systemd.tmpfiles.rules = [ "d /var/lib/drbd 0750 root root -" ];
|
||||||
|
|
||||||
|
# Prevent drbd.service from auto-starting at boot / nixos-rebuild switch.
|
||||||
|
# Pacemaker's OCF drbd agent calls drbdadm up/down directly when managing
|
||||||
|
# the resource. If drbd.service also runs drbdadm up all while DRBD is
|
||||||
|
# already Primary under Pacemaker, apply-al fails with "device busy" (exit 20).
|
||||||
|
systemd.services.drbd.wantedBy = lib.mkForce [ ];
|
||||||
|
|
||||||
|
services.drbd = {
|
||||||
|
enable = true;
|
||||||
|
config = ''
|
||||||
|
global {
|
||||||
|
usage-count yes;
|
||||||
|
}
|
||||||
|
|
||||||
|
common {
|
||||||
|
net {
|
||||||
|
protocol C;
|
||||||
|
ping-int 1;
|
||||||
|
verify-alg sha256;
|
||||||
|
after-sb-0pri discard-zero-changes;
|
||||||
|
after-sb-1pri discard-secondary;
|
||||||
|
}
|
||||||
|
disk {
|
||||||
|
fencing resource-only;
|
||||||
|
}
|
||||||
|
handlers {
|
||||||
|
fence-peer "${fencePeerWrapper}";
|
||||||
|
unfence-peer "${unfencePeerWrapper}";
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
resource ha-data {
|
||||||
|
volume 0 {
|
||||||
|
device /dev/drbd0;
|
||||||
|
disk ${vars.haServerDrbdDisk};
|
||||||
|
meta-disk internal;
|
||||||
|
}
|
||||||
|
|
||||||
|
on ${vars.haServer1Host} {
|
||||||
|
address ${vars.haServer1StorageIp}:${toString vars.ports.haServerDrbd};
|
||||||
|
}
|
||||||
|
|
||||||
|
on ${vars.haServer2Host} {
|
||||||
|
address ${vars.haServer2StorageIp}:${toString vars.ports.haServerDrbd};
|
||||||
|
}
|
||||||
|
}
|
||||||
|
'';
|
||||||
|
};
|
||||||
|
|
||||||
|
# /etc/corosync/authkey — sops binary secret, identical on both nodes.
|
||||||
|
# Decryptable by both ha-server host keys (added by sync-host-keys.sh).
|
||||||
|
sops.secrets.corosync_authkey = {
|
||||||
|
sopsFile = ../../secrets/ha-corosync-authkey;
|
||||||
|
format = "binary";
|
||||||
|
path = "/etc/corosync/authkey";
|
||||||
|
mode = "0400";
|
||||||
|
restartUnits = [ "corosync.service" ];
|
||||||
|
};
|
||||||
|
|
||||||
|
# NixOS common config enables NetworkManager by default; HA cluster nodes
|
||||||
|
# need stable static IPs with predictable interface names — NM is not suitable.
|
||||||
|
networking.networkmanager.enable = lib.mkForce false;
|
||||||
|
|
||||||
|
# services.corosync.enable is set by modules/ha/pacemaker-stack.nix.
|
||||||
|
services.corosync = {
|
||||||
|
clusterName = "ha-cluster";
|
||||||
|
nodelist = [
|
||||||
|
# ring0: cluster-internal vmbr1 (primary heartbeat + DRBD path)
|
||||||
|
# ring1: LAN vmbr0 (backup heartbeat only — never carries DRBD)
|
||||||
|
{ nodeid = 1; name = vars.haServer1Host; ring_addrs = [ vars.haServer1StorageIp vars.haServer1Ip ]; }
|
||||||
|
{ nodeid = 2; name = vars.haServer2Host; ring_addrs = [ vars.haServer2StorageIp vars.haServer2Ip ]; }
|
||||||
|
];
|
||||||
|
};
|
||||||
|
|
||||||
|
networking.firewall = {
|
||||||
|
allowedTCPPorts = [
|
||||||
|
vars.ports.haServerPacemakerRemoted
|
||||||
|
vars.ports.haServerPcsd
|
||||||
|
vars.ports.haServerDrbd
|
||||||
|
];
|
||||||
|
allowedUDPPorts = [
|
||||||
|
vars.ports.haServerCorosync1
|
||||||
|
vars.ports.haServerCorosync2
|
||||||
|
vars.ports.haServerCorosyncCrypto
|
||||||
|
];
|
||||||
|
# Protocol separation: iSCSI (VLAN 20 / storage clients only),
|
||||||
|
# NFS (VLAN 2 / LAN only). Cluster-internal subnets accepted wholesale
|
||||||
|
# since they are isolated bridges with no external uplink.
|
||||||
|
extraCommands = ''
|
||||||
|
iptables -A nixos-fw -s ${vars.haServer1Ip}/32 -j nixos-fw-accept
|
||||||
|
iptables -A nixos-fw -s ${vars.haServer2Ip}/32 -j nixos-fw-accept
|
||||||
|
iptables -A nixos-fw -s ${vars.haStorageCidr} -j nixos-fw-accept
|
||||||
|
|
||||||
|
iptables -A nixos-fw -p tcp -s ${vars.haClientCidr} --dport ${toString vars.ports.haServerIscsi} -j nixos-fw-accept
|
||||||
|
|
||||||
|
iptables -A nixos-fw -p tcp -s ${vars.lanCidr} --dport ${toString vars.ports.nfsRpcbind} -j nixos-fw-accept
|
||||||
|
iptables -A nixos-fw -p udp -s ${vars.lanCidr} --dport ${toString vars.ports.nfsRpcbind} -j nixos-fw-accept
|
||||||
|
iptables -A nixos-fw -p tcp -s ${vars.lanCidr} --dport ${toString vars.ports.nfsd} -j nixos-fw-accept
|
||||||
|
iptables -A nixos-fw -p udp -s ${vars.lanCidr} --dport ${toString vars.ports.nfsd} -j nixos-fw-accept
|
||||||
|
iptables -A nixos-fw -p tcp -s ${vars.lanCidr} --dport ${toString vars.ports.nfsMountd} -j nixos-fw-accept
|
||||||
|
iptables -A nixos-fw -p udp -s ${vars.lanCidr} --dport ${toString vars.ports.nfsMountd} -j nixos-fw-accept
|
||||||
|
|
||||||
|
iptables -A nixos-fw -p tcp -s ${vars.haClientCidr} --dport ${toString vars.ports.nfsRpcbind} -j nixos-fw-accept
|
||||||
|
iptables -A nixos-fw -p udp -s ${vars.haClientCidr} --dport ${toString vars.ports.nfsRpcbind} -j nixos-fw-accept
|
||||||
|
iptables -A nixos-fw -p tcp -s ${vars.haClientCidr} --dport ${toString vars.ports.nfsd} -j nixos-fw-accept
|
||||||
|
iptables -A nixos-fw -p udp -s ${vars.haClientCidr} --dport ${toString vars.ports.nfsd} -j nixos-fw-accept
|
||||||
|
iptables -A nixos-fw -p tcp -s ${vars.haClientCidr} --dport ${toString vars.ports.nfsMountd} -j nixos-fw-accept
|
||||||
|
iptables -A nixos-fw -p udp -s ${vars.haClientCidr} --dport ${toString vars.ports.nfsMountd} -j nixos-fw-accept
|
||||||
|
'';
|
||||||
|
};
|
||||||
|
}
|
||||||
@@ -0,0 +1,99 @@
|
|||||||
|
# LIO iSCSI target service (targetctl) for NixOS HA clusters.
|
||||||
|
#
|
||||||
|
# Provides the targetctl.service that saves/restores LIO configuration from
|
||||||
|
# /etc/target/saveconfig.json. Pacemaker manages this service via its
|
||||||
|
# systemd resource agent (class="systemd" type="targetctl").
|
||||||
|
#
|
||||||
|
# Why ExecStop is not simply "targetctl save":
|
||||||
|
# targetctl save writes the LIO config to JSON but does NOT remove the LIO
|
||||||
|
# target from the kernel's configfs. As a result, any fileio backing store
|
||||||
|
# that LIO has open (e.g. iscsi-lun.img on an XFS-over-DRBD filesystem)
|
||||||
|
# stays referenced in the kernel. The subsequent XFS umount from the
|
||||||
|
# Filesystem OCF resource then returns EBUSY and either hangs for the full
|
||||||
|
# op-stop timeout or fails outright, blocking the entire failover.
|
||||||
|
#
|
||||||
|
# The ExecStop script here additionally tears down the kernel LIO state
|
||||||
|
# via rtslib_fb after saving, so the backing-store file descriptor is
|
||||||
|
# released and umount succeeds immediately.
|
||||||
|
#
|
||||||
|
# Empty-config guard:
|
||||||
|
# The save step is skipped when no iSCSI targets are currently active.
|
||||||
|
# This prevents the secondary node (where LIO was never started) from
|
||||||
|
# overwriting a valid saveconfig.json with an empty one when Pacemaker
|
||||||
|
# stops the iscsi-target resource as part of a failover or cleanup.
|
||||||
|
{ pkgs, ... }:
|
||||||
|
|
||||||
|
let
|
||||||
|
python3 = pkgs.python3.withPackages (ps: [ ps.rtslib-fb ]);
|
||||||
|
targetctl = "${python3}/bin/targetctl";
|
||||||
|
|
||||||
|
targetctlStop = pkgs.writeScript "targetctl-stop" ''
|
||||||
|
#!${python3}/bin/python3
|
||||||
|
import subprocess, sys
|
||||||
|
import rtslib_fb
|
||||||
|
|
||||||
|
root = rtslib_fb.RTSRoot()
|
||||||
|
targets = list(root.targets)
|
||||||
|
if targets:
|
||||||
|
subprocess.run(
|
||||||
|
["${targetctl}", "save", "/etc/target/saveconfig.json"],
|
||||||
|
capture_output=True,
|
||||||
|
)
|
||||||
|
print(f"saved {len(targets)} iSCSI target(s)")
|
||||||
|
else:
|
||||||
|
print("no active LIO targets — saveconfig.json unchanged")
|
||||||
|
|
||||||
|
for target in targets:
|
||||||
|
try:
|
||||||
|
for tpg in list(target.tpgs):
|
||||||
|
tpg.enable = False
|
||||||
|
target.delete()
|
||||||
|
except Exception as e:
|
||||||
|
print(f"warn (target): {e}", file=sys.stderr)
|
||||||
|
for so in list(root.storage_objects):
|
||||||
|
try:
|
||||||
|
so.delete()
|
||||||
|
except Exception as e:
|
||||||
|
print(f"warn (backstore): {e}", file=sys.stderr)
|
||||||
|
print("LIO kernel target cleared")
|
||||||
|
'';
|
||||||
|
in
|
||||||
|
{
|
||||||
|
boot.kernelModules = [
|
||||||
|
"target_core_mod"
|
||||||
|
"iscsi_target_mod"
|
||||||
|
"target_core_file"
|
||||||
|
"target_core_pscsi"
|
||||||
|
"target_core_user"
|
||||||
|
"configfs"
|
||||||
|
];
|
||||||
|
|
||||||
|
systemd = {
|
||||||
|
mounts = [{
|
||||||
|
where = "/sys/kernel/config";
|
||||||
|
what = "configfs";
|
||||||
|
type = "configfs";
|
||||||
|
wantedBy = [ "multi-user.target" ];
|
||||||
|
before = [ "targetctl.service" ];
|
||||||
|
}];
|
||||||
|
services.targetctl = {
|
||||||
|
description = "LIO iSCSI target config save/restore";
|
||||||
|
wantedBy = [ "multi-user.target" ];
|
||||||
|
after = [ "sys-kernel-config.mount" "network.target" ];
|
||||||
|
requires = [ "sys-kernel-config.mount" ];
|
||||||
|
serviceConfig = {
|
||||||
|
Type = "oneshot";
|
||||||
|
RemainAfterExit = true;
|
||||||
|
ExecStart = "${targetctl} restore /etc/target/saveconfig.json";
|
||||||
|
ExecStop = "${targetctlStop}";
|
||||||
|
};
|
||||||
|
unitConfig.ConditionFileNotEmpty = "/etc/target/saveconfig.json";
|
||||||
|
};
|
||||||
|
tmpfiles.rules = [
|
||||||
|
"d /etc/target 0750 root root -"
|
||||||
|
"f /etc/target/saveconfig.json 0640 root root -"
|
||||||
|
];
|
||||||
|
};
|
||||||
|
|
||||||
|
environment.systemPackages = [ pkgs.targetcli-fb ];
|
||||||
|
}
|
||||||
@@ -0,0 +1,94 @@
|
|||||||
|
# Pacemaker + Corosync HA stack for NixOS with known-good workarounds.
|
||||||
|
#
|
||||||
|
# Issues fixed here (confirmed through live testing on NixOS 25.11):
|
||||||
|
#
|
||||||
|
# 1. StateDirectory ownership reset: systemd's StateDirectory=pacemaker
|
||||||
|
# creates /var/lib/pacemaker owned root:root. pacemaker-based (the CIB
|
||||||
|
# daemon) runs as the hacluster user and calls pcmk__daemon_can_write,
|
||||||
|
# which requires the CIB directory to be owned by hacluster or be
|
||||||
|
# group-writable by haclient. Workaround: remove StateDirectory and let
|
||||||
|
# ExecStartPre create every required subdirectory with correct ownership.
|
||||||
|
#
|
||||||
|
# 2. HA_SBIN_DIR wrong path: ocf-shellfuncs sets HA_SBIN_DIR to the Nix
|
||||||
|
# store path of the resource-agents derivation's /sbin, which doesn't
|
||||||
|
# exist. The DRBD OCF agent uses ${HA_SBIN_DIR}/crm_master, so it exits
|
||||||
|
# 127 without this override. Fix: export HA_SBIN_DIR=/run/current-system/sw/bin.
|
||||||
|
#
|
||||||
|
# 3. Broad PATH for OCF agents: the resource executor (pacemaker-execd) runs
|
||||||
|
# OCF agent scripts as children. NixOS provides no implicit PATH for
|
||||||
|
# system services; without an explicit PATH the agents can't find ip, ss,
|
||||||
|
# mount, umount, drbdadm, etc.
|
||||||
|
#
|
||||||
|
# 4. FUSER=true: the Filesystem OCF agent calls check_binary $FUSER (default:
|
||||||
|
# fuser from psmisc), which is not installed. Setting FUSER=true makes
|
||||||
|
# check_binary succeed (true is always in PATH) and the subsequent
|
||||||
|
# "$FUSER -km $mountpoint" becomes a no-op. Pair with force_unmount=false
|
||||||
|
# on each Filesystem resource unless you want lazy unmount behaviour.
|
||||||
|
{ lib, pkgs, ... }:
|
||||||
|
|
||||||
|
let
|
||||||
|
ocfBinPath = lib.concatStringsSep ":" [
|
||||||
|
"${pkgs.iproute2}/bin"
|
||||||
|
"${pkgs.iproute2}/sbin"
|
||||||
|
"${pkgs.iputils}/bin"
|
||||||
|
"${pkgs.util-linux}/bin"
|
||||||
|
"${pkgs.util-linux}/sbin"
|
||||||
|
"${pkgs.gawk}/bin"
|
||||||
|
"${pkgs.gnugrep}/bin"
|
||||||
|
"${pkgs.gnused}/bin"
|
||||||
|
"${pkgs.coreutils}/bin"
|
||||||
|
"${pkgs.bash}/bin"
|
||||||
|
"${pkgs.procps}/bin"
|
||||||
|
"${pkgs.xfsprogs}/bin"
|
||||||
|
"${pkgs.drbd}/bin"
|
||||||
|
"${pkgs.python3}/bin"
|
||||||
|
"/run/current-system/sw/bin"
|
||||||
|
"/run/current-system/sw/sbin"
|
||||||
|
"/usr/local/sbin"
|
||||||
|
"/usr/local/bin"
|
||||||
|
"/usr/sbin"
|
||||||
|
"/usr/bin"
|
||||||
|
"/sbin"
|
||||||
|
"/bin"
|
||||||
|
];
|
||||||
|
|
||||||
|
# Single pre-start script: schemas symlink + directory ownership.
|
||||||
|
# Runs before pacemakerd so pacemaker-based finds hacluster-owned dirs.
|
||||||
|
preStartCmd = "${pkgs.bash}/bin/bash -c '"
|
||||||
|
+ "ln -sfn ${pkgs.pacemaker}/share/pacemaker /var/lib/pacemaker/schemas; "
|
||||||
|
+ "for d in /var/lib/pacemaker /var/lib/pacemaker/cib /var/lib/pacemaker/cores "
|
||||||
|
+ "/var/lib/pacemaker/pengine /var/lib/pacemaker/blackbox "
|
||||||
|
+ "/var/lib/pacemaker/hostcache; do "
|
||||||
|
+ "mkdir -p \"\\$d\" && chown hacluster:pacemaker \"\\$d\" && chmod 2770 \"\\$d\"; "
|
||||||
|
+ "done'";
|
||||||
|
|
||||||
|
ocfEnv = {
|
||||||
|
PATH = lib.mkForce ocfBinPath;
|
||||||
|
OCF_ROOT = "${pkgs.ocf-resource-agents}/usr/lib/ocf";
|
||||||
|
HA_SBIN_DIR = "/run/current-system/sw/bin";
|
||||||
|
FUSER = "true";
|
||||||
|
};
|
||||||
|
in
|
||||||
|
{
|
||||||
|
users.groups.haclient = { };
|
||||||
|
|
||||||
|
services.corosync.enable = true;
|
||||||
|
services.pacemaker.enable = true;
|
||||||
|
|
||||||
|
systemd.services = {
|
||||||
|
pacemaker = {
|
||||||
|
serviceConfig = {
|
||||||
|
StateDirectory = lib.mkForce "";
|
||||||
|
ExecStartPre = lib.mkBefore [ preStartCmd ];
|
||||||
|
};
|
||||||
|
environment = ocfEnv;
|
||||||
|
};
|
||||||
|
pacemaker-execd.environment = ocfEnv;
|
||||||
|
};
|
||||||
|
|
||||||
|
environment.systemPackages = with pkgs; [
|
||||||
|
corosync
|
||||||
|
pacemaker
|
||||||
|
ocf-resource-agents
|
||||||
|
];
|
||||||
|
}
|
||||||
@@ -0,0 +1,23 @@
|
|||||||
|
# Adapted from the output of `nixos-generate-config`, run from a live GUI
|
||||||
|
# ISO boot on the actual gui-host hardware (AMD CPU). fileSystems and
|
||||||
|
# swapDevices are deliberately omitted -- the live ISO had no formatted
|
||||||
|
# disks to detect, and disko (modules/disko/baremetal.nix) generates both
|
||||||
|
# from the declarative zpool layout anyway.
|
||||||
|
{ config, lib, pkgs, modulesPath, ... }:
|
||||||
|
|
||||||
|
{
|
||||||
|
imports =
|
||||||
|
[
|
||||||
|
(modulesPath + "/installer/scan/not-detected.nix")
|
||||||
|
];
|
||||||
|
|
||||||
|
boot = {
|
||||||
|
initrd.availableKernelModules = [ "xhci_pci" "ahci" "usbhid" "usb_storage" "sd_mod" ];
|
||||||
|
initrd.kernelModules = [ ];
|
||||||
|
kernelModules = [ "kvm-amd" ];
|
||||||
|
extraModulePackages = [ ];
|
||||||
|
};
|
||||||
|
|
||||||
|
nixpkgs.hostPlatform = lib.mkDefault "x86_64-linux";
|
||||||
|
hardware.cpu.amd.updateMicrocode = lib.mkDefault config.hardware.enableRedistributableFirmware;
|
||||||
|
}
|
||||||
@@ -5,36 +5,39 @@
|
|||||||
|
|
||||||
{
|
{
|
||||||
imports =
|
imports =
|
||||||
[ (modulesPath + "/profiles/qemu-guest.nix")
|
[
|
||||||
|
(modulesPath + "/profiles/qemu-guest.nix")
|
||||||
];
|
];
|
||||||
|
|
||||||
boot.initrd.availableKernelModules = [ "virtio_pci" "virtio_scsi" "ahci" "sd_mod" ];
|
boot = {
|
||||||
boot.initrd.kernelModules = [ ];
|
initrd.availableKernelModules = [ "virtio_pci" "virtio_scsi" "ahci" "sd_mod" ];
|
||||||
boot.kernelModules = [ ];
|
initrd.kernelModules = [ ];
|
||||||
boot.extraModulePackages = [ ];
|
kernelModules = [ ];
|
||||||
boot.loader.grub.device = "/dev/sda";
|
extraModulePackages = [ ];
|
||||||
|
|
||||||
fileSystems."/" =
|
|
||||||
{ device = "/dev/sda";
|
|
||||||
fsType = "ext4";
|
|
||||||
};
|
|
||||||
|
|
||||||
swapDevices =
|
|
||||||
[ { device = "/dev/sdb"; }
|
|
||||||
];
|
|
||||||
|
|
||||||
# Enable LISH
|
# Enable LISH
|
||||||
boot.kernelParams = [ "console=ttyS0,19200n8" ];
|
kernelParams = [ "console=ttyS0,19200n8" ];
|
||||||
boot.loader.grub.extraConfig = ''
|
|
||||||
|
loader = {
|
||||||
|
grub = {
|
||||||
|
device = "/dev/sda";
|
||||||
|
extraConfig = ''
|
||||||
serial --speed=19200 --unit=0 --word=8 --parity=no --stop=1;
|
serial --speed=19200 --unit=0 --word=8 --parity=no --stop=1;
|
||||||
terminal_input serial;
|
terminal_input serial;
|
||||||
terminal_output serial;
|
terminal_output serial;
|
||||||
'';
|
'';
|
||||||
|
forceInstall = true;
|
||||||
|
# device = "nodev";
|
||||||
|
};
|
||||||
|
timeout = 10;
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
boot.loader.grub.forceInstall = true;
|
# fileSystems."/" and swapDevices are now owned by disko
|
||||||
# boot.loader.grub.device = "nodev";
|
# (../disko/linode.nix, imported from ../platforms/linode.nix) — same
|
||||||
boot.loader.timeout = 10;
|
# /dev/sda root + /dev/sdb swap layout, declared there instead so disko's
|
||||||
|
# (idempotent, non-destructive — see that file) format/mount scripts stay
|
||||||
|
# in sync with what NixOS actually mounts.
|
||||||
|
|
||||||
nixpkgs.hostPlatform = lib.mkDefault "x86_64-linux";
|
nixpkgs.hostPlatform = lib.mkDefault "x86_64-linux";
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -5,13 +5,16 @@
|
|||||||
|
|
||||||
{
|
{
|
||||||
imports =
|
imports =
|
||||||
[ (modulesPath + "/profiles/qemu-guest.nix")
|
[
|
||||||
|
(modulesPath + "/profiles/qemu-guest.nix")
|
||||||
];
|
];
|
||||||
|
|
||||||
boot.initrd.availableKernelModules = [ "ata_piix" "uhci_hcd" "virtio_pci" "virtio_scsi" "sd_mod" "sr_mod" ];
|
boot = {
|
||||||
boot.initrd.kernelModules = [ ];
|
initrd.availableKernelModules = [ "ata_piix" "uhci_hcd" "virtio_pci" "virtio_scsi" "sd_mod" "sr_mod" ];
|
||||||
boot.kernelModules = [ "kvm-amd" ];
|
initrd.kernelModules = [ ];
|
||||||
boot.extraModulePackages = [ ];
|
kernelModules = [ "kvm-amd" ];
|
||||||
|
extraModulePackages = [ ];
|
||||||
|
};
|
||||||
# boot.loader.grub.device = "/dev/sda2"; # or "nodev" for efi only
|
# boot.loader.grub.device = "/dev/sda2"; # or "nodev" for efi only
|
||||||
|
|
||||||
# fileSystems."/" =
|
# fileSystems."/" =
|
||||||
|
|||||||
@@ -0,0 +1,121 @@
|
|||||||
|
{ pkgs, lib, vars, ... }:
|
||||||
|
|
||||||
|
{
|
||||||
|
imports = [
|
||||||
|
./host-keys.nix
|
||||||
|
];
|
||||||
|
|
||||||
|
networking.useDHCP = lib.mkDefault true;
|
||||||
|
|
||||||
|
# Recommended over the true default (bypasses ZFS's own import safeguards)
|
||||||
|
# per the option's own docs. This installer environment has no ZFS pools
|
||||||
|
# of its own to import, so this is a no-op here — just silences the
|
||||||
|
# eval-time warning, matching modules/common/configuration.nix.
|
||||||
|
boot.zfs.forceImportRoot = false;
|
||||||
|
|
||||||
|
time.timeZone = vars.timeZone;
|
||||||
|
|
||||||
|
# Without this, the installer only ever sees cache.nixos.org, which
|
||||||
|
# doesn't carry sops-install-secrets (it's built straight from the
|
||||||
|
# sops-nix flake's own Go source, not part of nixpkgs) — every install
|
||||||
|
# would otherwise compile it from scratch, which is what ran an 8GB LXC
|
||||||
|
# container's disk out of space. Push a built copy to nix-cache once
|
||||||
|
# (from a machine with real disk headroom) and every future install,
|
||||||
|
# of any type, fetches instead of rebuilding.
|
||||||
|
nix.settings = {
|
||||||
|
substituters = [
|
||||||
|
"http://nix-cache"
|
||||||
|
"https://cache.nixos.org/"
|
||||||
|
];
|
||||||
|
trusted-public-keys = [
|
||||||
|
"cache.local-1:usoWYanY3Kpq2+kDIS2nhWoLZiRxanmdysdzqCFBHW4="
|
||||||
|
"cache.nixos.org-1:6NCHdD59X431o0gWypbMrAURkbJ16ZPMQFGspcDShjY="
|
||||||
|
];
|
||||||
|
};
|
||||||
|
|
||||||
|
environment = {
|
||||||
|
systemPackages = with pkgs; [
|
||||||
|
git
|
||||||
|
curl
|
||||||
|
jq
|
||||||
|
parted
|
||||||
|
e2fsprogs
|
||||||
|
btrfs-progs
|
||||||
|
util-linux
|
||||||
|
disko
|
||||||
|
];
|
||||||
|
|
||||||
|
# Auto-install script, kept as a real, version-controlled shell file at
|
||||||
|
# scripts/installer/auto-install.sh rather than an inline Nix string.
|
||||||
|
# It sources scripts/env.sh itself (for LAN_DOMAIN, same as every other
|
||||||
|
# script in this repo) rather than relying on Nix-level templating, so
|
||||||
|
# it behaves identically whether it's run straight from a git checkout
|
||||||
|
# or from here -- baking scripts/env.sh in alongside it at a matching
|
||||||
|
# relative path (installer/auto-install.sh -> ../env.sh) is what makes
|
||||||
|
# that resolve correctly in both places.
|
||||||
|
etc = {
|
||||||
|
"nixos-installer/env.sh".source = ../../scripts/env.sh;
|
||||||
|
|
||||||
|
"nixos-installer/installer/auto-install.sh" = {
|
||||||
|
source = ../../scripts/installer/auto-install.sh;
|
||||||
|
mode = "0755";
|
||||||
|
};
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
programs.git.enable = true;
|
||||||
|
|
||||||
|
# Run the installer on first login. Previously this copied an /etc file
|
||||||
|
# into the nixos user's ~/.bash_profile via an activation script that
|
||||||
|
# got dropped in a refactor (and only ever worked for that one user
|
||||||
|
# anyway) — loginShellInit is NixOS's native hook for this, applies to
|
||||||
|
# any user's login shell (root included), and needs no home-directory
|
||||||
|
# file-copying/chown.
|
||||||
|
programs.bash.loginShellInit = ''
|
||||||
|
if [ -n "$PS1" ] && [ ! -e "$HOME/.auto_install_ran" ]; then
|
||||||
|
sudo /etc/nixos-installer/installer/auto-install.sh
|
||||||
|
touch "$HOME/.auto_install_ran"
|
||||||
|
fi
|
||||||
|
'';
|
||||||
|
|
||||||
|
services.openssh.enable = true;
|
||||||
|
|
||||||
|
services.openssh.settings = {
|
||||||
|
PermitRootLogin = "yes";
|
||||||
|
PasswordAuthentication = true;
|
||||||
|
};
|
||||||
|
|
||||||
|
# nixpkgs' own installer profile (profiles/installation-device.nix, pulled
|
||||||
|
# in via installation-cd-minimal.nix) sets initialHashedPassword = "" for
|
||||||
|
# both users — its own passwordless-login convention for install media.
|
||||||
|
# That's a second, non-null password option alongside our hashedPassword
|
||||||
|
# below, which NixOS warns about as ambiguous precedence. Force it null
|
||||||
|
# rather than adopting passwordless login: this image now also boots over
|
||||||
|
# LAN PXE with PasswordAuthentication enabled, so passwordless root SSH
|
||||||
|
# would be reachable by anyone on the LAN, not just local console.
|
||||||
|
users.users.root = {
|
||||||
|
hashedPassword =
|
||||||
|
"$6$Kwv9KAyvcurAViQF$H4.u3feqGE7lVoNgkFXhE3n2Pmo//9JYDTCz8ifrVHBxPjwa1xMby7tEZ8Bpt5MXs9Rkx6/YbZWxs5CpH0s/70";
|
||||||
|
initialHashedPassword = lib.mkForce null;
|
||||||
|
};
|
||||||
|
|
||||||
|
users.users.${vars.primaryUser} = {
|
||||||
|
isNormalUser = true;
|
||||||
|
|
||||||
|
extraGroups = [
|
||||||
|
"wheel"
|
||||||
|
];
|
||||||
|
|
||||||
|
shell = pkgs.bashInteractive;
|
||||||
|
|
||||||
|
hashedPassword =
|
||||||
|
"$6$Kwv9KAyvcurAViQF$H4.u3feqGE7lVoNgkFXhE3n2Pmo//9JYDTCz8ifrVHBxPjwa1xMby7tEZ8Bpt5MXs9Rkx6/YbZWxs5CpH0s/70";
|
||||||
|
initialHashedPassword = lib.mkForce null;
|
||||||
|
|
||||||
|
openssh.authorizedKeys.keys = [
|
||||||
|
vars.adminSshKey
|
||||||
|
];
|
||||||
|
};
|
||||||
|
|
||||||
|
system.stateVersion = "26.05";
|
||||||
|
}
|
||||||
@@ -0,0 +1,37 @@
|
|||||||
|
{ lib, ... }:
|
||||||
|
|
||||||
|
let
|
||||||
|
# host-keys/ is gitignored (private key material must never be committed),
|
||||||
|
# which means flakes' git-filtered source tree can never see it via a
|
||||||
|
# normal relative path — referencing it at all requires stepping outside
|
||||||
|
# pure evaluation. builtins.getEnv is neutered to "" under normal
|
||||||
|
# `nix build`/`nix eval` (no error, just empty), so this whole module is a
|
||||||
|
# silent no-op unless the operator explicitly opts in with --impure and
|
||||||
|
# the env var set — safe by default, including in CI.
|
||||||
|
#
|
||||||
|
# NIXOS_HOST_KEYS_DIR=$(pwd)/host-keys nix build .#iso --impure
|
||||||
|
#
|
||||||
|
# See docs/auto-installer.md.
|
||||||
|
hostKeysDirStr = builtins.getEnv "NIXOS_HOST_KEYS_DIR";
|
||||||
|
hasHostKeysDir = hostKeysDirStr != "" && builtins.pathExists hostKeysDirStr;
|
||||||
|
hostKeysDir = /. + hostKeysDirStr;
|
||||||
|
|
||||||
|
keyFileNames =
|
||||||
|
if hasHostKeysDir
|
||||||
|
then
|
||||||
|
lib.filter
|
||||||
|
(name: lib.hasSuffix "_ssh_host_ed25519_key" name || lib.hasSuffix "_ssh_host_ed25519_key.pub" name)
|
||||||
|
(lib.attrNames (builtins.readDir hostKeysDir))
|
||||||
|
else [ ];
|
||||||
|
in
|
||||||
|
{
|
||||||
|
environment.etc = lib.listToAttrs (map
|
||||||
|
(name: {
|
||||||
|
name = "host-keys/${name}";
|
||||||
|
value = {
|
||||||
|
source = hostKeysDir + "/${name}";
|
||||||
|
mode = "0400";
|
||||||
|
};
|
||||||
|
})
|
||||||
|
keyFileNames);
|
||||||
|
}
|
||||||
@@ -0,0 +1,8 @@
|
|||||||
|
{ modulesPath, ... }:
|
||||||
|
|
||||||
|
{
|
||||||
|
imports = [
|
||||||
|
"${modulesPath}/installer/cd-dvd/installation-cd-minimal.nix"
|
||||||
|
./common.nix
|
||||||
|
];
|
||||||
|
}
|
||||||
@@ -0,0 +1,210 @@
|
|||||||
|
# Fully declarative FreeIPA domain membership.
|
||||||
|
#
|
||||||
|
# Imported by modules/common/configuration.nix — no per-host wiring needed.
|
||||||
|
# Enables itself automatically on any host that has a sops-encrypted keytab
|
||||||
|
# at secrets/<hostname>.keytab; is a no-op for all other hosts.
|
||||||
|
#
|
||||||
|
# To enroll a new host:
|
||||||
|
# 0. scripts/secrets/sync-host-keys.sh <flake-target>
|
||||||
|
# 1. scripts/ipa/create-nixos-ipa-host-account.sh [--ip <addr>] <hostname>
|
||||||
|
# (adds .sops.yaml rule, runs ipa host-add, encrypts keytab in one step)
|
||||||
|
# 2. git add secrets/<hostname>.keytab .sops.yaml && git commit
|
||||||
|
# 3. Deploy — no further steps required.
|
||||||
|
#
|
||||||
|
# Manual fallback (if the script isn't usable):
|
||||||
|
# a. On the FreeIPA server: ipa host-add <fqdn> [--ip-address=<ip>] --force
|
||||||
|
# b. On the FreeIPA server: ipa-getkeytab -s <ipa-server> -p host/<fqdn> -k /tmp/<host>.keytab
|
||||||
|
# c. From the repo root (path must match for sops creation rule to apply):
|
||||||
|
# cp /tmp/<host>.keytab secrets/<host>.keytab
|
||||||
|
# sops -e --input-type binary -i secrets/<host>.keytab
|
||||||
|
# d. Commit secrets/<host>.keytab and the updated .sops.yaml, then deploy.
|
||||||
|
#
|
||||||
|
# vars dependencies: homeDomain, ipaServer, domainControllerIp, ipaUser
|
||||||
|
|
||||||
|
{ config, lib, pkgs, vars, ... }:
|
||||||
|
|
||||||
|
let
|
||||||
|
keytabPath = ../../secrets + "/${config.networking.hostName}.keytab";
|
||||||
|
enabled = builtins.pathExists keytabPath;
|
||||||
|
|
||||||
|
realm = lib.strings.toUpper vars.homeDomain;
|
||||||
|
fqdn = "${config.networking.hostName}.${vars.homeDomain}";
|
||||||
|
# "sweet.home" -> "dc=sweet,dc=home"
|
||||||
|
basedn = lib.strings.concatMapStringsSep "," (c: "dc=${c}") (lib.strings.splitString "." vars.homeDomain);
|
||||||
|
# security.ipa.certificate expects a derivation (package), not a raw path.
|
||||||
|
caCertPkg = pkgs.writeText "ipa-ca.crt" (builtins.readFile ../../certs/ipa-ca.crt);
|
||||||
|
in
|
||||||
|
lib.mkIf enabled {
|
||||||
|
networking.domain = lib.mkDefault vars.homeDomain;
|
||||||
|
networking.nameservers = lib.mkDefault [ vars.domainControllerIp ];
|
||||||
|
|
||||||
|
security = {
|
||||||
|
ipa = {
|
||||||
|
enable = true;
|
||||||
|
domain = vars.homeDomain;
|
||||||
|
inherit realm;
|
||||||
|
server = vars.ipaServer;
|
||||||
|
certificate = caCertPkg;
|
||||||
|
inherit basedn;
|
||||||
|
ipaHostname = fqdn;
|
||||||
|
offlinePasswords = true;
|
||||||
|
cacheCredentials = true;
|
||||||
|
};
|
||||||
|
|
||||||
|
# Create the home directory on first login if it doesn't exist yet.
|
||||||
|
# IPA users have no pre-created home on the host; without this sshd
|
||||||
|
# opens a session to a non-existent directory and resets the connection.
|
||||||
|
# lightdm also needs this so the GUI login path can create the home dir
|
||||||
|
# if it was not pre-seeded by the tmpfiles rule above (e.g. on first boot
|
||||||
|
# before SSSD has resolved the user).
|
||||||
|
pam.services = {
|
||||||
|
sshd.makeHomeDir = true;
|
||||||
|
lightdm.makeHomeDir = true;
|
||||||
|
|
||||||
|
# pam_unix returns PAM_AUTHINFO_UNAVAIL without prompting when the local
|
||||||
|
# stub has "!" in shadow (account locked), so PAM_AUTHTOK is never set
|
||||||
|
# and pam_sss's use_first_pass fails with "No authentication token".
|
||||||
|
# Changing to try_first_pass makes pam_sss prompt independently when no
|
||||||
|
# prior module has set the token, restoring IPA password login via
|
||||||
|
# LightDM and su.
|
||||||
|
login.rules.auth.sss.settings = lib.mkForce { try_first_pass = true; };
|
||||||
|
su.rules.auth.sss.settings = lib.mkForce { try_first_pass = true; };
|
||||||
|
};
|
||||||
|
|
||||||
|
# HM with useUserPackages = true (flake.nix) sets users.users.${ipaUser}.packages,
|
||||||
|
# which forces the stub into /etc/passwd. pam_sss.so with the "localusers" flag
|
||||||
|
# (added by NixOS when SSSD is enabled) then skips SSSD for any user it finds in
|
||||||
|
# local /etc/passwd — including this stub — falling through to pam_unix, which has
|
||||||
|
# no password for the stub → sudo auth always fails.
|
||||||
|
#
|
||||||
|
# Fix: NOPASSWD for the IPA user. The IPA user already authenticated to reach a
|
||||||
|
# shell (SSH public key from IPA or Kerberos), so re-prompting via a broken PAM
|
||||||
|
# path is security theater on a single-admin homelab.
|
||||||
|
sudo.extraRules = [{
|
||||||
|
users = [ vars.ipaUser ];
|
||||||
|
commands = [{ command = "ALL"; options = [ "NOPASSWD" ]; }];
|
||||||
|
}];
|
||||||
|
};
|
||||||
|
|
||||||
|
systemd = {
|
||||||
|
# Fetch SSH public keys from IPA so users can log in with the key stored
|
||||||
|
# in their IPA profile rather than needing ~/.ssh/authorized_keys on every
|
||||||
|
# host. sss_ssh_authorizedkeys queries SSSD (which queries IPA LDAP).
|
||||||
|
#
|
||||||
|
# /nix/store is 1775 (group-writable by nixbld). OpenSSH 10.0+ rejects
|
||||||
|
# AuthorizedKeysCommand binaries whose path contains any group-writable
|
||||||
|
# component, silently skipping the command. Copy to /usr/local/bin (all
|
||||||
|
# components root-owned, 755) so the path passes sshd's safety check.
|
||||||
|
tmpfiles.rules = [
|
||||||
|
"d /usr/local 0755 root root - -"
|
||||||
|
"d /usr/local/bin 0755 root root - -"
|
||||||
|
"C+ /usr/local/bin/sss_ssh_authorizedkeys 0555 root root - ${pkgs.sssd}/bin/sss_ssh_authorizedkeys"
|
||||||
|
# Pre-create the IPA user's home dir so Home Manager activation succeeds
|
||||||
|
# even before their first login. On a fresh system SSSD may not have
|
||||||
|
# resolved the user yet — tmpfiles warns and skips in that case (non-fatal),
|
||||||
|
# and pam_mkhomedir covers the first-login path as a fallback.
|
||||||
|
"d /home/${vars.ipaUser} 0700 ${vars.ipaUser} ${vars.ipaUser} - -"
|
||||||
|
];
|
||||||
|
|
||||||
|
# security.ipa enables Kerberos (security.krb5) which causes systemd to
|
||||||
|
# start auth-rpcgss-module.service and rpc-gssd.service for Kerberos NFS
|
||||||
|
# authentication. LXC containers can't load the auth_rpcgss kernel module
|
||||||
|
# and don't have /var/lib/nfs/rpc_pipefs, so both services fail.
|
||||||
|
#
|
||||||
|
# The NixOS IPA module already adds a drop-in for auth-rpcgss-module.service
|
||||||
|
# with ConditionPathExists=/etc/krb5.keytab. We use lib.mkForce to win the
|
||||||
|
# text conflict and add ConditionVirtualization=!container alongside it so
|
||||||
|
# the service is skipped (not failed) in containers that do have a keytab.
|
||||||
|
# Same fix for rpc-gssd.service which also fails in containers.
|
||||||
|
units = lib.mkIf config.boot.isContainer {
|
||||||
|
"auth-rpcgss-module.service" = {
|
||||||
|
overrideStrategy = "asDropinIfExists";
|
||||||
|
text = lib.mkForce ''
|
||||||
|
[Unit]
|
||||||
|
ConditionPathExists=
|
||||||
|
ConditionPathExists=/etc/krb5.keytab
|
||||||
|
ConditionVirtualization=!container
|
||||||
|
'';
|
||||||
|
};
|
||||||
|
# rpc-gssd also has ConditionPathExists from the NixOS IPA module (and an
|
||||||
|
# X-Restart-Triggers store path from systemd.nix). Use mkForce to win;
|
||||||
|
# omit X-Restart-Triggers since this service is skipped in containers anyway.
|
||||||
|
"rpc-gssd.service" = {
|
||||||
|
overrideStrategy = "asDropinIfExists";
|
||||||
|
text = lib.mkForce ''
|
||||||
|
[Unit]
|
||||||
|
ConditionPathExists=
|
||||||
|
ConditionPathExists=/etc/krb5.keytab
|
||||||
|
ConditionVirtualization=!container
|
||||||
|
'';
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
# home-manager-<user>.service fails on first enrollment because /home/wayne
|
||||||
|
# doesn't exist until the user's first login (pam_mkhomedir creates it then).
|
||||||
|
# ConditionPathExists makes systemd skip the service (exit 0, condition not
|
||||||
|
# met) instead of failing. After first login the dir exists and subsequent
|
||||||
|
# rebuilds activate HM normally.
|
||||||
|
services."home-manager-${vars.ipaUser}".unitConfig.ConditionPathExists =
|
||||||
|
"/home/${vars.ipaUser}";
|
||||||
|
};
|
||||||
|
|
||||||
|
services.openssh.extraConfig = ''
|
||||||
|
AuthorizedKeysCommand /usr/local/bin/sss_ssh_authorizedkeys %u
|
||||||
|
AuthorizedKeysCommandUser nobody
|
||||||
|
'';
|
||||||
|
|
||||||
|
# Host keytab: pre-provisioned on the IPA server, sops-encrypted binary.
|
||||||
|
# Placed at /etc/krb5.keytab before SSSD starts so the host authenticates
|
||||||
|
# to IPA without running ipa-client-install.
|
||||||
|
sops.secrets."ipa-host-keytab" = {
|
||||||
|
sopsFile = keytabPath;
|
||||||
|
format = "binary";
|
||||||
|
path = "/etc/krb5.keytab";
|
||||||
|
owner = "root";
|
||||||
|
group = "root";
|
||||||
|
mode = "0600";
|
||||||
|
restartUnits = [ "sssd.service" ];
|
||||||
|
};
|
||||||
|
|
||||||
|
# NixOS requires isNormalUser/isSystemUser + group on any entry in
|
||||||
|
# users.users. HM with useUserPackages = true (set in flake.nix) adds a stub
|
||||||
|
# entry for each HM user so it can install packages to
|
||||||
|
# /etc/profiles/per-user/<name>/. This definition satisfies those assertions.
|
||||||
|
# With security.ipa setting "passwd: sss files" in nsswitch, SSSD's IPA entry
|
||||||
|
# takes priority for NSS lookups — this local stub is only a fallback when
|
||||||
|
# SSSD is unreachable (at which point auth fails anyway).
|
||||||
|
users.users.${vars.ipaUser} = {
|
||||||
|
isNormalUser = true;
|
||||||
|
group = "users";
|
||||||
|
extraGroups = [ "wheel" ];
|
||||||
|
createHome = false;
|
||||||
|
# "!" is not a password hash — it is the standard "account locked" marker.
|
||||||
|
# It cannot authenticate anyone locally. It exists solely so NixOS generates
|
||||||
|
# a shadow entry for this stub user; without one pam_unix returns
|
||||||
|
# PAM_AUTHINFO_UNAVAIL before prompting, which means PAM_AUTHTOK is never
|
||||||
|
# set and the subsequent pam_sss use_first_pass call has nothing to work
|
||||||
|
# with — blocking LightDM and su logins even when IPA/SSSD auth succeeds.
|
||||||
|
hashedPassword = "!";
|
||||||
|
};
|
||||||
|
|
||||||
|
# Home Manager config for the IPA primary user, applied on every enrolled
|
||||||
|
# host. Manages what IPA doesn't: dotfiles, user-scoped packages, session
|
||||||
|
# variables. Switch-nix/Test-nix/buildImage are system-wide (configuration.nix)
|
||||||
|
# so they don't need to be repeated here.
|
||||||
|
#
|
||||||
|
# homeDirectory uses mkForce because HM's NixOS integration module sets it to
|
||||||
|
# "/var/empty" for users not found in config.users.users at eval time (SSSD
|
||||||
|
# users aren't visible there).
|
||||||
|
home-manager.users.${vars.ipaUser} = { pkgs, ... }: {
|
||||||
|
home = {
|
||||||
|
username = vars.ipaUser;
|
||||||
|
homeDirectory = lib.mkForce "/home/${vars.ipaUser}";
|
||||||
|
stateVersion = "26.05";
|
||||||
|
packages = with pkgs; [ tmux sshfs ];
|
||||||
|
sessionVariables.EDITOR = lib.mkDefault "nano";
|
||||||
|
};
|
||||||
|
programs.home-manager.enable = true;
|
||||||
|
programs.bash.enable = true;
|
||||||
|
};
|
||||||
|
}
|
||||||
@@ -0,0 +1,43 @@
|
|||||||
|
{ config, lib, vars, ... }:
|
||||||
|
|
||||||
|
{
|
||||||
|
# Prestages a NetworkManager connection profile for vars.wifiSsid so the
|
||||||
|
# host associates on first boot with no manual nmtui/nmcli step. Guarded
|
||||||
|
# on a non-empty SSID so leaving the placeholder blank in variables.nix
|
||||||
|
# is a no-op rather than an empty, broken profile — fill it in once the
|
||||||
|
# network is known.
|
||||||
|
#
|
||||||
|
# The password itself lives in secrets/gui.yaml, not variables.nix --
|
||||||
|
# NetworkManager's ensureProfiles renders `psk = "$WIFI_PASSWORD"`
|
||||||
|
# literally into the store (see nixpkgs' own ensureProfiles example,
|
||||||
|
# which does the same for exactly this reason) and its systemd service
|
||||||
|
# envsubst-expands it from environmentFiles at activation time, so the
|
||||||
|
# real value only ever touches /run (root-only, UMask 0177), never the
|
||||||
|
# Nix store.
|
||||||
|
sops.secrets."wifi-password" = lib.mkIf (vars.wifiSsid != "") {
|
||||||
|
sopsFile = ../../secrets/gui.yaml;
|
||||||
|
};
|
||||||
|
|
||||||
|
sops.templates."wifi-password.env" = lib.mkIf (vars.wifiSsid != "") {
|
||||||
|
content = "WIFI_PASSWORD=${config.sops.placeholder."wifi-password"}";
|
||||||
|
};
|
||||||
|
|
||||||
|
networking.networkmanager.ensureProfiles = lib.mkIf (vars.wifiSsid != "") {
|
||||||
|
environmentFiles = [ config.sops.templates."wifi-password.env".path ];
|
||||||
|
|
||||||
|
profiles.${vars.wifiSsid} = {
|
||||||
|
connection = {
|
||||||
|
id = vars.wifiSsid;
|
||||||
|
type = "wifi";
|
||||||
|
};
|
||||||
|
wifi = {
|
||||||
|
mode = "infrastructure";
|
||||||
|
ssid = vars.wifiSsid;
|
||||||
|
};
|
||||||
|
wifi-security = {
|
||||||
|
key-mgmt = "wpa-psk";
|
||||||
|
psk = "$WIFI_PASSWORD";
|
||||||
|
};
|
||||||
|
};
|
||||||
|
};
|
||||||
|
}
|
||||||
@@ -1,9 +1,9 @@
|
|||||||
{ ... }:
|
{ vars, ... }:
|
||||||
|
|
||||||
{
|
{
|
||||||
nix.settings = {
|
nix.settings = {
|
||||||
substituters = [
|
substituters = [
|
||||||
"http://nix-cache"
|
"http://${vars.nixCacheHost}.${vars.homeDomain}"
|
||||||
"https://cache.nixos.org/"
|
"https://cache.nixos.org/"
|
||||||
];
|
];
|
||||||
trusted-public-keys = [
|
trusted-public-keys = [
|
||||||
|
|||||||
@@ -0,0 +1,40 @@
|
|||||||
|
{ pkgs, vars, ... }:
|
||||||
|
|
||||||
|
{
|
||||||
|
# Authenticate as nixremote using the client host's own default root SSH
|
||||||
|
# identity (/root/.ssh/id_ed25519) rather than a separately-named key --
|
||||||
|
# matches vars.remoteBuilderAuthorizedKeys, which already authorizes
|
||||||
|
# each host's own default key (one entry per host, not a shared
|
||||||
|
# dedicated keypair). If this host doesn't have one yet:
|
||||||
|
# sudo -u root ssh-keygen -t ed25519 -N '' -f /root/.ssh/id_ed25519
|
||||||
|
# # then add its .pub to vars.remoteBuilderAuthorizedKeys and rebuild nix-cache
|
||||||
|
# sudo ssh -i /root/.ssh/id_ed25519 nixremote@nix-cache.sweet.home nix-store --version
|
||||||
|
# Trust nix-cache's SSH host key declaratively so the nix-daemon (root)
|
||||||
|
# can connect the first time without a manual ssh-keyscan/known_hosts
|
||||||
|
# step on every new client.
|
||||||
|
programs.ssh.knownHosts."${vars.nixCacheHost}.${vars.homeDomain}" = {
|
||||||
|
hostNames = [ "${vars.nixCacheHost}.${vars.homeDomain}" ];
|
||||||
|
publicKey = vars.nixCacheHostKey;
|
||||||
|
};
|
||||||
|
|
||||||
|
nix = {
|
||||||
|
distributedBuilds = true;
|
||||||
|
|
||||||
|
buildMachines = [
|
||||||
|
{
|
||||||
|
hostName = "${vars.nixCacheHost}.${vars.homeDomain}";
|
||||||
|
sshUser = vars.remoteBuilderUser;
|
||||||
|
sshKey = "/root/.ssh/id_ed25519";
|
||||||
|
inherit (pkgs.stdenv.hostPlatform) system;
|
||||||
|
maxJobs = 4;
|
||||||
|
speedFactor = 2;
|
||||||
|
supportedFeatures = [ "nixos-test" "benchmark" "big-parallel" "kvm" ];
|
||||||
|
}
|
||||||
|
];
|
||||||
|
|
||||||
|
settings = {
|
||||||
|
builders-use-substitutes = true;
|
||||||
|
max-jobs = "auto";
|
||||||
|
};
|
||||||
|
};
|
||||||
|
}
|
||||||
@@ -1,54 +1,53 @@
|
|||||||
{ config, pkgs, ... }:
|
{ config, pkgs, vars, ... }:
|
||||||
|
|
||||||
{
|
{
|
||||||
# Generate the binary cache key pair on the nix-cache host:
|
# nix-serve's signing key has to be the *same* key on every host that
|
||||||
# sudo install -d -m 0700 /etc/nix
|
# ever plays the nix-cache role -- modules/nix-cache/client.nix hardcodes
|
||||||
# sudo nix-store --generate-binary-cache-key nix-cache-1 \
|
# every client's trust in one specific public key ("cache.local-1:..."),
|
||||||
# /etc/nix/cache-priv.pem \
|
# so a freshly self-generated key here wouldn't be trusted by anyone.
|
||||||
# /etc/nix/cache-pub.pem
|
# Managed via sops-nix like every other secret in this repo instead of
|
||||||
# sudo chmod 0600 /etc/nix/cache-priv.pem
|
# the old manual `nix-store --generate-binary-cache-key` step -- see
|
||||||
# sudo chmod 0644 /etc/nix/cache-pub.pem
|
# "Binary cache signing key" in docs/nix-cache.md for how to add/rotate
|
||||||
# cat /etc/nix/cache-pub.pem
|
# the value in secrets/nix-cache.yaml.
|
||||||
services.nix-serve = {
|
sops.secrets."cache-priv-key".sopsFile = ../../secrets/nix-cache.yaml;
|
||||||
|
|
||||||
|
services = {
|
||||||
|
nix-serve = {
|
||||||
enable = true;
|
enable = true;
|
||||||
secretKeyFile = "/etc/nix/cache-priv.pem";
|
secretKeyFile = config.sops.secrets."cache-priv-key".path;
|
||||||
};
|
};
|
||||||
|
|
||||||
services.nginx = {
|
nginx = {
|
||||||
enable = true;
|
enable = true;
|
||||||
recommendedProxySettings = true;
|
recommendedProxySettings = true;
|
||||||
virtualHosts."nix-cache" = {
|
virtualHosts."${vars.nixCacheHost}.${vars.homeDomain}" = {
|
||||||
locations."/" = {
|
locations."/" = {
|
||||||
proxyPass = "http://${config.services.nix-serve.bindAddress}:${toString config.services.nix-serve.port}";
|
proxyPass = "http://${config.services.nix-serve.bindAddress}:${toString config.services.nix-serve.port}";
|
||||||
};
|
};
|
||||||
};
|
};
|
||||||
};
|
};
|
||||||
|
|
||||||
networking.firewall.allowedTCPPorts = [ 80 ];
|
openssh.enable = true;
|
||||||
|
};
|
||||||
|
|
||||||
users.groups.nixremote = {};
|
networking.firewall.allowedTCPPorts = [ vars.ports.nixCacheHttp ];
|
||||||
|
|
||||||
users.users.nixremote = {
|
users.groups.${vars.remoteBuilderUser} = { };
|
||||||
|
|
||||||
|
users.users.${vars.remoteBuilderUser} = {
|
||||||
isSystemUser = true;
|
isSystemUser = true;
|
||||||
group = "nixremote";
|
group = vars.remoteBuilderUser;
|
||||||
createHome = true;
|
createHome = true;
|
||||||
home = "/var/lib/nixremote";
|
home = "/var/lib/nixremote";
|
||||||
shell = pkgs.bashInteractive;
|
shell = pkgs.bashInteractive;
|
||||||
# Provide remote builder public keys here (safe to commit public keys only):
|
# Client public keys allowed to use this host as a remote builder —
|
||||||
# openssh.authorizedKeys.keys = [ "ssh-ed25519 AAAA... client@host" ];
|
# single source of truth is vars.remoteBuilderAuthorizedKeys (safe to
|
||||||
#
|
# commit public keys only).
|
||||||
# Avoid absolute keyFiles paths here because they break pure flake evaluation.
|
openssh.authorizedKeys.keys = vars.remoteBuilderAuthorizedKeys;
|
||||||
openssh.authorizedKeys.keys = ["ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIFDEA1S2ikpObREgbP5uVBWMxIOGbY8B+Wx7VTZK1m6t root@server"
|
|
||||||
"ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIPAYIT9ormlmxZ0SziyDQaUntnKI8HK9/s3Qac1ZKjP2 root@docker"
|
|
||||||
"ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIKKKzoEPl/ZW9KBRHBcp6/ThOngGpwMv5EhkTlgC4aDf root@nixos"
|
|
||||||
"ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIIGtOWOCS+ImHc7NehguoyD7PbonGosKMZqc9+QR3v/h root@nixos"
|
|
||||||
"ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIHxXTQxFnArK5HXG7czeoybZebCGfxpUdusJkPn+BCSp root@server"];
|
|
||||||
};
|
};
|
||||||
|
|
||||||
services.openssh.enable = true;
|
|
||||||
|
|
||||||
nix.settings = {
|
nix.settings = {
|
||||||
trusted-users = [ "root" "nixremote" ];
|
trusted-users = [ "root" vars.remoteBuilderUser ];
|
||||||
experimental-features = [ "nix-command" "flakes" ];
|
experimental-features = [ "nix-command" "flakes" ];
|
||||||
auto-optimise-store = true;
|
auto-optimise-store = true;
|
||||||
builders-use-substitutes = true;
|
builders-use-substitutes = true;
|
||||||
@@ -57,6 +56,6 @@
|
|||||||
nix.gc = {
|
nix.gc = {
|
||||||
automatic = true;
|
automatic = true;
|
||||||
dates = "weekly";
|
dates = "weekly";
|
||||||
options = "--delete-older-than 30d";
|
options = "--delete-older-than ${vars.nixCacheGcMaxAge}";
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,38 @@
|
|||||||
|
{ ... }:
|
||||||
|
|
||||||
|
{
|
||||||
|
imports = [
|
||||||
|
../hardware-configuration/baremetal.nix
|
||||||
|
../boot/efi.nix
|
||||||
|
../disko/baremetal.nix
|
||||||
|
../services/zfs/enable-service.nix
|
||||||
|
];
|
||||||
|
|
||||||
|
# Needed for real wifi/bluetooth/GPU firmware blobs and CPU microcode
|
||||||
|
# updates (hardware-configuration/baremetal.nix's amd.updateMicrocode
|
||||||
|
# keys off this) -- irrelevant on the linode/proxmox/lxc platforms,
|
||||||
|
# which are all VMs with no real hardware to load firmware for.
|
||||||
|
hardware.enableRedistributableFirmware = true;
|
||||||
|
|
||||||
|
# AMD GPU: the amdgpu kernel driver autoloads from the PCI ID with no
|
||||||
|
# extra boot.kernelModules entry needed; this is the userspace half --
|
||||||
|
# the dedicated Xorg driver (not just the generic modesetting fallback)
|
||||||
|
# plus Mesa OpenGL/Vulkan (amdgpu/RADV), same firmware blobs as above.
|
||||||
|
# 32-bit support is for compatibility with 32-bit apps/games.
|
||||||
|
services.xserver.videoDrivers = [ "amdgpu" ];
|
||||||
|
|
||||||
|
hardware.graphics = {
|
||||||
|
enable = true;
|
||||||
|
enable32Bit = true;
|
||||||
|
};
|
||||||
|
|
||||||
|
# The systemd-based initrd (default here since this host has a ZFS root --
|
||||||
|
# see modules/disko/baremetal.nix) locks the root account by default, so
|
||||||
|
# sulogin refuses to hand over a shell if something in the initrd (e.g.
|
||||||
|
# the ZFS pool import) fails and it drops to emergency mode -- confirmed
|
||||||
|
# live: it just loops re-entering the target instead of prompting. This
|
||||||
|
# only affects the pre-switch-root initrd shell, not the installed
|
||||||
|
# system's own login, and is worth the tradeoff on a box already reachable
|
||||||
|
# at the physical console.
|
||||||
|
boot.initrd.systemd.emergencyAccess = true;
|
||||||
|
}
|
||||||
@@ -3,6 +3,7 @@
|
|||||||
{
|
{
|
||||||
imports = [
|
imports = [
|
||||||
../hardware-configuration/vm/linode.nix
|
../hardware-configuration/vm/linode.nix
|
||||||
|
../disko/linode.nix
|
||||||
];
|
];
|
||||||
|
|
||||||
networking = {
|
networking = {
|
||||||
|
|||||||
+204
-4
@@ -1,8 +1,208 @@
|
|||||||
{ ... }:
|
{ config, lib, modulesPath, flakeTarget, ... }:
|
||||||
|
|
||||||
|
let
|
||||||
|
# Bakes this exact flake target's pre-generated SSH host key straight
|
||||||
|
# into /etc/ssh/ -- mirrors modules/installer/host-keys.nix's
|
||||||
|
# builtins.getEnv pattern (impure and empty under normal `nix
|
||||||
|
# build`/`nix eval`, so this is a no-op unless explicitly opted into
|
||||||
|
# with NIXOS_HOST_KEYS_DIR=... --impure), but places the key directly
|
||||||
|
# rather than staging it under /etc/host-keys/ for a later manual copy
|
||||||
|
# -- this is the whole system for a `lxc-*` host, built straight to a
|
||||||
|
# pct-restorable tarball with no install step, so there's no later copy
|
||||||
|
# step to stage for.
|
||||||
|
#
|
||||||
|
# Without this, config.system.build.tarball's built-in system just
|
||||||
|
# generates a fresh host key at first boot like any other host would --
|
||||||
|
# but sops-nix derives its decryption key from *this* file, and
|
||||||
|
# .sops.yaml only trusts whatever key scripts/secrets/sync-host-keys.sh already
|
||||||
|
# registered for this exact target name. A freshly-generated key can
|
||||||
|
# never match that, so every secret (including this host's own login)
|
||||||
|
# permanently fails to decrypt. Confirmed live: sops-install-secrets
|
||||||
|
# errored with "Error getting data key: 0 successful groups required,
|
||||||
|
# got 0" -- the container's actual host key's age fingerprint didn't
|
||||||
|
# match the one registered in .sops.yaml at all.
|
||||||
|
hostKeysDirStr = builtins.getEnv "NIXOS_HOST_KEYS_DIR";
|
||||||
|
hasHostKeysDir = hostKeysDirStr != "" && builtins.pathExists hostKeysDirStr;
|
||||||
|
hostKeysDir = /. + hostKeysDirStr;
|
||||||
|
|
||||||
|
# flakeTarget ("${platform}-${buildType}") comes in via specialArgs from
|
||||||
|
# flake.nix's mkTarget -- exactly the name scripts/secrets/sync-host-keys.sh
|
||||||
|
# registers keys under. Deliberately not read back from
|
||||||
|
# config.environment.etc."flake-target" (which is set to the same value)
|
||||||
|
# -- this module also *contributes* to environment.etc below, and a
|
||||||
|
# module reading the merged value of an option it's still defining is a
|
||||||
|
# circular dependency (confirmed: "infinite recursion encountered").
|
||||||
|
privKeyFile = hostKeysDir + "/${flakeTarget}_ssh_host_ed25519_key";
|
||||||
|
pubKeyFile = hostKeysDir + "/${flakeTarget}_ssh_host_ed25519_key.pub";
|
||||||
|
hasKeyForThisTarget =
|
||||||
|
hasHostKeysDir
|
||||||
|
&& builtins.pathExists privKeyFile
|
||||||
|
&& builtins.pathExists pubKeyFile;
|
||||||
|
in
|
||||||
{
|
{
|
||||||
boot.isContainer = true;
|
# LXC containers share the host kernel — Proxmox starts them by exec'ing
|
||||||
|
# /sbin/init directly, no bootloader/initrd involved — and Proxmox has its
|
||||||
|
# own container hostname/network provisioning outside Nix. nixpkgs' own
|
||||||
|
# virtualisation/proxmox-lxc.nix module already handles all of this
|
||||||
|
# correctly (boot.isContainer, loader.initScript, systemd-networkd) and,
|
||||||
|
# critically, provides config.system.build.tarball — a directly
|
||||||
|
# `pct restore`-able container image, no nixos-install/bind-mount needed
|
||||||
|
# (nixos-install refuses to touch the filesystem it's currently running
|
||||||
|
# on, which is exactly what bind-mounting / onto /mnt for an installer
|
||||||
|
# LXC container does).
|
||||||
|
imports = [
|
||||||
|
(modulesPath + "/virtualisation/proxmox-lxc.nix")
|
||||||
|
../common/preserve-ssh-host-key.nix
|
||||||
|
];
|
||||||
|
|
||||||
boot.loader.grub.enable = false;
|
proxmoxLXC = {
|
||||||
boot.loader.systemd-boot.enable = false;
|
# host.nix declares each host's real hostname (networking.hostName);
|
||||||
|
# keep that instead of letting Proxmox's ambient container config win.
|
||||||
|
manageHostName = true;
|
||||||
|
# Unprivileged by default -- matches how these containers are actually
|
||||||
|
# created (scripts/proxmox/create-proxmox-resource.sh reads this value
|
||||||
|
# back to decide `pct create`'s --unprivileged flag, so the two stay
|
||||||
|
# in sync).
|
||||||
|
#
|
||||||
|
# Any lxc-* host with an NFS fileSystem must be privileged: the kernel's
|
||||||
|
# NFS client doesn't set FS_USERNS_MOUNT, so mounting NFS from inside
|
||||||
|
# *any* non-init user namespace -- which is exactly what an unprivileged
|
||||||
|
# container's UID-mapped root runs in -- is rejected at the VFS layer
|
||||||
|
# with EPERM, no matter what Proxmox's own `mount=nfs;nfs4` container
|
||||||
|
# feature allows at the AppArmor layer (confirmed live: TCP to the NFS
|
||||||
|
# server succeeds, the server's export table matches the container's IP,
|
||||||
|
# and `mount.nfs: Operation not permitted` still fires immediately with
|
||||||
|
# no corresponding denial anywhere in the server's logs -- a kernel-level
|
||||||
|
# rejection, not a network or export-permission one). Deriving this from
|
||||||
|
# fileSystems rather than a per-host override keeps it self-consistent:
|
||||||
|
# any new lxc-* host that declares an NFS mount automatically gets the
|
||||||
|
# privilege level it needs without a separate manual flag.
|
||||||
|
privileged = builtins.any
|
||||||
|
(fs: fs.fsType == "nfs" || fs.fsType == "nfs4")
|
||||||
|
(builtins.attrValues config.fileSystems);
|
||||||
|
};
|
||||||
|
|
||||||
|
boot.loader = {
|
||||||
|
grub.enable = false;
|
||||||
|
systemd-boot.enable = false;
|
||||||
|
};
|
||||||
|
|
||||||
|
# NetworkManager depends on a running udevd to enumerate/classify devices,
|
||||||
|
# which boot.isContainer disables (see nixpkgs' container-config.nix) —
|
||||||
|
# that's what broke DHCP-hostname registration in Pi-hole. The imported
|
||||||
|
# proxmox-lxc.nix module already switches networking to systemd-networkd
|
||||||
|
# for the same reason; it just doesn't disable NetworkManager itself,
|
||||||
|
# which modules/common/configuration.nix enables for every host.
|
||||||
|
networking.networkmanager.enable = lib.mkForce false;
|
||||||
|
|
||||||
|
environment.etc = lib.mkIf hasKeyForThisTarget {
|
||||||
|
"ssh/ssh_host_ed25519_key" = {
|
||||||
|
source = privKeyFile;
|
||||||
|
mode = "0600";
|
||||||
|
};
|
||||||
|
"ssh/ssh_host_ed25519_key.pub" = {
|
||||||
|
source = pubKeyFile;
|
||||||
|
mode = "0644";
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
# virtualisation/proxmox-lxc.nix (imported above) registers the Nix
|
||||||
|
# store DB via a systemd service (register-nix-paths) -- it never runs
|
||||||
|
# an activation script at all. Confirmed live this means neither
|
||||||
|
# sops-nix's "for users" secrets (password hashes -- installed by the
|
||||||
|
# activation script itself, not a systemd service, since they need to
|
||||||
|
# exist *before* user creation) nor the user-creation step that
|
||||||
|
# consumes them ever run on a real lxc-* boot. In this config sops-nix
|
||||||
|
# does NOT generate its own boot-time service (confirmed live: no
|
||||||
|
# sops-nix.service in systemctl list-unit-files on a deployed
|
||||||
|
# lxc-tor-relay container); /run/secrets is a tmpfs cleared on every
|
||||||
|
# reboot, so secrets must be reinstalled on each non-first boot by
|
||||||
|
# nixos-lxc-sops-reinstall (below).
|
||||||
|
#
|
||||||
|
# A systemd service, not boot.postBootCommands: tried that first (it's
|
||||||
|
# a genuine, generally-invoked hook -- nixos/modules/system/boot/stage-2-init.sh,
|
||||||
|
# which becomes this container's actual /sbin/init, unconditionally
|
||||||
|
# runs it) but switch-to-configuration behaves differently that early in
|
||||||
|
# boot (raw stage-2-init.sh, before systemd itself has even started) --
|
||||||
|
# confirmed live it silently failed to rewrite /etc/shadow from there
|
||||||
|
# even in "test" mode, despite the exact same command working reliably
|
||||||
|
# every time when run post-boot (i.e. as a normal systemd service, which
|
||||||
|
# is what this is). Not fully root-caused why the early context
|
||||||
|
# specifically breaks it; a real systemd service sidesteps needing to.
|
||||||
|
#
|
||||||
|
# /etc/shadow already has PLACEHOLDER entries for every declared user
|
||||||
|
# baked in at build time (part of constructing the system closure).
|
||||||
|
# update-users-groups.pl deliberately never overwrites an *existing*
|
||||||
|
# shadow entry -- a correct safety property in general (don't clobber a
|
||||||
|
# real user's real password on a config rebuild) -- but on a genuine
|
||||||
|
# first boot that only means the real hashedPasswordFile-derived hash
|
||||||
|
# never gets the chance to be applied either, since the placeholder is
|
||||||
|
# already "seen". Safe to clear here specifically: there is no real
|
||||||
|
# password yet to protect on a first boot.
|
||||||
|
#
|
||||||
|
# "test" mode, not "boot": confirmed live "boot" mode aborts partway
|
||||||
|
# through (before rewriting /etc/shadow) on a warning that "/boot" is on
|
||||||
|
# a different filesystem -- a real check for a host with a bootloader to
|
||||||
|
# update, meaningless for a container that has none
|
||||||
|
# (boot.loader.{grub,systemd-boot}.enable are both false above), but it
|
||||||
|
# still aborts the script. "test" runs every activation step without
|
||||||
|
# touching boot-loader state at all.
|
||||||
|
#
|
||||||
|
# ConditionPathExists (systemd-native, not a bash-level check) means
|
||||||
|
# this only ever runs once, on the genuine first boot -- systemd itself
|
||||||
|
# skips even starting it on every later boot once the marker exists.
|
||||||
|
# switch-to-configuration is otherwise the operator's call per this
|
||||||
|
# repo's own safety rules, not something to run on every boot.
|
||||||
|
systemd.services.nixos-lxc-first-boot-activate = {
|
||||||
|
description = "Complete first-boot NixOS activation (users, secrets) for this LXC container";
|
||||||
|
wantedBy = [ "multi-user.target" ];
|
||||||
|
unitConfig.ConditionPathExists = "!/var/lib/nixos-lxc-first-boot-activated";
|
||||||
|
serviceConfig = {
|
||||||
|
Type = "oneshot";
|
||||||
|
RemainAfterExit = true;
|
||||||
|
};
|
||||||
|
script = ''
|
||||||
|
rm -f /etc/shadow
|
||||||
|
/run/current-system/bin/switch-to-configuration test
|
||||||
|
mkdir -p /var/lib
|
||||||
|
touch /var/lib/nixos-lxc-first-boot-activated
|
||||||
|
'';
|
||||||
|
};
|
||||||
|
|
||||||
|
# Reinstalls sops secrets on every non-first boot. /run/secrets is a
|
||||||
|
# tmpfs that is cleared on each reboot; without this service, secrets
|
||||||
|
# are permanently absent after the first boot and every service that
|
||||||
|
# reads from /run/secrets fails on start.
|
||||||
|
#
|
||||||
|
# wantedBy/before network.target: switch-to-configuration test requires
|
||||||
|
# D-Bus to restart systemd targets after running activation scripts. D-Bus
|
||||||
|
# is available once basic.target completes (the default After=basic.target
|
||||||
|
# that DefaultDependencies would otherwise add). Placing the service before
|
||||||
|
# network.target ensures secrets are ready before any network-dependent
|
||||||
|
# service (including beszel-agent and nix-serve) starts, while running late
|
||||||
|
# enough that D-Bus is already up.
|
||||||
|
#
|
||||||
|
# ConditionPathExists=... skips this service on the genuine first boot
|
||||||
|
# (the marker doesn't exist yet); nixos-lxc-first-boot-activate handles
|
||||||
|
# that case. On every subsequent boot the condition passes and secrets
|
||||||
|
# are reinstalled before user services start.
|
||||||
|
#
|
||||||
|
# SuccessExitStatus=11: switch-to-configuration exits 11 when it cannot
|
||||||
|
# acquire the activation lock (another switch is already in progress).
|
||||||
|
# During a nixos-rebuild switch the activation already installs secrets, so
|
||||||
|
# treating the lock-held case as success is correct.
|
||||||
|
systemd.services.nixos-lxc-sops-reinstall = {
|
||||||
|
description = "Reinstall sops secrets on each non-first boot (LXC, /run is tmpfs)";
|
||||||
|
wantedBy = [ "network.target" ];
|
||||||
|
before = [ "network.target" ];
|
||||||
|
unitConfig.ConditionPathExists = "/var/lib/nixos-lxc-first-boot-activated";
|
||||||
|
serviceConfig = {
|
||||||
|
Type = "oneshot";
|
||||||
|
RemainAfterExit = true;
|
||||||
|
SuccessExitStatus = "11";
|
||||||
|
};
|
||||||
|
script = ''
|
||||||
|
/run/current-system/bin/switch-to-configuration test
|
||||||
|
'';
|
||||||
|
};
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,9 +1,50 @@
|
|||||||
{ ... }:
|
{ lib, flakeTarget, ... }:
|
||||||
|
|
||||||
|
let
|
||||||
|
# Bakes this exact flake target's pre-generated SSH host key straight
|
||||||
|
# into /etc/ssh/ -- mirrors lxc.nix's builtins.getEnv pattern (impure
|
||||||
|
# and empty under normal `nix build`/`nix eval`, so this is a no-op
|
||||||
|
# unless explicitly opted into with NIXOS_HOST_KEYS_DIR=... --impure).
|
||||||
|
#
|
||||||
|
# Unlike --pre-format-files (which places files on the QEMU builder VM's
|
||||||
|
# rootfs, not the target disk), embedding via environment.etc here means
|
||||||
|
# nixos-install's own activation step installs the key onto the target
|
||||||
|
# disk. sshd-keygen then finds it already present and skips generation,
|
||||||
|
# so the disk image boots with the clan-registered key and sops can
|
||||||
|
# decrypt on first boot.
|
||||||
|
#
|
||||||
|
# Without this, nixos-install's sshd-keygen activation generates a fresh
|
||||||
|
# key (unregistered in .sops.yaml), sops decryption fails permanently,
|
||||||
|
# and password hashes are never applied -- confirmed live: passwords
|
||||||
|
# stayed '!' even with mutableUsers = false because hashedPasswordFile
|
||||||
|
# pointed to a path that sops never wrote.
|
||||||
|
hostKeysDirStr = builtins.getEnv "NIXOS_HOST_KEYS_DIR";
|
||||||
|
hasHostKeysDir = hostKeysDirStr != "" && builtins.pathExists hostKeysDirStr;
|
||||||
|
hostKeysDir = /. + hostKeysDirStr;
|
||||||
|
|
||||||
|
privKeyFile = hostKeysDir + "/${flakeTarget}_ssh_host_ed25519_key";
|
||||||
|
pubKeyFile = hostKeysDir + "/${flakeTarget}_ssh_host_ed25519_key.pub";
|
||||||
|
hasKeyForThisTarget =
|
||||||
|
hasHostKeysDir
|
||||||
|
&& builtins.pathExists privKeyFile
|
||||||
|
&& builtins.pathExists pubKeyFile;
|
||||||
|
in
|
||||||
{
|
{
|
||||||
imports = [
|
imports = [
|
||||||
../hardware-configuration/vm/proxmox.nix
|
../hardware-configuration/vm/proxmox.nix
|
||||||
../boot/efi.nix
|
../boot/efi.nix
|
||||||
../disko/proxmox.nix
|
../disko/proxmox.nix
|
||||||
|
../common/preserve-ssh-host-key.nix
|
||||||
];
|
];
|
||||||
|
|
||||||
|
environment.etc = lib.mkIf hasKeyForThisTarget {
|
||||||
|
"ssh/ssh_host_ed25519_key" = {
|
||||||
|
source = privKeyFile;
|
||||||
|
mode = "0600";
|
||||||
|
};
|
||||||
|
"ssh/ssh_host_ed25519_key.pub" = {
|
||||||
|
source = pubKeyFile;
|
||||||
|
mode = "0644";
|
||||||
|
};
|
||||||
|
};
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,42 @@
|
|||||||
|
{ config, lib, vars, ... }:
|
||||||
|
|
||||||
|
let
|
||||||
|
# FQDN of the LAN NFS VIP (Pacemaker vip-lan, 192.168.2.229). Defined in
|
||||||
|
# variables.nix as haLanNfsFqdn; using the FQDN avoids systemd-resolved
|
||||||
|
# LLMNR quirks and survives a future VIP renumber via a DNS-only update.
|
||||||
|
nfsServer = vars.haLanNfsFqdn;
|
||||||
|
in
|
||||||
|
{
|
||||||
|
fileSystems.${vars.nfsShares.pxebootImages.mountpoint} = {
|
||||||
|
device = "${nfsServer}:${vars.haStorageRoot}/${vars.nfsShares.pxebootImages.subpath}";
|
||||||
|
fsType = "nfs";
|
||||||
|
options = [
|
||||||
|
"_netdev"
|
||||||
|
"noatime"
|
||||||
|
] ++ (if config.boot.isContainer
|
||||||
|
# NFSv4 requires rpc_pipefs (sunrpc filesystem), which Proxmox LXC
|
||||||
|
# containers block unless `features: mount=nfs` is set. Use NFSv3+nolock
|
||||||
|
# instead: no rpc_pipefs dependency at the protocol level, and rpcbind
|
||||||
|
# on the server handles port resolution without needing client-side
|
||||||
|
# sunrpc infrastructure. nofail keeps boot clean if server is unreachable.
|
||||||
|
then [ "nfsvers=3" "proto=tcp" "nolock" "nofail" ]
|
||||||
|
else [ "nfsvers=4.2" "x-systemd.automount" ]);
|
||||||
|
};
|
||||||
|
|
||||||
|
# NixOS pulls var-lib-nfs-rpc_pipefs.mount (the sunrpc filesystem) into
|
||||||
|
# nfs-client.target for any nfs fileSystems entry. In LXC containers the
|
||||||
|
# sunrpc mount is blocked by Proxmox's AppArmor profile, causing it to fail
|
||||||
|
# and the activation to report an error even though our mount uses nofail.
|
||||||
|
# Add ConditionVirtualization=!container via drop-in so systemd skips the
|
||||||
|
# unit entirely in containers (skip = inactive, not failed), which keeps
|
||||||
|
# nfs-client.target green and activation clean.
|
||||||
|
systemd.units = lib.mkIf config.boot.isContainer {
|
||||||
|
"var-lib-nfs-rpc_pipefs.mount" = {
|
||||||
|
overrideStrategy = "asDropin";
|
||||||
|
text = ''
|
||||||
|
[Unit]
|
||||||
|
ConditionVirtualization=!container
|
||||||
|
'';
|
||||||
|
};
|
||||||
|
};
|
||||||
|
}
|
||||||
@@ -0,0 +1,32 @@
|
|||||||
|
{ netbootSystem, netbootMinimalSystem, ... }:
|
||||||
|
|
||||||
|
let
|
||||||
|
# config.system.build.kernel and .netbootRamdisk are directories, not the
|
||||||
|
# files themselves — nixpkgs' own system.build.kexecTree does the same
|
||||||
|
# ${...}/<file> dereference for the same reason.
|
||||||
|
mkStageRules = { dirName, system }:
|
||||||
|
let
|
||||||
|
inherit (system.config.system.boot.loader) kernelFile;
|
||||||
|
dir = "/srv/pxe/http/${dirName}";
|
||||||
|
in
|
||||||
|
[
|
||||||
|
# Declared here too (not just in build-types/pxe-boot.nix) so this
|
||||||
|
# module's C+ rules don't depend on cross-module list-merge ordering —
|
||||||
|
# tmpfiles' C type needs the target directory to already exist.
|
||||||
|
"d ${dir} 0755 root root -"
|
||||||
|
"C+ ${dir}/${kernelFile} 0644 root root - ${system.config.system.build.kernel}/${kernelFile}"
|
||||||
|
"C+ ${dir}/initrd 0644 root root - ${system.config.system.build.netbootRamdisk}/initrd"
|
||||||
|
"C+ ${dir}/netboot.ipxe 0644 root root - ${system.config.system.build.netbootIpxeScript}/netboot.ipxe"
|
||||||
|
];
|
||||||
|
in
|
||||||
|
{
|
||||||
|
# Builds this flake's own installer netboot image (the same one
|
||||||
|
# `nix build .#pxe` produces) plus the vanilla NixOS minimal netboot image
|
||||||
|
# (`nix build .#pxe-minimal`), and stages both where menu.ipxe's
|
||||||
|
# :auto-installer / :nixos-minimal entries expect them, so the pxe-boot
|
||||||
|
# host is self-contained — no manual operator step to populate
|
||||||
|
# /srv/pxe/http after deploy.
|
||||||
|
systemd.tmpfiles.rules =
|
||||||
|
mkStageRules { dirName = "auto-installer"; system = netbootSystem; }
|
||||||
|
++ mkStageRules { dirName = "nixos-minimal"; system = netbootMinimalSystem; };
|
||||||
|
}
|
||||||
@@ -1,14 +1,23 @@
|
|||||||
{ ... }:
|
{ config, lib, vars, ... }:
|
||||||
|
|
||||||
{
|
{
|
||||||
fileSystems."/mnt/raspi" = {
|
fileSystems.${vars.raspiMountpoint} = {
|
||||||
device = "raspberrypi.tail13f623.ts.net:/home/raspi/raspi";
|
device = "${vars.raspberryPiHost}.${vars.tailnetDomain}:${vars.raspiNfsPath}";
|
||||||
fsType = "nfs4";
|
fsType = "nfs4";
|
||||||
options = [
|
options = [
|
||||||
"nofail"
|
"nofail"
|
||||||
"_netdev"
|
"_netdev"
|
||||||
"noatime"
|
"noatime"
|
||||||
|
|
||||||
|
# Explicitly use NFSv4.2 if supported
|
||||||
|
"nfsvers=4.2"
|
||||||
|
] ++ lib.optionals (!config.boot.isContainer) [
|
||||||
|
# `x-systemd.automount` never works inside a Linux container (LXC
|
||||||
|
# included) -- confirmed live on lxc-docker: systemd logs "Starting
|
||||||
|
# of <unit>.automount unsupported" and never mounts it. `nofail`
|
||||||
|
# above already keeps boot non-blocking there, so plain eager
|
||||||
|
# mounting is fine.
|
||||||
|
|
||||||
# Don't mount until first access
|
# Don't mount until first access
|
||||||
"x-systemd.automount"
|
"x-systemd.automount"
|
||||||
|
|
||||||
@@ -17,9 +26,6 @@ fileSystems."/mnt/raspi" = {
|
|||||||
|
|
||||||
# Give the Pi/Tailscale a little time to appear
|
# Give the Pi/Tailscale a little time to appear
|
||||||
"x-systemd.device-timeout=10s"
|
"x-systemd.device-timeout=10s"
|
||||||
|
|
||||||
# Explicitly use NFSv4.2 if supported
|
|
||||||
"nfsvers=4.2"
|
|
||||||
];
|
];
|
||||||
};
|
};
|
||||||
|
|
||||||
|
|||||||
@@ -1,26 +0,0 @@
|
|||||||
{ pkgs, ... }:
|
|
||||||
|
|
||||||
{
|
|
||||||
# Install the remote builder key on each client host (do not commit private keys):
|
|
||||||
# sudo install -d -m 0700 /root/.ssh
|
|
||||||
# sudo install -m 0600 ./nixremote /root/.ssh/nixremote
|
|
||||||
# sudo ssh -i /root/.ssh/nixremote nixremote@nix-cache nix-store --version
|
|
||||||
nix.distributedBuilds = true;
|
|
||||||
|
|
||||||
nix.buildMachines = [
|
|
||||||
{
|
|
||||||
hostName = "nix-cache";
|
|
||||||
sshUser = "nixremote";
|
|
||||||
sshKey = "/root/.ssh/nixremote";
|
|
||||||
system = pkgs.stdenv.hostPlatform.system;
|
|
||||||
maxJobs = 4;
|
|
||||||
speedFactor = 2;
|
|
||||||
supportedFeatures = [ "nixos-test" "benchmark" "big-parallel" "kvm" ];
|
|
||||||
}
|
|
||||||
];
|
|
||||||
|
|
||||||
nix.settings = {
|
|
||||||
builders-use-substitutes = true;
|
|
||||||
max-jobs = "auto";
|
|
||||||
};
|
|
||||||
}
|
|
||||||
@@ -1,20 +0,0 @@
|
|||||||
{ ... }:
|
|
||||||
|
|
||||||
{
|
|
||||||
services.logrotate = {
|
|
||||||
enable = true;
|
|
||||||
|
|
||||||
settings = {
|
|
||||||
"/mnt/docker/volumes/traefik-data/logs/*.log" = {
|
|
||||||
daily = true;
|
|
||||||
size = "100M";
|
|
||||||
rotate = 20;
|
|
||||||
compress = true;
|
|
||||||
missingok = true;
|
|
||||||
notifempty = true;
|
|
||||||
copytruncate = true;
|
|
||||||
};
|
|
||||||
|
|
||||||
};
|
|
||||||
};
|
|
||||||
}
|
|
||||||
@@ -1,4 +1,4 @@
|
|||||||
{ ... }:
|
_:
|
||||||
|
|
||||||
{
|
{
|
||||||
services.rpcbind.enable = true;
|
services.rpcbind.enable = true;
|
||||||
|
|||||||
@@ -1,22 +0,0 @@
|
|||||||
{ pkgs, ... }:
|
|
||||||
|
|
||||||
{
|
|
||||||
# Create nextcloud cron scheduled task
|
|
||||||
systemd.services.nextcloud = {
|
|
||||||
description = "Nextcloud scheduled task";
|
|
||||||
script = ''${pkgs.bash}/bin/bash ~/docker/services-up.sh --profile nextcloud exec -u 33 nextcloud-webapp php ./cron.php'';
|
|
||||||
serviceConfig = {
|
|
||||||
Type = "oneshot";
|
|
||||||
User = "nixos";
|
|
||||||
};
|
|
||||||
path = with pkgs; [ docker docker-compose ];
|
|
||||||
};
|
|
||||||
|
|
||||||
systemd.timers.nextcloud = {
|
|
||||||
wantedBy = [ "timers.target" ];
|
|
||||||
timerConfig = {
|
|
||||||
OnCalendar = "*:0/5";
|
|
||||||
Persistent = true;
|
|
||||||
};
|
|
||||||
};
|
|
||||||
}
|
|
||||||
@@ -1,9 +1,15 @@
|
|||||||
{ pkgs, ... }:
|
{ pkgs, ... }:
|
||||||
|
|
||||||
{
|
{
|
||||||
boot.supportedFilesystems = [ "zfs" ];
|
boot = {
|
||||||
boot.zfs.forceImportRoot = false;
|
supportedFilesystems = [ "zfs" ];
|
||||||
boot.zfs.package = pkgs.zfs_unstable;
|
zfs = {
|
||||||
|
forceImportRoot = false;
|
||||||
|
package = pkgs.zfs_unstable;
|
||||||
|
devNodes = "/dev/disk/by-id";
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
services.zfs = {
|
services.zfs = {
|
||||||
autoScrub.enable = true;
|
autoScrub.enable = true;
|
||||||
autoSnapshot.enable = true;
|
autoSnapshot.enable = true;
|
||||||
@@ -12,5 +18,4 @@ boot.supportedFilesystems = [ "zfs" ];
|
|||||||
|
|
||||||
#systemd.services.zfs-import-cache.enable = true;
|
#systemd.services.zfs-import-cache.enable = true;
|
||||||
systemd.services.zfs-mount.enable = true;
|
systemd.services.zfs-mount.enable = true;
|
||||||
boot.zfs.devNodes = "/dev/disk/by-id";
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,4 +1,4 @@
|
|||||||
{ ... }:
|
_:
|
||||||
|
|
||||||
{
|
{
|
||||||
services.tailscale.enable = true;
|
services.tailscale.enable = true;
|
||||||
|
|||||||
@@ -1,12 +0,0 @@
|
|||||||
{ ... }:
|
|
||||||
|
|
||||||
{
|
|
||||||
services.tailscale = {
|
|
||||||
enable = true;
|
|
||||||
|
|
||||||
extraUpFlags = [
|
|
||||||
"--advertise-exit-node"
|
|
||||||
"--advertise-routes=192.168.2.0/24"
|
|
||||||
];
|
|
||||||
};
|
|
||||||
}
|
|
||||||
@@ -0,0 +1,35 @@
|
|||||||
|
{ pkgs, ... }:
|
||||||
|
|
||||||
|
{
|
||||||
|
imports = [ ./enable-service.nix ];
|
||||||
|
|
||||||
|
services.tailscale = {
|
||||||
|
# Enables the sysctl forwarding settings subnet routers need;
|
||||||
|
# without this, --advertise-routes has no effect.
|
||||||
|
useRoutingFeatures = "server";
|
||||||
|
|
||||||
|
# Lets peers reach this node directly over the tailscale UDP port
|
||||||
|
# instead of relaying through DERP.
|
||||||
|
openFirewall = true;
|
||||||
|
};
|
||||||
|
|
||||||
|
# Tailscale recommends these ethtool flags on the uplink interface to get
|
||||||
|
# full UDP GRO throughput on subnet routers (https://tailscale.com/s/ethtool-config-udp-gro).
|
||||||
|
# The interface is derived from the default route so it works regardless of
|
||||||
|
# what the NIC is named on a given host.
|
||||||
|
systemd.services.tailscale-udp-gro = {
|
||||||
|
description = "Enable UDP GRO forwarding on uplink for Tailscale subnet router";
|
||||||
|
after = [ "network-online.target" ];
|
||||||
|
wants = [ "network-online.target" ];
|
||||||
|
wantedBy = [ "multi-user.target" ];
|
||||||
|
path = [ pkgs.ethtool pkgs.iproute2 ];
|
||||||
|
serviceConfig = {
|
||||||
|
Type = "oneshot";
|
||||||
|
RemainAfterExit = true;
|
||||||
|
ExecStart = pkgs.writeShellScript "tailscale-udp-gro" ''
|
||||||
|
NETDEV=$(ip -o route get 8.8.8.8 | cut -f 5 -d " ")
|
||||||
|
ethtool -K "$NETDEV" rx-udp-gro-forwarding on rx-gro-list off
|
||||||
|
'';
|
||||||
|
};
|
||||||
|
};
|
||||||
|
}
|
||||||
@@ -0,0 +1,53 @@
|
|||||||
|
{ vars, ... }:
|
||||||
|
|
||||||
|
{
|
||||||
|
# Run dnsmasq on the LAN interface as a forwarding-only resolver for
|
||||||
|
# *.ts.net (Tailscale MagicDNS names). FreeIPA's bind-dyndb-ldap
|
||||||
|
# cannot reach vars.tailscaleResolverIp directly because the DC is not a
|
||||||
|
# Tailscale node. This host IS a Tailscale node and can reach it via
|
||||||
|
# tailscale0, so it acts as an intermediary: FreeIPA has a conditional
|
||||||
|
# forward zone for ts.net pointing here (vars.tailscaleRouterIp), and this
|
||||||
|
# dnsmasq instance forwards those queries onward to Tailscale's resolver.
|
||||||
|
#
|
||||||
|
# Configure FreeIPA once after deploying this host:
|
||||||
|
# kinit admin
|
||||||
|
# ipa dnsforwardzone-add ${vars.tailnetDomain} \
|
||||||
|
# --forwarder=${vars.tailscaleRouterIp} \
|
||||||
|
# --forward-policy=only
|
||||||
|
# Note: IPA refuses to shadow ts.net (a real public TLD); use the
|
||||||
|
# tailnet-specific subdomain (vars.tailnetDomain) instead.
|
||||||
|
services.dnsmasq = {
|
||||||
|
enable = true;
|
||||||
|
# NixOS's dnsmasq module defaults resolveLocalQueries to true, which adds
|
||||||
|
# 127.0.0.1 to networking.nameservers and makes dnsmasq bind to
|
||||||
|
# listen-address=127.0.0.1. This instance is not the host's local
|
||||||
|
# resolver — it only serves IPA's conditional forwarder for tailnet names.
|
||||||
|
# The host uses domainControllerIp directly (networking.nameservers in
|
||||||
|
# host.nix). Without this, all host DNS goes through dnsmasq, which has
|
||||||
|
# no upstream for general queries (no-resolv=true), breaking resolution.
|
||||||
|
resolveLocalQueries = false;
|
||||||
|
settings = {
|
||||||
|
# Listen only on the LAN interface — not tailscale0 or loopback.
|
||||||
|
# bind-interfaces prevents dnsmasq from binding to 0.0.0.0 and then
|
||||||
|
# filtering by interface later; combined with `interface` this ensures
|
||||||
|
# it genuinely listens only on eth0.
|
||||||
|
bind-interfaces = true;
|
||||||
|
interface = [ vars.lxcLanInterface ];
|
||||||
|
|
||||||
|
# Forward-only: no local /etc/hosts or /etc/resolv.conf reading, no
|
||||||
|
# negative caching of NXDOMAIN for names this instance doesn't serve.
|
||||||
|
# All ts.net queries come from FreeIPA's conditional forwarder and must
|
||||||
|
# be answered by Tailscale's resolver.
|
||||||
|
no-hosts = true;
|
||||||
|
no-resolv = true;
|
||||||
|
|
||||||
|
# Forward *.tailnetDomain to Tailscale's internal resolver, scoped to
|
||||||
|
# the tailnet-specific subdomain rather than all of ts.net (FreeIPA
|
||||||
|
# refuses to shadow ts.net, a real public TLD).
|
||||||
|
server = [ "/${vars.tailnetDomain}/${vars.tailscaleResolverIp}" ];
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
networking.firewall.allowedUDPPorts = [ vars.ports.dns ];
|
||||||
|
networking.firewall.allowedTCPPorts = [ vars.ports.dns ];
|
||||||
|
}
|
||||||
@@ -0,0 +1,35 @@
|
|||||||
|
{ pkgs, vars, ... }:
|
||||||
|
|
||||||
|
{
|
||||||
|
services.tor = {
|
||||||
|
enable = true;
|
||||||
|
|
||||||
|
# Opens settings.ORPort (and DirPort, unset here) in the firewall —
|
||||||
|
# see the nixpkgs tor module's own networking.firewall.mkIf block.
|
||||||
|
openFirewall = true;
|
||||||
|
|
||||||
|
relay = {
|
||||||
|
enable = true;
|
||||||
|
# Plain middle/guard relay, not "exit" — relays onion traffic between
|
||||||
|
# other Tor nodes without ever making requests to the public internet
|
||||||
|
# on a user's behalf, avoiding the abuse complaints and legal exposure
|
||||||
|
# an exit node invites.
|
||||||
|
role = "relay";
|
||||||
|
};
|
||||||
|
|
||||||
|
settings.ORPort = vars.ports.torRelayOrPort;
|
||||||
|
|
||||||
|
# Unix control socket at /run/tor/control (GroupWritable, group "tor")
|
||||||
|
# -- what nyx below actually monitors the relay through. Nyx's own
|
||||||
|
# default control-socket path (/var/run/tor/control) resolves to the
|
||||||
|
# same place, so no extra nyx config is needed.
|
||||||
|
controlSocket.enable = true;
|
||||||
|
};
|
||||||
|
|
||||||
|
# Lets the primary user's shell session read/write the control socket
|
||||||
|
# above without being root -- otherwise nyx fails to authenticate against
|
||||||
|
# it at all.
|
||||||
|
users.users.${vars.primaryUser}.extraGroups = [ "tor" ];
|
||||||
|
|
||||||
|
environment.systemPackages = [ pkgs.nyx ];
|
||||||
|
}
|
||||||
@@ -0,0 +1,20 @@
|
|||||||
|
{ vars, ... }:
|
||||||
|
|
||||||
|
{
|
||||||
|
services.logrotate = {
|
||||||
|
enable = true;
|
||||||
|
|
||||||
|
settings = {
|
||||||
|
"${vars.nfsShares.dockerVolumes.mountpoint}/traefik-data/logs/*.log" = {
|
||||||
|
daily = true;
|
||||||
|
size = vars.traefikLogRotate.maxSize;
|
||||||
|
rotate = vars.traefikLogRotate.keep;
|
||||||
|
compress = true;
|
||||||
|
missingok = true;
|
||||||
|
notifempty = true;
|
||||||
|
copytruncate = true;
|
||||||
|
};
|
||||||
|
|
||||||
|
};
|
||||||
|
};
|
||||||
|
}
|
||||||
-30
@@ -1,30 +0,0 @@
|
|||||||
#create MBR table
|
|
||||||
parted /dev/sda -- mklabel msdos
|
|
||||||
#create nixos partition
|
|
||||||
parted /dev/sda -- mkpart primary 1MB -8GB
|
|
||||||
#set nixos partition to bootable
|
|
||||||
parted /dev/sda -- set 1 boot on
|
|
||||||
# create swap partition
|
|
||||||
parted /dev/sda -- mkpart primary linux-swap -8GB 100%
|
|
||||||
|
|
||||||
#format OS partition
|
|
||||||
mkfs.ext4 -L nixos /dev/sda1
|
|
||||||
#format swap
|
|
||||||
mkswap -L swap /dev/sda2
|
|
||||||
|
|
||||||
#activate swap
|
|
||||||
swapon /dev/sda2
|
|
||||||
|
|
||||||
#mount nixos partition
|
|
||||||
mount /dev/disk/by-label/nixos /mnt
|
|
||||||
export TMPDIR=/mnt/install-tmp
|
|
||||||
mkdir -p /mnt/install-tmp
|
|
||||||
#Generate config
|
|
||||||
#nixos-generate-config --root /mnt/
|
|
||||||
|
|
||||||
#copy customised configuration over
|
|
||||||
#cp configuration.nix /mnt/etc/nixos/configuration.nix
|
|
||||||
|
|
||||||
#nixos-install --no-root-passwd
|
|
||||||
|
|
||||||
#reboot
|
|
||||||
@@ -1,134 +0,0 @@
|
|||||||
# Spec: Remove Sensitive Information from NixOS Flake
|
|
||||||
|
|
||||||
## Goal
|
|
||||||
|
|
||||||
Every secret currently readable in plaintext anywhere in this repo (working tree *and* git history) gets removed, replaced with `sops-nix`-managed encrypted references, and rotated. When this is done, the repo should be safe to make public without exposing anything about the systems it configures.
|
|
||||||
|
|
||||||
Treat this as three sequential milestones. Do not start git history rewriting (Milestone 3) until Milestones 1 and 2 are fully verified and the flake still builds. This should be its own branch (`refactor/secrets`) until fully verified, then merged.
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Milestone 1 — Audit
|
|
||||||
|
|
||||||
Before touching anything, produce a complete inventory. Do not guess at scope — grep the whole tree and the whole history.
|
|
||||||
|
|
||||||
1. Run a secret scanner across the working tree and full history. Use both, since they catch different things:
|
|
||||||
- `gitleaks detect --source . -v --log-opts="--all"` (scans history too)
|
|
||||||
- `trufflehog git file://. --since-commit=$(git rev-list --max-parents=0 HEAD) --only-verified=false`
|
|
||||||
If neither is installed, add them via a temporary `nix-shell -p gitleaks trufflehog` — don't install anything globally on the host.
|
|
||||||
|
|
||||||
2. Manually grep for the categories below, since scanners miss config-specific patterns:
|
|
||||||
- `hashedPassword`, `password`, `initialPassword`, `initialHashedPassword` in any `users.users.*` block
|
|
||||||
- `age.secrets`, `sops.secrets` (if any partial secrets work already exists — check for it)
|
|
||||||
- PSK / `preSharedKey`, `privateKeyFile` inline values (vs. file references) for WireGuard
|
|
||||||
- `authKey`, `apiToken`, `api_key`, `token =`, `secret =` in service modules (Tailscale, Cloudflare, backup tools, etc.)
|
|
||||||
- SSH private key material: search for `BEGIN OPENSSH PRIVATE KEY` / `BEGIN RSA PRIVATE KEY` literals
|
|
||||||
- TLS cert/key pairs committed under e.g. `secrets/`, `certs/`, `pki/`
|
|
||||||
- Real name, personal email, home address, or anything in comments/hostnames that maps a machine to your physical identity or network layout (e.g. hostnames like `wayne-desktop`, static LAN IPs, ISP-identifying info)
|
|
||||||
- `.env` files, `secrets.nix`, `secrets.yaml`, or any file that looks like it was meant to be gitignored but wasn't
|
|
||||||
|
|
||||||
3. Produce `secrets-inventory.md` (temporary, delete before finishing) listing: file path, line, secret type, and which host/service it belongs to. This becomes the checklist for Milestone 2 — every row must be either migrated to sops or deleted, with nothing left unaccounted for.
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Milestone 2 — Migrate to sops-nix
|
|
||||||
|
|
||||||
### 2.1 Set up sops-nix
|
|
||||||
|
|
||||||
1. Add the flake input:
|
|
||||||
```nix
|
|
||||||
sops-nix.url = "github:Mic92/sops-nix";
|
|
||||||
sops-nix.inputs.nixpkgs.follows = "nixpkgs";
|
|
||||||
```
|
|
||||||
2. Import `sops-nix.nixosModules.sops` into each host's module list (or into a shared `common.nix` if all hosts use it).
|
|
||||||
3. Generate an age keypair **per host** (not one shared key for everything — a compromised host shouldn't decrypt every other host's secrets):
|
|
||||||
```
|
|
||||||
nix-shell -p age --run "age-keygen -o /var/lib/sops-nix/key.txt"
|
|
||||||
```
|
|
||||||
Print the public key (`age-keygen -y`) for each host — you'll need it for `.sops.yaml`.
|
|
||||||
4. Also generate one age key for yourself (your admin workstation) so you can edit secrets without needing to SSH into a host: store it at `~/.config/sops/age/keys.txt`, back it up somewhere outside this repo (password manager, offline). **If this key is lost, every secret encrypted with it is unrecoverable — losing the age key is equivalent to losing the secrets.**
|
|
||||||
5. Create `.sops.yaml` at the repo root defining creation rules: which age public keys can decrypt which secrets files, keyed by path regex, so e.g. `secrets/hostA.yaml` is decryptable by your admin key + hostA's key, `secrets/hostB.yaml` by your admin key + hostB's key.
|
|
||||||
|
|
||||||
### 2.2 Migrate each secret category from the inventory
|
|
||||||
|
|
||||||
For each row in `secrets-inventory.md`:
|
|
||||||
|
|
||||||
- **Password hashes**: generate hash with `mkpasswd -m sha-512` (or `bcrypt` if your setup wants that), store under `sops.secrets."<name>/hashedPassword"`, reference via `users.users.<name>.hashedPasswordFile = config.sops.secrets."<name>/hashedPassword".path;`. Do not put the *plaintext* password anywhere, only the hash, and only the hash goes into the encrypted sops file.
|
|
||||||
- **API tokens / auth keys**: move the raw value into the per-host sops YAML, reference in the module via `config.sops.secrets."<service>/token".path` — most NixOS service modules that take a token also accept a `*File` variant (e.g. `environmentFile`, `tokenFile`); use that instead of passing the value directly.
|
|
||||||
- **Private keys / certs**: move the PEM/key content wholesale into a sops secret, output as a file with appropriate `sops.secrets.<name>.path`, `owner`, `mode`, `restartUnits` so the depending service (sshd, wireguard, nginx) reloads when the secret changes.
|
|
||||||
- **Personal/identifying info**: this doesn't belong in sops (it's not "secret," it's just information you don't want public). Replace real names/emails with placeholders or move to a small untracked `local.nix` that's `.gitignore`'d and imported conditionally, with a documented template (`local.nix.example`) committed instead.
|
|
||||||
|
|
||||||
### 2.3 Verify before moving on
|
|
||||||
|
|
||||||
- `nixos-rebuild dry-build --flake .#<host>` succeeds for every host.
|
|
||||||
- `sudo nixos-rebuild switch --flake .#<host>` on at least one real machine (or a VM) confirms secrets decrypt and services start.
|
|
||||||
- Confirm decrypted secrets land under `/run/secrets/` (not the Nix store — anything placed in `/nix/store` is world-readable by design, so sops-nix's runtime-only placement is the whole point; double check no module accidentally pulls a secret path into a store-built config file).
|
|
||||||
- Re-run the grep/scanner sweep from Milestone 1 against the *working tree only* (not history yet) — it should now come back clean.
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Milestone 3 — Scrub git history
|
|
||||||
|
|
||||||
Do this only after Milestone 2 is merged to your main branch and confirmed working, since it rewrites every commit SHA from the point of the earliest offending commit onward.
|
|
||||||
|
|
||||||
**This is destructive and irreversible on your local clone. Back up first:**
|
|
||||||
```
|
|
||||||
cp -r /path/to/nixos-repo /path/to/nixos-repo-backup-$(date +%F)
|
|
||||||
```
|
|
||||||
|
|
||||||
1. Install `git-filter-repo` (not the older `git filter-branch` / BFG — filter-repo is the currently maintained, faster, safer tool):
|
|
||||||
```
|
|
||||||
nix-shell -p git-filter-repo
|
|
||||||
```
|
|
||||||
2. Use the `secrets-inventory.md` list to build a list of literal strings/paths to strip. Two approaches, use both:
|
|
||||||
- Path-based: if whole files were secret (e.g. `secrets.nix`, a `.env`, a private key file), remove them entirely from history:
|
|
||||||
```
|
|
||||||
git filter-repo --path secrets.nix --path .env --invert-paths
|
|
||||||
```
|
|
||||||
- Value-based: for secrets embedded inline in files you're keeping (not deleting the whole file), use `--replace-text` with a file listing each literal secret string to replace with `***REMOVED***`:
|
|
||||||
```
|
|
||||||
git filter-repo --replace-text expressions.txt
|
|
||||||
```
|
|
||||||
3. After filtering, verify: run the Milestone 1 scanners again against full history (`--log-opts="--all"`). They must come back clean.
|
|
||||||
4. Force-push the rewritten history:
|
|
||||||
```
|
|
||||||
git push origin --force --all
|
|
||||||
git push origin --force --tags
|
|
||||||
```
|
|
||||||
5. **Every other clone of this repo (other machines, WSL instances, CI) must be deleted and re-cloned fresh** — a `git pull` against rewritten history will not work cleanly and risks resurrecting the old commits. Don't try to reconcile old clones; throw them away and re-clone.
|
|
||||||
6. If this repo has ever been pushed to a public host (GitHub, etc.) or a fork/mirror exists, treat every secret that was ever in history as **permanently compromised regardless of the rewrite** — caches, forks, and Wayback-style archives can retain old commits indefinitely. History scrubbing prevents *future* exposure via `git clone`; it does not undo past exposure.
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Milestone 4 — Rotate everything
|
|
||||||
|
|
||||||
Because the secrets were exposed in history (even briefly, even in a private repo), the migration is not complete until every credential in the inventory has been **rotated**, not just re-encrypted. Re-encrypting an already-leaked value protects it going forward but doesn't undo the leak.
|
|
||||||
|
|
||||||
For each row in the original inventory:
|
|
||||||
- Password hashes → change the actual account password, regenerate the hash, update the sops file.
|
|
||||||
- API tokens/auth keys → revoke the old token in the issuing service's dashboard (Cloudflare, Tailscale, backup provider, etc.) and generate a new one.
|
|
||||||
- SSH/WireGuard private keys → generate new keypairs, update the corresponding public key wherever it's trusted (authorized_keys, peer configs, etc.), retire the old ones.
|
|
||||||
- TLS certs → reissue if the private key was exposed.
|
|
||||||
|
|
||||||
Keep `secrets-inventory.md` open during this step and check off each row as rotated. Delete the file only once every row is checked off — it should not be committed.
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Ongoing prevention
|
|
||||||
|
|
||||||
Add a pre-commit hook (or a `nix flake check` step) running `gitleaks protect --staged` so a secret can't be committed again by accident. Document in the repo README (briefly) that new secrets go through `sops <file>` to edit, never as plaintext in a tracked file.
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Definition of done
|
|
||||||
|
|
||||||
- [ ] Milestone 1 inventory complete and reviewed
|
|
||||||
- [ ] All hosts have per-host age keys; admin key backed up outside the repo
|
|
||||||
- [ ] Every inventoried secret migrated to sops-nix, referenced via `*File`/`sops.secrets.*.path`, nothing plaintext in the working tree
|
|
||||||
- [ ] `nixos-rebuild dry-build` and at least one real `switch` verified per host
|
|
||||||
- [ ] Working-tree scanner sweep clean
|
|
||||||
- [ ] History rewritten with `git-filter-repo`, force-pushed, full-history scanner sweep clean
|
|
||||||
- [ ] All other clones deleted and re-cloned from the rewritten history
|
|
||||||
- [ ] Every credential in the original inventory rotated (not just re-encrypted)
|
|
||||||
- [ ] Pre-commit secret scanning hook added
|
|
||||||
- [ ] `secrets-inventory.md` deleted from the working directory (never committed)
|
|
||||||
Executable
+134
@@ -0,0 +1,134 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# Bumps the NixOS release branch this flake tracks — flake.nix's
|
||||||
|
# `nixpkgs.url` and `home-manager.url` — in one place, via targeted
|
||||||
|
# substitution of just those two lines. Deliberately does NOT touch any
|
||||||
|
# `system.stateVersion` anywhere in the repo: per NixOS's own docs, that
|
||||||
|
# value must stay fixed at whatever it was on a host's first install (it
|
||||||
|
# pins on-disk data-format defaults, not "which nixpkgs release am I on"),
|
||||||
|
# so it's never something a channel bump should follow.
|
||||||
|
#
|
||||||
|
# scripts/codex-maintenance.sh's own `nixos-25.11` pin (used only to fetch
|
||||||
|
# nixpkgs-fmt/statix — see CLAUDE.md) is a separate, independently-versioned
|
||||||
|
# reference on purpose: it doesn't have to track the flake's own nixpkgs
|
||||||
|
# input, since the tooling just needs to build, not match. Bump it with
|
||||||
|
# --tooling if you want it moved too; the default run leaves it alone.
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
repo_root="$(cd "$(dirname "$0")/.." && pwd)"
|
||||||
|
flake_nix="${repo_root}/flake.nix"
|
||||||
|
maintenance_sh="${repo_root}/scripts/codex-maintenance.sh"
|
||||||
|
claude_md="${repo_root}/CLAUDE.md"
|
||||||
|
|
||||||
|
usage() {
|
||||||
|
cat <<EOF
|
||||||
|
Usage: $0 <release> [--tooling <release>]
|
||||||
|
|
||||||
|
<release> New NixOS release for flake.nix's nixpkgs.url and
|
||||||
|
home-manager.url, e.g. 26.11
|
||||||
|
--tooling <release> Also bump scripts/codex-maintenance.sh's separate
|
||||||
|
nixpkgs-fmt/statix pin (and its mention in
|
||||||
|
CLAUDE.md) to this release. Independent of the
|
||||||
|
first argument — pass the same value if you want
|
||||||
|
both in sync, a different one if you don't.
|
||||||
|
|
||||||
|
Examples:
|
||||||
|
$0 26.11
|
||||||
|
$0 26.11 --tooling 26.11
|
||||||
|
EOF
|
||||||
|
}
|
||||||
|
|
||||||
|
release_re='^[0-9]{2}\.(05|11)$'
|
||||||
|
|
||||||
|
if [[ $# -eq 0 || "$1" == "-h" || "$1" == "--help" ]]; then
|
||||||
|
usage
|
||||||
|
exit 0
|
||||||
|
fi
|
||||||
|
|
||||||
|
new_release="$1"
|
||||||
|
shift
|
||||||
|
|
||||||
|
tooling_release=""
|
||||||
|
while [[ $# -gt 0 ]]; do
|
||||||
|
case "$1" in
|
||||||
|
--tooling)
|
||||||
|
tooling_release="${2:?--tooling requires a release argument}"
|
||||||
|
shift 2
|
||||||
|
;;
|
||||||
|
*)
|
||||||
|
echo "Unknown argument: $1" >&2
|
||||||
|
usage >&2
|
||||||
|
exit 1
|
||||||
|
;;
|
||||||
|
esac
|
||||||
|
done
|
||||||
|
|
||||||
|
check_release_format() {
|
||||||
|
local release="$1"
|
||||||
|
if [[ ! "$release" =~ $release_re ]]; then
|
||||||
|
echo "ERROR: '$release' doesn't look like a NixOS release (expected e.g. 26.11)" >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
check_branch_exists() {
|
||||||
|
local repo_url="$1" branch="$2"
|
||||||
|
echo "Checking '$branch' exists on $repo_url..."
|
||||||
|
if ! git ls-remote --exit-code --heads "$repo_url" "$branch" >/dev/null; then
|
||||||
|
echo "ERROR: branch '$branch' not found on $repo_url. Typo, or not cut yet?" >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
check_release_format "$new_release"
|
||||||
|
|
||||||
|
current_release="$(grep -oE 'nixos-[0-9]{2}\.[0-9]{2}' "$flake_nix" | head -1 | sed 's/^nixos-//')"
|
||||||
|
if [[ -z "$current_release" ]]; then
|
||||||
|
echo "ERROR: couldn't find flake.nix's current nixpkgs release" >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
if [[ "$current_release" == "$new_release" ]]; then
|
||||||
|
echo "flake.nix is already on $new_release."
|
||||||
|
else
|
||||||
|
echo "Bumping flake.nix's nixpkgs/home-manager release: $current_release -> $new_release"
|
||||||
|
check_branch_exists "https://github.com/NixOS/nixpkgs.git" "nixos-$new_release"
|
||||||
|
check_branch_exists "https://github.com/nix-community/home-manager.git" "release-$new_release"
|
||||||
|
|
||||||
|
sed -i \
|
||||||
|
-e "s|github:NixOS/nixpkgs/nixos-${current_release}|github:NixOS/nixpkgs/nixos-${new_release}|" \
|
||||||
|
-e "s|github:nix-community/home-manager/release-${current_release}|github:nix-community/home-manager/release-${new_release}|" \
|
||||||
|
"$flake_nix"
|
||||||
|
|
||||||
|
echo "Updated:"
|
||||||
|
grep -n "nixos-${new_release}\|release-${new_release}" "$flake_nix"
|
||||||
|
fi
|
||||||
|
|
||||||
|
if [[ -n "$tooling_release" ]]; then
|
||||||
|
check_release_format "$tooling_release"
|
||||||
|
|
||||||
|
current_tooling_release="$(grep -oE 'nixos-[0-9]{2}\.[0-9]{2}' "$maintenance_sh" | head -1 | sed 's/^nixos-//')"
|
||||||
|
|
||||||
|
if [[ "$current_tooling_release" == "$tooling_release" ]]; then
|
||||||
|
echo "codex-maintenance.sh's tooling pin is already on $tooling_release."
|
||||||
|
else
|
||||||
|
echo "Bumping codex-maintenance.sh's nixpkgs-fmt/statix pin: $current_tooling_release -> $tooling_release"
|
||||||
|
check_branch_exists "https://github.com/NixOS/nixpkgs.git" "nixos-$tooling_release"
|
||||||
|
|
||||||
|
sed -i "s|github:NixOS/nixpkgs/nixos-${current_tooling_release}|github:NixOS/nixpkgs/nixos-${tooling_release}|g" \
|
||||||
|
"$maintenance_sh"
|
||||||
|
sed -i "s|nixos-${current_tooling_release}|nixos-${tooling_release}|g" \
|
||||||
|
"$claude_md"
|
||||||
|
|
||||||
|
echo "Updated:"
|
||||||
|
grep -n "nixos-${tooling_release}" "$maintenance_sh" "$claude_md"
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
|
||||||
|
echo
|
||||||
|
echo "flake.lock still points at the old input revisions until refreshed. Either:"
|
||||||
|
echo " nix flake update nixpkgs home-manager # just these two inputs"
|
||||||
|
echo " nix flake update # everything — see docs/flake-lock-automation.md"
|
||||||
|
echo
|
||||||
|
echo "Then run 'bash scripts/codex-maintenance.sh --full-check --dry-run' before"
|
||||||
|
echo "committing — a channel bump can shift option defaults across every host,"
|
||||||
|
echo "and only --dry-run actually builds anything to catch that."
|
||||||
+293
-31
@@ -1,22 +1,84 @@
|
|||||||
#!/usr/bin/env bash
|
#!/usr/bin/env bash
|
||||||
|
# Validation entry point for CI and local/agent review.
|
||||||
|
#
|
||||||
|
# Default mode (what CI runs on every push/PR): fmt-check, statix, and eval
|
||||||
|
# are scoped to files that actually changed against a base ref, plus
|
||||||
|
# whichever hosts/packages those changes can affect. This exists because
|
||||||
|
# the unscoped sweep below is slow enough to time out CI runners -- see
|
||||||
|
# --full-check.
|
||||||
|
#
|
||||||
|
# --full-check: the historical full sweep (every host, every package,
|
||||||
|
# fmt --check ./statix check . over the whole tree). Slow -- minutes, not
|
||||||
|
# seconds. CI never passes this; run it locally before a release or after
|
||||||
|
# touching modules/common/*, flake.nix, or variables.nix if you want extra
|
||||||
|
# confidence beyond what the changed-files scope already covers for those
|
||||||
|
# paths (see below).
|
||||||
|
#
|
||||||
|
# --dry-run: adds `nix build --dry-run --no-link` for whatever scope is
|
||||||
|
# active (changed-files scope by default, full scope under --full-check).
|
||||||
|
#
|
||||||
|
# Per-host/per-package eval and dry-run build calls run concurrently (see
|
||||||
|
# scripts/lib/nix-parallel.sh) since they're independent of each other.
|
||||||
|
# Concurrency defaults to core count capped by available memory (~1GB/job)
|
||||||
|
# rather than plain core count, since each concurrent `nix eval` evaluates a
|
||||||
|
# whole NixOS system closure and can OOM a small/memory-constrained CI
|
||||||
|
# runner otherwise; override via NIX_PARALLEL_JOBS if a runner has more (or
|
||||||
|
# less) room than that estimate assumes.
|
||||||
set -euo pipefail
|
set -euo pipefail
|
||||||
|
|
||||||
export NIX_CONFIG="${NIX_CONFIG:-}
|
script_dir="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||||
experimental-features = nix-command flakes
|
# shellcheck source=lib/nix-bootstrap.sh
|
||||||
accept-flake-config = false
|
source "${script_dir}/lib/nix-bootstrap.sh"
|
||||||
warn-dirty = false
|
# shellcheck source=lib/nix-eval.sh
|
||||||
"
|
source "${script_dir}/lib/nix-eval.sh"
|
||||||
|
# shellcheck source=lib/nix-parallel.sh
|
||||||
|
source "${script_dir}/lib/nix-parallel.sh"
|
||||||
|
|
||||||
MODE="${1:-validate}"
|
repo_root="$(cd "${script_dir}/.." && pwd)"
|
||||||
|
cd "$repo_root"
|
||||||
|
|
||||||
ensure_nix_profile() {
|
# When this repo is a subdirectory of a larger git repo (e.g. a mono-repo
|
||||||
if [ -f /nix/var/nix/profiles/default/etc/profile.d/nix-daemon.sh ]; then
|
# subtree), `git diff --name-only` outputs paths relative to the outer git
|
||||||
. /nix/var/nix/profiles/default/etc/profile.d/nix-daemon.sh
|
# root, not this directory. Compute a prefix to strip so pattern matching
|
||||||
elif [ -f "$HOME/.nix-profile/etc/profile.d/nix.sh" ]; then
|
# below works correctly regardless of nesting depth.
|
||||||
. "$HOME/.nix-profile/etc/profile.d/nix.sh"
|
_git_root="$(git rev-parse --show-toplevel 2>/dev/null || echo "$repo_root")"
|
||||||
|
if [[ "$repo_root" != "$_git_root" ]]; then
|
||||||
|
_subtree_prefix="${repo_root#"$_git_root"/}/"
|
||||||
|
else
|
||||||
|
_subtree_prefix=""
|
||||||
fi
|
fi
|
||||||
|
|
||||||
|
full_check=false
|
||||||
|
dry_run=false
|
||||||
|
|
||||||
|
usage() {
|
||||||
|
cat <<'EOF'
|
||||||
|
Usage: scripts/codex-maintenance.sh [--full-check] [--dry-run]
|
||||||
|
|
||||||
|
--full-check Run the full sweep: fmt-check and statix over the whole
|
||||||
|
repo, eval every host and package. Slow. Never run by CI.
|
||||||
|
--dry-run Additionally run `nix build --dry-run --no-link` for
|
||||||
|
whatever scope is active.
|
||||||
|
|
||||||
|
With neither flag (the CI default), fmt-check/statix/eval are scoped to
|
||||||
|
files changed against a base ref (env MAINT_BASE_SHA, else the PR base,
|
||||||
|
else HEAD^), plus the hosts/packages those changes can affect.
|
||||||
|
EOF
|
||||||
}
|
}
|
||||||
|
|
||||||
|
for arg in "$@"; do
|
||||||
|
case "$arg" in
|
||||||
|
--full-check) full_check=true ;;
|
||||||
|
--dry-run) dry_run=true ;;
|
||||||
|
-h|--help) usage; exit 0 ;;
|
||||||
|
*)
|
||||||
|
echo "Unknown argument: $arg" >&2
|
||||||
|
usage >&2
|
||||||
|
exit 1
|
||||||
|
;;
|
||||||
|
esac
|
||||||
|
done
|
||||||
|
|
||||||
ensure_nix_profile
|
ensure_nix_profile
|
||||||
|
|
||||||
if ! command -v nix >/dev/null 2>&1; then
|
if ! command -v nix >/dev/null 2>&1; then
|
||||||
@@ -24,13 +86,6 @@ if ! command -v nix >/dev/null 2>&1; then
|
|||||||
exit 127
|
exit 127
|
||||||
fi
|
fi
|
||||||
|
|
||||||
hosts_json="$(nix eval --json --no-use-registries --no-accept-flake-config .#nixosConfigurations --apply builtins.attrNames)"
|
|
||||||
hosts="$(echo "$hosts_json" | jq -r '.[]')"
|
|
||||||
|
|
||||||
echo "Hosts:"
|
|
||||||
echo "$hosts"
|
|
||||||
|
|
||||||
echo
|
|
||||||
echo "Checking for obvious committed secrets..."
|
echo "Checking for obvious committed secrets..."
|
||||||
if grep -RInE 'github_pat_|ghp_|access-tokens|hashedPassword[[:space:]]*=' \
|
if grep -RInE 'github_pat_|ghp_|access-tokens|hashedPassword[[:space:]]*=' \
|
||||||
--exclude-dir=.git \
|
--exclude-dir=.git \
|
||||||
@@ -42,28 +97,235 @@ else
|
|||||||
echo "No obvious token patterns found."
|
echo "No obvious token patterns found."
|
||||||
fi
|
fi
|
||||||
|
|
||||||
|
mapfile -t all_hosts < <(list_flake_targets .)
|
||||||
|
mapfile -t all_packages < <(nix eval --json "${NIX_EVAL_FLAGS[@]}" .#packages.x86_64-linux --apply builtins.attrNames | jq -r '.[]')
|
||||||
|
|
||||||
|
# host_targets_for_dir <hosts-subdir-name>
|
||||||
|
# Prints the nixosConfigurations target names whose hostPath is
|
||||||
|
# ./hosts/<dir>/host.nix, derived straight from flake.nix's generatedTargets
|
||||||
|
# (one mkTarget { ... } call per line) rather than a hand-maintained table,
|
||||||
|
# so it can't drift the way a copied mapping would.
|
||||||
|
host_targets_for_dir() {
|
||||||
|
local dir="$1"
|
||||||
|
grep -oE '^[[:space:]]*[A-Za-z0-9_-]+ = mkTarget \{[^}]*hostPath = \./hosts/'"${dir}"'/host\.nix;[^}]*\};' flake.nix \
|
||||||
|
| sed -E 's/^[[:space:]]*([A-Za-z0-9_-]+) = mkTarget.*/\1/' \
|
||||||
|
|| true
|
||||||
|
}
|
||||||
|
|
||||||
|
declare -a changed_files=()
|
||||||
|
scope_desc="full repo"
|
||||||
|
|
||||||
|
if ! $full_check; then
|
||||||
|
resolve_base_ref() {
|
||||||
|
if [[ -n "${MAINT_BASE_SHA:-}" ]] && git cat-file -e "${MAINT_BASE_SHA}^{commit}" 2>/dev/null; then
|
||||||
|
echo "$MAINT_BASE_SHA"
|
||||||
|
return
|
||||||
|
fi
|
||||||
|
if git rev-parse --verify -q HEAD^ >/dev/null 2>&1; then
|
||||||
|
echo "HEAD^"
|
||||||
|
return
|
||||||
|
fi
|
||||||
|
git hash-object -t tree /dev/null
|
||||||
|
}
|
||||||
|
|
||||||
|
base_ref="$(resolve_base_ref)"
|
||||||
|
echo
|
||||||
|
echo "Changed-files scope: diffing against ${base_ref}"
|
||||||
|
mapfile -t changed_files < <(git diff --name-only --diff-filter=ACMR "$base_ref" -- . | sort -u | sed "s|^${_subtree_prefix}||")
|
||||||
|
|
||||||
|
if [[ ${#changed_files[@]} -eq 0 ]]; then
|
||||||
|
echo "No changed files detected."
|
||||||
|
else
|
||||||
|
printf ' %s\n' "${changed_files[@]}"
|
||||||
|
fi
|
||||||
|
scope_desc="changed files only (base: ${base_ref})"
|
||||||
|
fi
|
||||||
|
|
||||||
|
# Whole-tree fmt/lint always run under --full-check; otherwise scoped below.
|
||||||
|
declare -a changed_nix_files=()
|
||||||
|
for f in "${changed_files[@]:-}"; do
|
||||||
|
[[ "$f" == *.nix && -f "$f" ]] && changed_nix_files+=("$f")
|
||||||
|
done
|
||||||
|
|
||||||
echo
|
echo
|
||||||
echo "Checking Nix formatting with nixpkgs-fmt..."
|
echo "Checking Nix formatting with nixpkgs-fmt..."
|
||||||
nix run --no-use-registries --no-accept-flake-config github:NixOS/nixpkgs/nixos-25.11#nixpkgs-fmt -- --check .
|
if $full_check; then
|
||||||
|
nix run "${NIX_EVAL_FLAGS[@]}" github:NixOS/nixpkgs/nixos-25.11#nixpkgs-fmt -- --check .
|
||||||
|
elif [[ ${#changed_nix_files[@]} -gt 0 ]]; then
|
||||||
|
nix run "${NIX_EVAL_FLAGS[@]}" github:NixOS/nixpkgs/nixos-25.11#nixpkgs-fmt -- --check "${changed_nix_files[@]}"
|
||||||
|
else
|
||||||
|
echo "No changed .nix files; skipping."
|
||||||
|
fi
|
||||||
|
|
||||||
echo
|
echo
|
||||||
echo "Running statix lint..."
|
echo "Running statix lint..."
|
||||||
nix run --no-use-registries --no-accept-flake-config github:NixOS/nixpkgs/nixos-25.11#statix -- check .
|
if $full_check; then
|
||||||
|
nix run "${NIX_EVAL_FLAGS[@]}" github:NixOS/nixpkgs/nixos-25.11#statix -- check .
|
||||||
|
elif [[ ${#changed_nix_files[@]} -gt 0 ]]; then
|
||||||
|
for f in "${changed_nix_files[@]}"; do
|
||||||
|
nix run "${NIX_EVAL_FLAGS[@]}" github:NixOS/nixpkgs/nixos-25.11#statix -- check "$f"
|
||||||
|
done
|
||||||
|
else
|
||||||
|
echo "No changed .nix files; skipping."
|
||||||
|
fi
|
||||||
|
|
||||||
echo
|
# Figure out which hosts/packages this run needs to eval (and, under
|
||||||
echo "Evaluating host toplevel derivations..."
|
# --dry-run, build). full_check always means "everything"; otherwise a
|
||||||
for host in $hosts; do
|
# change to flake.nix/flake.lock/variables.nix/modules/common/* (repo-wide
|
||||||
echo "==> $host"
|
# inputs) or to any other modules/*.nix outside platforms//build-types
|
||||||
nix eval --raw --no-use-registries --no-accept-flake-config ".#nixosConfigurations.${host}.config.system.build.toplevel.drvPath"
|
# (whose blast radius isn't safely inferable from the path alone -- see
|
||||||
|
# CLAUDE.md's "Grep modules/build-types/*.nix for each build type's imports
|
||||||
|
# list") also falls back to everything, on the same reasoning CLAUDE.md
|
||||||
|
# already gives interactive sessions for when to run the full sweep.
|
||||||
|
# Anything more targeted -- a host.nix, a platform module, a build-type
|
||||||
|
# module -- narrows to just the hosts it can affect.
|
||||||
|
declare -A affected_hosts=()
|
||||||
|
eval_packages=false
|
||||||
|
|
||||||
|
if $full_check; then
|
||||||
|
for h in "${all_hosts[@]}"; do affected_hosts[$h]=1; done
|
||||||
|
eval_packages=true
|
||||||
|
else
|
||||||
|
full_fallback=false
|
||||||
|
for f in "${changed_files[@]:-}"; do
|
||||||
|
case "$f" in
|
||||||
|
flake.nix|flake.lock|variables.nix|modules/common/*)
|
||||||
|
full_fallback=true
|
||||||
|
;;
|
||||||
|
esac
|
||||||
done
|
done
|
||||||
|
|
||||||
if [[ "$MODE" == "dry-run" ]]; then
|
if ! $full_fallback; then
|
||||||
echo
|
for f in "${changed_files[@]:-}"; do
|
||||||
echo "Running dry-run builds for all hosts. This will not create result symlinks."
|
case "$f" in
|
||||||
for host in $hosts; do
|
hosts/*/*)
|
||||||
echo "==> Dry-run build: $host"
|
hostdir="${f#hosts/}"
|
||||||
nix build --dry-run --no-link --no-use-registries --no-accept-flake-config ".#nixosConfigurations.${host}.config.system.build.toplevel"
|
hostdir="${hostdir%%/*}"
|
||||||
|
while IFS= read -r t; do
|
||||||
|
[[ -n "$t" ]] && affected_hosts[$t]=1
|
||||||
|
done < <(host_targets_for_dir "$hostdir")
|
||||||
|
;;
|
||||||
|
modules/platforms/*.nix)
|
||||||
|
platform="$(basename "$f" .nix)"
|
||||||
|
for h in "${all_hosts[@]}"; do
|
||||||
|
[[ "$h" == "${platform}-"* ]] && affected_hosts[$h]=1
|
||||||
done
|
done
|
||||||
|
;;
|
||||||
|
modules/build-types/*.nix)
|
||||||
|
buildtype="$(basename "$f" .nix)"
|
||||||
|
for h in "${all_hosts[@]}"; do
|
||||||
|
[[ "$h" == *"-${buildtype}" ]] && affected_hosts[$h]=1
|
||||||
|
done
|
||||||
|
;;
|
||||||
|
modules/installer/*)
|
||||||
|
# iso.nix (imported by both the "installer" nixosConfigurations
|
||||||
|
# target and netbootSystem, which backs packages.pxe) pulls in
|
||||||
|
# common.nix, so a common.nix change reaches all three.
|
||||||
|
affected_hosts[installer]=1
|
||||||
|
eval_packages=true
|
||||||
|
;;
|
||||||
|
modules/pxe-boot/*)
|
||||||
|
# stage-installer-artifacts.nix is imported by
|
||||||
|
# modules/build-types/pxe-boot.nix only -- same blast radius as a
|
||||||
|
# build-types/*.nix change, not a packages one.
|
||||||
|
for h in "${all_hosts[@]}"; do
|
||||||
|
[[ "$h" == *"-pxe-boot" ]] && affected_hosts[$h]=1
|
||||||
|
done
|
||||||
|
;;
|
||||||
|
modules/*)
|
||||||
|
full_fallback=true
|
||||||
|
;;
|
||||||
|
esac
|
||||||
|
done
|
||||||
|
fi
|
||||||
|
|
||||||
|
if $full_fallback; then
|
||||||
|
echo
|
||||||
|
echo "Changed files affect shared config; falling back to evaluating every host/package."
|
||||||
|
for h in "${all_hosts[@]}"; do affected_hosts[$h]=1; done
|
||||||
|
eval_packages=true
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
|
||||||
|
mapfile -t hosts < <(for h in "${!affected_hosts[@]}"; do echo "$h"; done | sort)
|
||||||
|
|
||||||
|
echo
|
||||||
|
echo "Checking nix-cache host key for drift..."
|
||||||
|
if bash "${script_dir}/secrets/sync-nix-cache-host-key.sh" --check; then
|
||||||
|
:
|
||||||
|
else
|
||||||
|
drift_status=$?
|
||||||
|
if [[ "$drift_status" -eq 2 ]]; then
|
||||||
|
echo "nix-cache unreachable from here -- skipping host-key drift check."
|
||||||
|
else
|
||||||
|
echo "WARNING: nix-cache's host key has drifted from variables.nix (see above)." >&2
|
||||||
|
echo " Run 'bash scripts/secrets/sync-nix-cache-host-key.sh' to fix." >&2
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
|
||||||
|
echo
|
||||||
|
if [[ ${#hosts[@]} -eq 0 ]]; then
|
||||||
|
echo "No hosts affected by changed files; skipping host eval."
|
||||||
|
else
|
||||||
|
echo "Evaluating host toplevel derivations (${scope_desc}, up to ${NIX_PARALLEL_JOBS} at a time)..."
|
||||||
|
# lxc-* hosts deploy via a directly pct-restore-able tarball instead of
|
||||||
|
# nixos-install (see docs/auto-installer.md); proxmox-* hosts can
|
||||||
|
# alternatively be built as a standalone disk image (see
|
||||||
|
# docs/proxmox-images.md). Both are otherwise-unvalidated buildable
|
||||||
|
# surface, easy to silently break without this.
|
||||||
|
declare -a host_eval_jobs=()
|
||||||
|
for host in "${hosts[@]}"; do
|
||||||
|
host_eval_jobs+=("${host}${NIX_PARALLEL_SEP}.#nixosConfigurations.${host}.config.system.build.toplevel.drvPath")
|
||||||
|
case "$host" in
|
||||||
|
lxc-*)
|
||||||
|
host_eval_jobs+=("${host} (tarball)${NIX_PARALLEL_SEP}.#nixosConfigurations.${host}.config.system.build.tarball.drvPath")
|
||||||
|
;;
|
||||||
|
proxmox-*)
|
||||||
|
host_eval_jobs+=("${host} (diskoImagesScript)${NIX_PARALLEL_SEP}.#nixosConfigurations.${host}.config.system.build.diskoImagesScript.drvPath")
|
||||||
|
;;
|
||||||
|
esac
|
||||||
|
done
|
||||||
|
run_nix_parallel host_eval_jobs eval --raw "${NIX_EVAL_FLAGS[@]}"
|
||||||
|
fi
|
||||||
|
|
||||||
|
echo
|
||||||
|
if ! $eval_packages; then
|
||||||
|
echo "No packages affected by changed files; skipping package eval."
|
||||||
|
else
|
||||||
|
echo "Evaluating buildable packages (up to ${NIX_PARALLEL_JOBS} at a time)..."
|
||||||
|
declare -a package_eval_jobs=()
|
||||||
|
for pkg in "${all_packages[@]}"; do
|
||||||
|
package_eval_jobs+=("packages.x86_64-linux.${pkg}${NIX_PARALLEL_SEP}.#packages.x86_64-linux.${pkg}")
|
||||||
|
done
|
||||||
|
run_nix_parallel package_eval_jobs eval --raw "${NIX_EVAL_FLAGS[@]}"
|
||||||
|
fi
|
||||||
|
|
||||||
|
if $dry_run; then
|
||||||
|
echo
|
||||||
|
echo "Running dry-run builds for the active scope (up to ${NIX_PARALLEL_JOBS} at a time). This will not create result symlinks."
|
||||||
|
declare -a host_build_jobs=()
|
||||||
|
for host in "${hosts[@]:-}"; do
|
||||||
|
host_build_jobs+=("Dry-run build: ${host}${NIX_PARALLEL_SEP}.#nixosConfigurations.${host}.config.system.build.toplevel")
|
||||||
|
case "$host" in
|
||||||
|
lxc-*)
|
||||||
|
host_build_jobs+=("Dry-run build: ${host} (tarball)${NIX_PARALLEL_SEP}.#nixosConfigurations.${host}.config.system.build.tarball")
|
||||||
|
;;
|
||||||
|
proxmox-*)
|
||||||
|
host_build_jobs+=("Dry-run build: ${host} (diskoImagesScript)${NIX_PARALLEL_SEP}.#nixosConfigurations.${host}.config.system.build.diskoImagesScript")
|
||||||
|
;;
|
||||||
|
esac
|
||||||
|
done
|
||||||
|
run_nix_parallel host_build_jobs build --dry-run --no-link "${NIX_EVAL_FLAGS[@]}"
|
||||||
|
|
||||||
|
if $eval_packages; then
|
||||||
|
echo
|
||||||
|
echo "Running dry-run builds for packages."
|
||||||
|
declare -a package_build_jobs=()
|
||||||
|
for pkg in "${all_packages[@]}"; do
|
||||||
|
package_build_jobs+=("Dry-run build: packages.x86_64-linux.${pkg}${NIX_PARALLEL_SEP}.#packages.x86_64-linux.${pkg}")
|
||||||
|
done
|
||||||
|
run_nix_parallel package_build_jobs build --dry-run --no-link "${NIX_EVAL_FLAGS[@]}"
|
||||||
|
fi
|
||||||
fi
|
fi
|
||||||
|
|
||||||
echo
|
echo
|
||||||
|
|||||||
+19
-22
@@ -1,19 +1,11 @@
|
|||||||
#!/usr/bin/env bash
|
#!/usr/bin/env bash
|
||||||
set -euo pipefail
|
set -euo pipefail
|
||||||
|
|
||||||
export NIX_CONFIG="${NIX_CONFIG:-}
|
script_dir="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||||
experimental-features = nix-command flakes
|
# shellcheck source=lib/nix-bootstrap.sh
|
||||||
accept-flake-config = false
|
source "${script_dir}/lib/nix-bootstrap.sh"
|
||||||
warn-dirty = false
|
# shellcheck source=lib/nix-eval.sh
|
||||||
"
|
source "${script_dir}/lib/nix-eval.sh"
|
||||||
|
|
||||||
ensure_nix_profile() {
|
|
||||||
if [ -f /nix/var/nix/profiles/default/etc/profile.d/nix-daemon.sh ]; then
|
|
||||||
. /nix/var/nix/profiles/default/etc/profile.d/nix-daemon.sh
|
|
||||||
elif [ -f "$HOME/.nix-profile/etc/profile.d/nix.sh" ]; then
|
|
||||||
. "$HOME/.nix-profile/etc/profile.d/nix.sh"
|
|
||||||
fi
|
|
||||||
}
|
|
||||||
|
|
||||||
install_nix_if_missing() {
|
install_nix_if_missing() {
|
||||||
if command -v nix >/dev/null 2>&1; then
|
if command -v nix >/dev/null 2>&1; then
|
||||||
@@ -49,6 +41,17 @@ warn-dirty = false
|
|||||||
build-users-group = nixbld
|
build-users-group = nixbld
|
||||||
EOF
|
EOF
|
||||||
|
|
||||||
|
# The official installer's single-user root path still shells out to
|
||||||
|
# `sudo` to create /nix even though it already knows it's running as
|
||||||
|
# root -- confirmed live against a sudo-less minimal Debian/Proxmox
|
||||||
|
# node, where it fails with "sudo: not found" and prints this exact
|
||||||
|
# mkdir/chown as the manual fix. Pre-create it so that branch of the
|
||||||
|
# installer is skipped entirely.
|
||||||
|
if [ ! -d /nix ]; then
|
||||||
|
mkdir -m 0755 /nix
|
||||||
|
chown root /nix
|
||||||
|
fi
|
||||||
|
|
||||||
sh <(curl -L https://nixos.org/nix/install) --no-daemon
|
sh <(curl -L https://nixos.org/nix/install) --no-daemon
|
||||||
else
|
else
|
||||||
sh <(curl -L https://nixos.org/nix/install) --no-daemon
|
sh <(curl -L https://nixos.org/nix/install) --no-daemon
|
||||||
@@ -65,6 +68,7 @@ cat > "$HOME/.config/nix/nix.conf" <<'EOF'
|
|||||||
experimental-features = nix-command flakes
|
experimental-features = nix-command flakes
|
||||||
accept-flake-config = false
|
accept-flake-config = false
|
||||||
warn-dirty = false
|
warn-dirty = false
|
||||||
|
build-users-group =
|
||||||
EOF
|
EOF
|
||||||
|
|
||||||
echo "Nix version:"
|
echo "Nix version:"
|
||||||
@@ -79,13 +83,6 @@ if ! command -v jq >/dev/null 2>&1; then
|
|||||||
fi
|
fi
|
||||||
|
|
||||||
echo "Available NixOS hosts:"
|
echo "Available NixOS hosts:"
|
||||||
hosts="$(nix eval --json --no-use-registries --no-accept-flake-config .#nixosConfigurations --apply builtins.attrNames | jq -r '.[]')"
|
list_flake_targets .
|
||||||
echo "$hosts"
|
|
||||||
|
|
||||||
echo "Evaluating all host toplevel derivations..."
|
echo "Codex setup complete. Run bash scripts/codex-maintenance.sh to validate changes."
|
||||||
for host in $hosts; do
|
|
||||||
echo "==> Evaluating $host"
|
|
||||||
nix eval --raw --no-use-registries --no-accept-flake-config ".#nixosConfigurations.${host}.config.system.build.toplevel.drvPath"
|
|
||||||
done
|
|
||||||
|
|
||||||
echo "Codex setup complete."
|
|
||||||
|
|||||||
@@ -1,9 +0,0 @@
|
|||||||
#!/usr/bin/env bash
|
|
||||||
set -euo pipefail
|
|
||||||
|
|
||||||
#boot to rescue mode
|
|
||||||
# set root password
|
|
||||||
scp $RESULT_ISO root@$LINODE_IP:/tmp/nixos-auto.iso
|
|
||||||
|
|
||||||
#in LISH or ssh to rescue mode
|
|
||||||
dd if=/tmp/nixos.iso of=/dev/sda bs=4M status=progress conv=fsync
|
|
||||||
Executable
+595
@@ -0,0 +1,595 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# deploy.sh — Full lifecycle management for the Docker Swarm HA cluster.
|
||||||
|
#
|
||||||
|
# Provisions two NixOS Proxmox VMs (ha-docker-1, ha-docker-2) as dual-manager
|
||||||
|
# Docker Swarm nodes sharing NFS storage from the existing HA file-server
|
||||||
|
# cluster. Both nodes are managers so either can accept Docker API and
|
||||||
|
# `docker stack` commands.
|
||||||
|
#
|
||||||
|
# Usage:
|
||||||
|
# scripts/docker-swarm/deploy.sh [options]
|
||||||
|
# scripts/docker-swarm/deploy.sh --destroy [options]
|
||||||
|
#
|
||||||
|
# Phases (all run by default; skip any with --skip-<phase>):
|
||||||
|
# 1. ensure-bridge Create vmbr3 (swarm cluster bridge) on the Proxmox node.
|
||||||
|
# 2. sync-keys Generate SSH host keys for both nodes (clan vars).
|
||||||
|
# 3. ipa-hosts Create IPA host objects + sops-encrypted keytabs.
|
||||||
|
# 4. create-vms Build NixOS disk images and create VMs via create-proxmox-resource.sh.
|
||||||
|
# 5. add-hardware Attach vmbr2 (storage) and vmbr3 (swarm) NICs; start VMs.
|
||||||
|
# 6. boot-wait Wait for SSH on both LAN IPs.
|
||||||
|
# 7. refresh-sops-keys Detect disko key drift; re-encrypt secrets; commit.
|
||||||
|
# 8. init-swarm docker swarm init on node1; manager join on node2; label nodes.
|
||||||
|
# 9. dns Register storage.home and swarm.home A records in FreeIPA.
|
||||||
|
# 10. verify docker node ls; NFS mount check; swarm health.
|
||||||
|
#
|
||||||
|
# Options:
|
||||||
|
# --node <host> Proxmox host (default: pve1.sweet.home)
|
||||||
|
# --vmid1 <n> VMID for ha-docker-1 (default: 202)
|
||||||
|
# --vmid2 <n> VMID for ha-docker-2 (default: 203)
|
||||||
|
# --storage <pool> Proxmox storage pool (default: local-zfs)
|
||||||
|
# --swarm-bridge <br> Bridge for Docker Swarm cluster network (default: vmbr3)
|
||||||
|
# --storage-bridge <br> Bridge for NFS storage network (default: vmbr2)
|
||||||
|
# --memory <MB> RAM per node (default: 4096)
|
||||||
|
# --cores <n> vCPUs per node (default: 4)
|
||||||
|
# --skip-ensure-bridge Skip vmbr3 creation/check
|
||||||
|
# --skip-sync-keys Skip sync-host-keys.sh (clan vars already exist)
|
||||||
|
# --skip-ipa-hosts Skip IPA host account creation (keytabs already exist)
|
||||||
|
# --skip-create-vms Skip VM creation (VMs already exist)
|
||||||
|
# --skip-add-hardware Skip NIC attachment (already attached)
|
||||||
|
# --skip-boot-wait Skip boot/SSH wait (VMs already running)
|
||||||
|
# --skip-refresh-sops-keys Skip sops host-key drift fix
|
||||||
|
# --skip-init-swarm Skip swarm initialisation (already initialised)
|
||||||
|
# --skip-dns Skip FreeIPA DNS record creation
|
||||||
|
# --skip-verify Skip post-deploy health checks
|
||||||
|
# --force-rebuild Pass --force-rebuild to create-proxmox-resource.sh
|
||||||
|
# --destroy Stop and delete both VMs (skip all other phases)
|
||||||
|
# --dry-run Print what would run without executing
|
||||||
|
# -h|--help Show this message
|
||||||
|
#
|
||||||
|
# Prerequisites:
|
||||||
|
# - SSH access to the Proxmox node as $PROXMOX_SSH_USER (wayne).
|
||||||
|
# - sops age key in the standard location (used by sync-host-keys.sh).
|
||||||
|
# - SSH access to domain-controller.sweet.home as $PROXMOX_SSH_USER for DNS phase.
|
||||||
|
# - For --skip-sync-keys: clan vars already in vars/per-machine/proxmox-ha-docker-{1,2}/.
|
||||||
|
# - For --skip-ipa-hosts: secrets/ha-docker-{1,2}.keytab already exist and are committed.
|
||||||
|
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||||
|
REPO_ROOT="$(cd "${SCRIPT_DIR}/../.." && pwd)"
|
||||||
|
|
||||||
|
# shellcheck source=../env.sh
|
||||||
|
source "${REPO_ROOT}/scripts/env.sh"
|
||||||
|
|
||||||
|
# ── Defaults ──────────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
NODE="${PVE1_HOST}" # deploy.sh targets pve1 by default (authorised for this cluster)
|
||||||
|
VMID1=202
|
||||||
|
VMID2=203
|
||||||
|
STORAGE="${PROXMOX_STORAGE:-local-zfs}"
|
||||||
|
SWARM_BRIDGE="vmbr3"
|
||||||
|
STORAGE_BRIDGE="vmbr2"
|
||||||
|
MEMORY_MB=4096
|
||||||
|
CORES=4
|
||||||
|
|
||||||
|
SKIP_ENSURE_BRIDGE=false
|
||||||
|
SKIP_SYNC_KEYS=false
|
||||||
|
SKIP_IPA_HOSTS=false
|
||||||
|
SKIP_CREATE_VMS=false
|
||||||
|
SKIP_ADD_HARDWARE=false
|
||||||
|
SKIP_BOOT_WAIT=false
|
||||||
|
SKIP_REFRESH_SOPS_KEYS=false
|
||||||
|
SKIP_INIT_SWARM=false
|
||||||
|
SKIP_DNS=false
|
||||||
|
SKIP_VERIFY=false
|
||||||
|
FORCE_REBUILD=false
|
||||||
|
DESTROY=false
|
||||||
|
DRY_RUN=false
|
||||||
|
|
||||||
|
# ── Variables from repo (mirrors variables.nix) ───────────────────────────────
|
||||||
|
|
||||||
|
NODE1_HOST="ha-docker-1"
|
||||||
|
NODE2_HOST="ha-docker-2"
|
||||||
|
NODE1_LAN_IP="192.168.2.230"
|
||||||
|
NODE2_LAN_IP="192.168.2.231"
|
||||||
|
NODE1_SWARM_IP="192.168.30.230"
|
||||||
|
NODE2_SWARM_IP="192.168.30.231"
|
||||||
|
NODE1_STORAGE_IP="192.168.20.230"
|
||||||
|
NODE2_STORAGE_IP="192.168.20.231"
|
||||||
|
SWARM_CIDR="192.168.30.0/24"
|
||||||
|
STORAGE_CIDR="192.168.20.0/24"
|
||||||
|
STORAGE_ZONE="storage.home"
|
||||||
|
SWARM_ZONE="swarm.home"
|
||||||
|
SSH_USER="${PROXMOX_SSH_USER:-wayne}"
|
||||||
|
DC_HOST="${IPA_SERVER:-domain-controller.sweet.home}"
|
||||||
|
|
||||||
|
# ── Argument parsing ──────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
usage() {
|
||||||
|
sed -n '/^# Usage:/,/^[^#]/{ /^#/{ s/^# \?//; p } }' "$0"
|
||||||
|
exit "${1:-0}"
|
||||||
|
}
|
||||||
|
|
||||||
|
while [[ $# -gt 0 ]]; do
|
||||||
|
case "$1" in
|
||||||
|
--node) NODE="$2"; shift 2 ;;
|
||||||
|
--vmid1) VMID1="$2"; shift 2 ;;
|
||||||
|
--vmid2) VMID2="$2"; shift 2 ;;
|
||||||
|
--storage) STORAGE="$2"; shift 2 ;;
|
||||||
|
--swarm-bridge) SWARM_BRIDGE="$2"; shift 2 ;;
|
||||||
|
--storage-bridge) STORAGE_BRIDGE="$2"; shift 2 ;;
|
||||||
|
--memory) MEMORY_MB="$2"; shift 2 ;;
|
||||||
|
--cores) CORES="$2"; shift 2 ;;
|
||||||
|
--skip-ensure-bridge) SKIP_ENSURE_BRIDGE=true; shift ;;
|
||||||
|
--skip-sync-keys) SKIP_SYNC_KEYS=true; shift ;;
|
||||||
|
--skip-ipa-hosts) SKIP_IPA_HOSTS=true; shift ;;
|
||||||
|
--skip-create-vms) SKIP_CREATE_VMS=true; shift ;;
|
||||||
|
--skip-add-hardware) SKIP_ADD_HARDWARE=true; shift ;;
|
||||||
|
--skip-boot-wait) SKIP_BOOT_WAIT=true; shift ;;
|
||||||
|
--skip-refresh-sops-keys) SKIP_REFRESH_SOPS_KEYS=true; shift ;;
|
||||||
|
--skip-init-swarm) SKIP_INIT_SWARM=true; shift ;;
|
||||||
|
--skip-dns) SKIP_DNS=true; shift ;;
|
||||||
|
--skip-verify) SKIP_VERIFY=true; shift ;;
|
||||||
|
--force-rebuild) FORCE_REBUILD=true; shift ;;
|
||||||
|
--destroy) DESTROY=true; shift ;;
|
||||||
|
--dry-run) DRY_RUN=true; shift ;;
|
||||||
|
-h|--help) usage 0 ;;
|
||||||
|
*) echo "Unknown option: $1" >&2; usage 1 ;;
|
||||||
|
esac
|
||||||
|
done
|
||||||
|
|
||||||
|
# ── Helpers ───────────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
log() { echo "==> $*"; }
|
||||||
|
logn() { echo " $*"; }
|
||||||
|
err() { echo "ERROR: $*" >&2; exit 1; }
|
||||||
|
|
||||||
|
run() {
|
||||||
|
if $DRY_RUN; then
|
||||||
|
echo "[dry-run] $*"
|
||||||
|
else
|
||||||
|
"$@"
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
pve() {
|
||||||
|
if $DRY_RUN; then
|
||||||
|
echo "[dry-run] ssh ${SSH_USER}@${NODE} sudo $*"
|
||||||
|
else
|
||||||
|
ssh -i ~/.ssh/id_ed25519 "${SSH_USER}@${NODE}" "sudo $*"
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
pve_check() {
|
||||||
|
# Read-only probe — always executes even in dry-run.
|
||||||
|
ssh -i ~/.ssh/id_ed25519 "${SSH_USER}@${NODE}" "sudo $*"
|
||||||
|
}
|
||||||
|
|
||||||
|
SWARM_USER="nixos"
|
||||||
|
|
||||||
|
n1() {
|
||||||
|
ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no -o ConnectTimeout=5 \
|
||||||
|
"${SWARM_USER}@${NODE1_LAN_IP}" "$@" 2>/dev/null
|
||||||
|
}
|
||||||
|
|
||||||
|
n2() {
|
||||||
|
ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no -o ConnectTimeout=5 \
|
||||||
|
"${SWARM_USER}@${NODE2_LAN_IP}" "$@" 2>/dev/null
|
||||||
|
}
|
||||||
|
|
||||||
|
dc() {
|
||||||
|
# Run ipa commands on domain-controller as $SSH_USER.
|
||||||
|
if $DRY_RUN; then
|
||||||
|
echo "[dry-run] ssh ${SSH_USER}@${DC_HOST} $*"
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
|
ssh -i ~/.ssh/id_ed25519 "${SSH_USER}@${DC_HOST}" "$@"
|
||||||
|
}
|
||||||
|
|
||||||
|
wait_for_ssh() {
|
||||||
|
local ip="$1" label="$2"
|
||||||
|
if $DRY_RUN; then
|
||||||
|
logn "[dry-run] Skipping SSH wait for ${label} (${ip})"
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
|
local deadline=$(( $(date +%s) + 300 ))
|
||||||
|
log "Waiting for SSH on ${label} (${ip}) — up to 5 min..."
|
||||||
|
while [[ $(date +%s) -lt $deadline ]]; do
|
||||||
|
if ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no -o ConnectTimeout=3 \
|
||||||
|
-o BatchMode=yes "${SWARM_USER}@${ip}" true 2>/dev/null; then
|
||||||
|
logn "${label} is up."
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
|
sleep 5
|
||||||
|
done
|
||||||
|
err "Timed out waiting for SSH on ${label} (${ip})"
|
||||||
|
}
|
||||||
|
|
||||||
|
# ── Destroy mode ──────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
if $DESTROY; then
|
||||||
|
log "Destroying Docker Swarm VMs (${VMID1}=${NODE1_HOST}, ${VMID2}=${NODE2_HOST}) on ${NODE}"
|
||||||
|
for vmid in "$VMID1" "$VMID2"; do
|
||||||
|
STATUS=$(pve "qm status ${vmid} 2>/dev/null" 2>/dev/null || true)
|
||||||
|
if echo "$STATUS" | grep -q "running"; then
|
||||||
|
log "Stopping VMID ${vmid}..."
|
||||||
|
pve "qm stop ${vmid} --skiplock 1"
|
||||||
|
sleep 5
|
||||||
|
fi
|
||||||
|
if $DRY_RUN || pve "qm config ${vmid} >/dev/null 2>&1"; then
|
||||||
|
log "Deleting VMID ${vmid}..."
|
||||||
|
run pve "qm destroy ${vmid} --purge 1"
|
||||||
|
else
|
||||||
|
logn "VMID ${vmid} not found — already gone."
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
log "Done — swarm VMs destroyed."
|
||||||
|
exit 0
|
||||||
|
fi
|
||||||
|
|
||||||
|
# ── Phase 1: Ensure swarm bridge ──────────────────────────────────────────────
|
||||||
|
|
||||||
|
if ! $SKIP_ENSURE_BRIDGE; then
|
||||||
|
log "Phase 1: Ensuring swarm bridge ${SWARM_BRIDGE} on ${NODE}"
|
||||||
|
if pve_check "test -d /sys/class/net/${SWARM_BRIDGE}" &>/dev/null; then
|
||||||
|
logn "${SWARM_BRIDGE} already exists — skipping."
|
||||||
|
else
|
||||||
|
logn "Creating isolated internal bridge ${SWARM_BRIDGE} (no upstream port, ${SWARM_CIDR})"
|
||||||
|
BRIDGE_CONF="auto ${SWARM_BRIDGE}
|
||||||
|
iface ${SWARM_BRIDGE} inet manual
|
||||||
|
bridge-ports none
|
||||||
|
bridge-stp off
|
||||||
|
bridge-fd 0"
|
||||||
|
if $DRY_RUN; then
|
||||||
|
echo "[dry-run] Would write /etc/network/interfaces.d/${SWARM_BRIDGE}.conf and ifup it"
|
||||||
|
else
|
||||||
|
ssh -i ~/.ssh/id_ed25519 "${SSH_USER}@${NODE}" \
|
||||||
|
"echo '${BRIDGE_CONF}' | sudo tee /etc/network/interfaces.d/${SWARM_BRIDGE}.conf > /dev/null && sudo ifup ${SWARM_BRIDGE}"
|
||||||
|
logn "${SWARM_BRIDGE} created and brought up."
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
|
||||||
|
# ── Phase 2: Sync host keys ───────────────────────────────────────────────────
|
||||||
|
|
||||||
|
if ! $SKIP_SYNC_KEYS; then
|
||||||
|
log "Phase 2: Syncing SSH host keys for both swarm targets"
|
||||||
|
for target in proxmox-ha-docker-1 proxmox-ha-docker-2; do
|
||||||
|
CLAN_DIR="${REPO_ROOT}/vars/per-machine/${target}/openssh"
|
||||||
|
if [[ -d "$CLAN_DIR" ]]; then
|
||||||
|
logn "Clan vars for ${target} already exist — skipping."
|
||||||
|
else
|
||||||
|
logn "Generating host keys for ${target}..."
|
||||||
|
run bash "${REPO_ROOT}/scripts/secrets/sync-host-keys.sh" "$target"
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
fi
|
||||||
|
|
||||||
|
# ── Phase 3: IPA host accounts ────────────────────────────────────────────────
|
||||||
|
|
||||||
|
if ! $SKIP_IPA_HOSTS; then
|
||||||
|
log "Phase 3: Creating IPA host accounts and keytabs"
|
||||||
|
IPA_SCRIPT="${REPO_ROOT}/scripts/ipa/create-nixos-ipa-host-account.sh"
|
||||||
|
for host in "${NODE1_HOST}" "${NODE2_HOST}"; do
|
||||||
|
KEYTAB="${REPO_ROOT}/secrets/${host}.keytab"
|
||||||
|
if [[ -f "$KEYTAB" ]]; then
|
||||||
|
logn "Keytab for ${host} already exists — skipping."
|
||||||
|
else
|
||||||
|
logn "Creating IPA host account and keytab for ${host}..."
|
||||||
|
run bash "$IPA_SCRIPT" "$host"
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
|
||||||
|
if ! $DRY_RUN; then
|
||||||
|
# Keytabs must be committed and pushed before VMs rebuild from Gitea.
|
||||||
|
CURRENT_BRANCH="$(git -C "$REPO_ROOT" rev-parse --abbrev-ref HEAD)"
|
||||||
|
logn "Committing keytabs and pushing to Gitea (branch: ${CURRENT_BRANCH})..."
|
||||||
|
(cd "${REPO_ROOT}" && \
|
||||||
|
git add secrets/ha-docker-1.keytab secrets/ha-docker-2.keytab .sops.yaml && \
|
||||||
|
git commit -m "secrets(ha-docker): add IPA keytabs for ha-docker-1 and ha-docker-2" || true && \
|
||||||
|
git push origin "${CURRENT_BRANCH}")
|
||||||
|
logn "Pushed."
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
|
||||||
|
# ── Phase 3.5: Prepare Proxmox node for building ─────────────────────────────
|
||||||
|
|
||||||
|
if ! $SKIP_CREATE_VMS && ! $DRY_RUN; then
|
||||||
|
CURRENT_BRANCH="$(git -C "$REPO_ROOT" rev-parse --abbrev-ref HEAD)"
|
||||||
|
|
||||||
|
local_ssh() { ssh -i ~/.ssh/id_ed25519 "${SSH_USER}@${NODE}" "$*"; }
|
||||||
|
if local_ssh "test -d /nix" &>/dev/null && ! local_ssh "test -w /nix" &>/dev/null; then
|
||||||
|
logn "/nix exists but not writable by ${SSH_USER} — fixing ownership with sudo..."
|
||||||
|
local_ssh "sudo chown -R ${SSH_USER} /nix"
|
||||||
|
logn "Done."
|
||||||
|
fi
|
||||||
|
unset -f local_ssh
|
||||||
|
|
||||||
|
# Use PROXMOX_REMOTE_REPO_DIR (from env.sh) so the build path is consistent
|
||||||
|
# with what create-proxmox-resource.sh will use. The default is
|
||||||
|
# /home/<user>/nixos (the standalone nixos repo clone on pve1), but can be
|
||||||
|
# overridden to e.g. /home/<user>/infrastructure/nixos when the infrastructure
|
||||||
|
# mono-repo is checked out on pve1 instead.
|
||||||
|
REMOTE_REPO="${PROXMOX_REMOTE_REPO_DIR:-/home/${SSH_USER}/nixos}"
|
||||||
|
if pve_check "test -d ${REMOTE_REPO}/.git" &>/dev/null; then
|
||||||
|
REMOTE_BRANCH=$(ssh -i ~/.ssh/id_ed25519 "${SSH_USER}@${NODE}" \
|
||||||
|
"cd ${REMOTE_REPO} && git rev-parse --abbrev-ref HEAD 2>/dev/null")
|
||||||
|
if [[ "$REMOTE_BRANCH" != "$CURRENT_BRANCH" ]]; then
|
||||||
|
logn "Remote repo (${REMOTE_REPO}) is on '${REMOTE_BRANCH}', switching to '${CURRENT_BRANCH}'..."
|
||||||
|
if ssh -i ~/.ssh/id_ed25519 "${SSH_USER}@${NODE}" \
|
||||||
|
"cd ${REMOTE_REPO} && git fetch origin && git checkout '${CURRENT_BRANCH}' && git pull --ff-only" 2>&1; then
|
||||||
|
logn "Done."
|
||||||
|
else
|
||||||
|
logn "WARNING: branch switch failed — proceeding anyway (create-proxmox-resource.sh will retry)"
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
else
|
||||||
|
logn "Remote repo ${REMOTE_REPO} not found on ${NODE} — create-proxmox-resource.sh will clone it."
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
|
||||||
|
# ── Phase 4: Create VMs ───────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
if ! $SKIP_CREATE_VMS; then
|
||||||
|
log "Phase 4: Building and creating swarm VMs on ${NODE}"
|
||||||
|
CREATE="${REPO_ROOT}/scripts/proxmox/create-proxmox-resource.sh"
|
||||||
|
REBUILD_FLAG=""
|
||||||
|
$FORCE_REBUILD && REBUILD_FLAG="--force-rebuild"
|
||||||
|
|
||||||
|
for spec in "${VMID1}:${NODE1_HOST}:proxmox-ha-docker-1" "${VMID2}:${NODE2_HOST}:proxmox-ha-docker-2"; do
|
||||||
|
IFS=: read -r vmid host_name flake_target <<< "$spec"
|
||||||
|
log "Creating ${flake_target} (VMID ${vmid}) on ${NODE}..."
|
||||||
|
# --force-rebuild is always passed: create-proxmox-resource.sh only calls
|
||||||
|
# sync_remote_host_keys (which bakes the clan-var SSH key into the disk
|
||||||
|
# image) when it actually builds. Reusing a cached image skips that step
|
||||||
|
# and leaves the VM unable to decrypt sops secrets on first boot.
|
||||||
|
run bash "$CREATE" \
|
||||||
|
--type vm \
|
||||||
|
--host "$host_name" \
|
||||||
|
--vmid "$vmid" \
|
||||||
|
--node "$NODE" \
|
||||||
|
--storage "$STORAGE" \
|
||||||
|
--memory "$MEMORY_MB" \
|
||||||
|
--cores "$CORES" \
|
||||||
|
--force-rebuild
|
||||||
|
done
|
||||||
|
fi
|
||||||
|
|
||||||
|
# ── Phase 5: Add NICs and start VMs ──────────────────────────────────────────
|
||||||
|
|
||||||
|
if ! $SKIP_ADD_HARDWARE; then
|
||||||
|
log "Phase 5: Attaching storage (${STORAGE_BRIDGE}) and swarm (${SWARM_BRIDGE}) NICs"
|
||||||
|
for vmid in "$VMID1" "$VMID2"; do
|
||||||
|
logn "VMID ${vmid}: stopping to add NICs..."
|
||||||
|
pve "qm stop ${vmid} --skiplock 1 2>/dev/null; sleep 3" || true
|
||||||
|
|
||||||
|
logn "Adding net1 (${STORAGE_BRIDGE} — NFS storage)..."
|
||||||
|
pve "qm set ${vmid} --net1 virtio,bridge=${STORAGE_BRIDGE},firewall=0"
|
||||||
|
|
||||||
|
logn "Adding net2 (${SWARM_BRIDGE} — Docker Swarm)..."
|
||||||
|
pve "qm set ${vmid} --net2 virtio,bridge=${SWARM_BRIDGE},firewall=0"
|
||||||
|
|
||||||
|
logn "Starting VMID ${vmid}..."
|
||||||
|
pve "qm start ${vmid}"
|
||||||
|
done
|
||||||
|
fi
|
||||||
|
|
||||||
|
# ── Phase 6: Wait for SSH ─────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
if ! $SKIP_BOOT_WAIT; then
|
||||||
|
log "Phase 6: Waiting for both nodes to come up on LAN IPs"
|
||||||
|
wait_for_ssh "$NODE1_LAN_IP" "$NODE1_HOST"
|
||||||
|
wait_for_ssh "$NODE2_LAN_IP" "$NODE2_HOST"
|
||||||
|
logn "Both nodes are SSHable."
|
||||||
|
sleep 10 # let systemd finish activation
|
||||||
|
fi
|
||||||
|
|
||||||
|
# ── Phase 7: Refresh sops host-key registrations ─────────────────────────────
|
||||||
|
#
|
||||||
|
# Disko builds raw disk images: each new VM boots with a freshly-generated SSH
|
||||||
|
# host key, not the one pre-seeded in clan vars. Scan the running VMs; if
|
||||||
|
# their ed25519 keys differ from the clan var, update the clan var, rewrite
|
||||||
|
# the .sops.yaml anchor, and re-encrypt all affected sops files.
|
||||||
|
|
||||||
|
if ! $SKIP_REFRESH_SOPS_KEYS; then
|
||||||
|
if $DRY_RUN; then
|
||||||
|
logn "[dry-run] Would scan VM host keys and refresh .sops.yaml / secrets if needed"
|
||||||
|
else
|
||||||
|
log "Phase 7: Refreshing sops host-key registrations (disko key drift fix)"
|
||||||
|
SOPS_UPDATED=false
|
||||||
|
|
||||||
|
for spec in \
|
||||||
|
"${NODE1_LAN_IP}:proxmox-ha-docker-1:${NODE1_HOST}" \
|
||||||
|
"${NODE2_LAN_IP}:proxmox-ha-docker-2:${NODE2_HOST}"; do
|
||||||
|
IFS=: read -r node_ip flake_target host_name <<< "$spec"
|
||||||
|
CLAN_PUB="${REPO_ROOT}/vars/per-machine/${flake_target}/openssh/ssh_host_ed25519_key.pub/value"
|
||||||
|
|
||||||
|
logn "Scanning ed25519 host key from ${host_name} (${node_ip})..."
|
||||||
|
RAW=$(ssh-keyscan -t ed25519 "${node_ip}" 2>/dev/null | grep -v "^#") || true
|
||||||
|
if [[ -z "$RAW" ]]; then
|
||||||
|
logn "WARNING: no ed25519 key returned for ${node_ip} — skipping"
|
||||||
|
continue
|
||||||
|
fi
|
||||||
|
SCANNED_TYPE=$(awk '{print $2}' <<< "$RAW")
|
||||||
|
SCANNED_KEY=$(awk '{print $3}' <<< "$RAW")
|
||||||
|
SCANNED_PUBKEY="${SCANNED_TYPE} ${SCANNED_KEY} ${host_name}"
|
||||||
|
|
||||||
|
CURRENT=$(tr -d '\n' < "$CLAN_PUB" 2>/dev/null || true)
|
||||||
|
if [[ "$SCANNED_PUBKEY" == "$CURRENT" ]]; then
|
||||||
|
logn "${host_name}: clan var matches running key — no update needed"
|
||||||
|
continue
|
||||||
|
fi
|
||||||
|
|
||||||
|
logn "${host_name}: key drift detected — updating clan var"
|
||||||
|
logn " old: ${CURRENT}"
|
||||||
|
logn " new: ${SCANNED_PUBKEY}"
|
||||||
|
echo "$SCANNED_PUBKEY" > "$CLAN_PUB"
|
||||||
|
SOPS_UPDATED=true
|
||||||
|
|
||||||
|
ANCHOR="${flake_target}"
|
||||||
|
NEW_AGE=$(echo "$SCANNED_PUBKEY" | \
|
||||||
|
nix run --quiet --no-warn-dirty nixpkgs#ssh-to-age 2>/dev/null)
|
||||||
|
[[ -z "$NEW_AGE" ]] && err "ssh-to-age produced no output for ${host_name}"
|
||||||
|
logn " new age key: ${NEW_AGE}"
|
||||||
|
sed -i "/&${ANCHOR} /s| age[a-z0-9]*$| ${NEW_AGE}|" "${REPO_ROOT}/.sops.yaml"
|
||||||
|
done
|
||||||
|
|
||||||
|
if $SOPS_UPDATED; then
|
||||||
|
logn "Running sops updatekeys on affected secrets..."
|
||||||
|
SOPS="nix run --quiet --no-warn-dirty nixpkgs#sops --"
|
||||||
|
(cd "${REPO_ROOT}" && \
|
||||||
|
$SOPS updatekeys -y secrets/common.yaml && \
|
||||||
|
$SOPS updatekeys -y secrets/ha-docker-1.keytab && \
|
||||||
|
$SOPS updatekeys -y secrets/ha-docker-2.keytab)
|
||||||
|
|
||||||
|
logn "Committing refreshed host keys and re-encrypted secrets..."
|
||||||
|
(cd "${REPO_ROOT}" && \
|
||||||
|
git add \
|
||||||
|
vars/per-machine/proxmox-ha-docker-1/openssh/ssh_host_ed25519_key.pub/value \
|
||||||
|
vars/per-machine/proxmox-ha-docker-2/openssh/ssh_host_ed25519_key.pub/value \
|
||||||
|
.sops.yaml \
|
||||||
|
secrets/common.yaml \
|
||||||
|
secrets/ha-docker-1.keytab \
|
||||||
|
secrets/ha-docker-2.keytab && \
|
||||||
|
git commit -m "secrets(ha-docker): refresh sops host-key registrations for new VM instances" || true)
|
||||||
|
logn "Sops keys refreshed and committed."
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
|
||||||
|
# ── Phase 8: Initialise Docker Swarm ─────────────────────────────────────────
|
||||||
|
|
||||||
|
if ! $SKIP_INIT_SWARM; then
|
||||||
|
log "Phase 8: Initialising Docker Swarm"
|
||||||
|
|
||||||
|
if $DRY_RUN; then
|
||||||
|
logn "[dry-run] Would run: docker swarm init --advertise-addr ${NODE1_SWARM_IP} --data-path-addr ${NODE1_SWARM_IP} on ${NODE1_HOST}"
|
||||||
|
logn "[dry-run] Would join ${NODE2_HOST} as manager"
|
||||||
|
logn "[dry-run] Would label both nodes"
|
||||||
|
else
|
||||||
|
# Check if node1 is already a swarm manager.
|
||||||
|
if n1 "docker info --format '{{.Swarm.LocalNodeState}}'" 2>/dev/null | grep -qx "active"; then
|
||||||
|
logn "${NODE1_HOST} is already in a swarm — skipping init."
|
||||||
|
else
|
||||||
|
logn "Initialising swarm on ${NODE1_HOST} (advertise: ${NODE1_SWARM_IP})..."
|
||||||
|
n1 "docker swarm init \
|
||||||
|
--advertise-addr ${NODE1_SWARM_IP} \
|
||||||
|
--data-path-addr ${NODE1_SWARM_IP}"
|
||||||
|
logn "Swarm initialised on ${NODE1_HOST}."
|
||||||
|
fi
|
||||||
|
|
||||||
|
# Check if node2 is already joined.
|
||||||
|
if n2 "docker info --format '{{.Swarm.LocalNodeState}}'" 2>/dev/null | grep -qx "active"; then
|
||||||
|
logn "${NODE2_HOST} is already in the swarm — skipping join."
|
||||||
|
else
|
||||||
|
logn "Fetching manager join token from ${NODE1_HOST}..."
|
||||||
|
JOIN_TOKEN=$(n1 "docker swarm join-token manager -q")
|
||||||
|
[[ -z "$JOIN_TOKEN" ]] && err "Failed to get swarm manager join token from ${NODE1_HOST}"
|
||||||
|
|
||||||
|
logn "Joining ${NODE2_HOST} as manager (advertise: ${NODE2_SWARM_IP})..."
|
||||||
|
n2 "docker swarm join \
|
||||||
|
--token ${JOIN_TOKEN} \
|
||||||
|
--advertise-addr ${NODE2_SWARM_IP} \
|
||||||
|
--data-path-addr ${NODE2_SWARM_IP} \
|
||||||
|
${NODE1_SWARM_IP}:2377"
|
||||||
|
logn "${NODE2_HOST} joined as manager."
|
||||||
|
fi
|
||||||
|
|
||||||
|
# Label nodes for service placement constraints.
|
||||||
|
logn "Labelling swarm nodes..."
|
||||||
|
n1 "docker node update --label-add node=${NODE1_HOST} ${NODE1_HOST}" || true
|
||||||
|
n1 "docker node update --label-add node=${NODE2_HOST} ${NODE2_HOST}" || true
|
||||||
|
logn "Labels applied."
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
|
||||||
|
# ── Phase 9: DNS registration ─────────────────────────────────────────────────
|
||||||
|
|
||||||
|
if ! $SKIP_DNS; then
|
||||||
|
log "Phase 9: Registering DNS records in FreeIPA"
|
||||||
|
|
||||||
|
if $DRY_RUN; then
|
||||||
|
logn "[dry-run] Would create/verify ${SWARM_ZONE} zone and add A records"
|
||||||
|
else
|
||||||
|
# Check for and create the swarm.home zone if absent.
|
||||||
|
if ! dc "ipa dnszone-show ${SWARM_ZONE}" >/dev/null 2>&1; then
|
||||||
|
logn "Creating ${SWARM_ZONE} DNS zone..."
|
||||||
|
dc "ipa dnszone-add ${SWARM_ZONE} \
|
||||||
|
--name-server=${DC_HOST}. \
|
||||||
|
--admin-email=hostmaster@${SWARM_ZONE}"
|
||||||
|
# Reverse zone for 192.168.30.x
|
||||||
|
dc "ipa dnszone-add 30.168.192.in-addr.arpa \
|
||||||
|
--name-server=${DC_HOST}. \
|
||||||
|
--admin-email=hostmaster@${SWARM_ZONE}" 2>/dev/null || \
|
||||||
|
logn " (reverse zone 30.168.192.in-addr.arpa already exists or skipped)"
|
||||||
|
else
|
||||||
|
logn "${SWARM_ZONE} zone already exists."
|
||||||
|
fi
|
||||||
|
|
||||||
|
# storage.home A records (zone already exists from HA cluster setup).
|
||||||
|
for spec in "${NODE1_HOST}:${NODE1_STORAGE_IP}" "${NODE2_HOST}:${NODE2_STORAGE_IP}"; do
|
||||||
|
IFS=: read -r hostname ip <<< "$spec"
|
||||||
|
logn "Adding ${hostname}.${STORAGE_ZONE} → ${ip}"
|
||||||
|
dc "ipa dnsrecord-add ${STORAGE_ZONE} ${hostname} --a-rec=${ip} --a-create-reverse" 2>/dev/null || \
|
||||||
|
logn " (record already exists or reverse zone missing — continuing)"
|
||||||
|
done
|
||||||
|
|
||||||
|
# swarm.home A records.
|
||||||
|
for spec in "${NODE1_HOST}:${NODE1_SWARM_IP}" "${NODE2_HOST}:${NODE2_SWARM_IP}"; do
|
||||||
|
IFS=: read -r hostname ip <<< "$spec"
|
||||||
|
logn "Adding ${hostname}.${SWARM_ZONE} → ${ip}"
|
||||||
|
dc "ipa dnsrecord-add ${SWARM_ZONE} ${hostname} --a-rec=${ip} --a-create-reverse" 2>/dev/null || \
|
||||||
|
logn " (record already exists — continuing)"
|
||||||
|
done
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
|
||||||
|
# ── Phase 10: Verify ──────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
if ! $SKIP_VERIFY; then
|
||||||
|
log "Phase 10: Verifying swarm health"
|
||||||
|
|
||||||
|
if $DRY_RUN; then
|
||||||
|
logn "[dry-run] Would verify swarm node list and NFS mounts"
|
||||||
|
else
|
||||||
|
logn "Swarm node list:"
|
||||||
|
n1 "docker node ls" || err "docker node ls failed on ${NODE1_HOST}"
|
||||||
|
|
||||||
|
logn "Checking swarm state on both nodes..."
|
||||||
|
for spec in "${NODE1_LAN_IP}:${NODE1_HOST}" "${NODE2_LAN_IP}:${NODE2_HOST}"; do
|
||||||
|
IFS=: read -r ip hostname <<< "$spec"
|
||||||
|
STATE=$(ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no \
|
||||||
|
"${SWARM_USER}@${ip}" "docker info --format '{{.Swarm.LocalNodeState}}'" 2>/dev/null)
|
||||||
|
if [[ "$STATE" != "active" ]]; then
|
||||||
|
err "${hostname} swarm state is '${STATE}', expected 'active'"
|
||||||
|
fi
|
||||||
|
logn " ${hostname}: swarm=${STATE} ✓"
|
||||||
|
done
|
||||||
|
|
||||||
|
logn "Checking NFS mounts on both nodes..."
|
||||||
|
for spec in "${NODE1_LAN_IP}:${NODE1_HOST}" "${NODE2_LAN_IP}:${NODE2_HOST}"; do
|
||||||
|
IFS=: read -r ip hostname <<< "$spec"
|
||||||
|
NFS_OK=$(ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no \
|
||||||
|
"${SWARM_USER}@${ip}" "df -h /mnt/docker/config 2>/dev/null | grep -c nfs || echo 0" 2>/dev/null)
|
||||||
|
if [[ "$NFS_OK" -ge 1 ]]; then
|
||||||
|
logn " ${hostname}: /mnt/docker/config NFS mount ✓"
|
||||||
|
else
|
||||||
|
logn " WARNING: ${hostname}: /mnt/docker/config does not appear to be NFS-mounted"
|
||||||
|
logn " (automount may still be pending — try: ssh nixos@${ip} 'ls /mnt/docker/config')"
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
|
||||||
|
logn "Checking overlay network..."
|
||||||
|
NETWORKS=$(n1 "docker network ls --filter driver=overlay --format '{{.Name}}'")
|
||||||
|
if echo "$NETWORKS" | grep -q "ingress"; then
|
||||||
|
logn " ingress overlay network present ✓"
|
||||||
|
else
|
||||||
|
logn " WARNING: ingress overlay network not found — swarm may not be fully initialised"
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
|
||||||
|
log "Deploy complete. Both nodes are ready for 'docker stack deploy'."
|
||||||
|
log "Connect to either manager:"
|
||||||
|
log " ssh nixos@${NODE1_LAN_IP} (${NODE1_HOST})"
|
||||||
|
log " ssh nixos@${NODE2_LAN_IP} (${NODE2_HOST})"
|
||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user