Archived
Compare commits
382
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f43de81a0b | ||
|
|
11f8ca4d58 | ||
|
|
86a1660adc | ||
|
|
6522a35115 | ||
|
|
fe8693b0e5 | ||
|
|
00bb9b53ac | ||
|
|
e0fb6ebcb2 | ||
|
|
b2dda869a4 | ||
|
|
f4a2d2d830 | ||
|
|
460459cda1 | ||
|
|
49a7068bc0 | ||
|
|
4963ce9ff9 | ||
|
|
0135db8875 | ||
|
|
419f17be2d | ||
|
|
2904522138 | ||
|
|
25c7fa9119 | ||
|
|
532dc03cfa | ||
|
|
b68bb3a997 | ||
|
|
9748a3f772 | ||
|
|
7e7e294371 | ||
|
|
a3e85b5079 | ||
|
|
2858891c20 | ||
|
|
0e5aa044c0 | ||
|
|
87eb8f7c39 | ||
|
|
89308730ab | ||
|
|
48255ba1da | ||
|
|
a9a204eaf2 | ||
|
|
ff021ca6f7 | ||
|
|
cc8566cc2e | ||
|
|
f5935a5826 | ||
|
|
9ef2efad88 | ||
|
|
f55e33b807 | ||
|
|
f0fae441ca | ||
|
|
9bfd804f7a | ||
|
|
0f7ab6fe7f | ||
|
|
75aef6ba4c | ||
|
|
bd65e71413 | ||
|
|
3a821158ca | ||
|
|
94bc27bcbb | ||
|
|
326e764f51 | ||
|
|
bbc6124dc0 | ||
|
|
b64ef613c3 | ||
|
|
29d7059f70 | ||
|
|
e667215ea6 | ||
|
|
9c0b74d731 | ||
|
|
6dcbae5659 | ||
|
|
2f7831aea8 | ||
|
|
c05e3a3821 | ||
|
|
0b090cab07 | ||
|
|
14accceb02 | ||
|
|
90baa168d1 | ||
|
|
d3ae41a031 | ||
|
|
a3230128b2 | ||
|
|
1395fdaccc | ||
|
|
8259e57191 | ||
|
|
9d42450c65 | ||
|
|
add9a29908 | ||
|
|
13b0c028a1 | ||
|
|
fdc4ea4596 | ||
|
|
c2c86f0854 | ||
|
|
40b68c2e16 | ||
|
|
2ea39994f8 | ||
|
|
331246a1cb | ||
|
|
442e7986f2 | ||
|
|
6064c69d3b | ||
|
|
56a2f3ff51 | ||
|
|
a29efce29e | ||
|
|
8e8183e371 | ||
|
|
4c53cce35c | ||
|
|
dea6bfec25 | ||
|
|
04de84d03b | ||
|
|
7bfe8cb0b2 | ||
|
|
8460eff2de | ||
|
|
bcfeb08b11 | ||
|
|
783bda00c1 | ||
|
|
e3ac63aefa | ||
|
|
4f1aff9a16 | ||
|
|
befab7f3c6 | ||
|
|
b9610dbc9d | ||
|
|
b605d19870 | ||
|
|
679dfd80c0 | ||
|
|
298f615929 | ||
|
|
5e17baba72 | ||
|
|
345b5ca657 | ||
|
|
2c4fe8af2b | ||
|
|
c128bff2bf | ||
|
|
8aff9fff29 | ||
|
|
8f468d5703 | ||
|
|
690ecb1eb0 | ||
|
|
1ed1b388f5 | ||
|
|
f338843166 | ||
|
|
1c1f65560c | ||
|
|
258ebfe1bf | ||
|
|
45cb9c38e8 | ||
|
|
b481324c7d | ||
|
|
d5467e9a5b | ||
|
|
c3c467bb41 | ||
|
|
80eff2c050 | ||
|
|
1cff38be43 | ||
|
|
960548c2ac | ||
|
|
229cd06f73 | ||
|
|
6e332a6bbd | ||
|
|
fe3041c06c | ||
|
|
591731e3b8 | ||
|
|
5da5401e91 | ||
|
|
aca46255f8 | ||
|
|
f85559bb18 | ||
|
|
07534e64fc | ||
|
|
54cefce90a | ||
|
|
fcb8f5e01e | ||
|
|
985d4e4bfa | ||
|
|
9462d80ccc | ||
|
|
14deedc12d | ||
|
|
893a7c997c | ||
|
|
d068195fbe | ||
|
|
51fba54d94 | ||
|
|
f541688973 | ||
|
|
3bb2b83370 | ||
|
|
8ff22175e5 | ||
|
|
bc719afe39 | ||
|
|
20dc28069c | ||
|
|
28ae709100 | ||
|
|
3de625309e | ||
|
|
f2744bf3a7 | ||
|
|
5b118b05da | ||
|
|
f139333980 | ||
|
|
5d5428014a | ||
|
|
529a518cd1 | ||
|
|
dc714b41c1 | ||
|
|
c470737316 | ||
|
|
8266f43980 | ||
|
|
474d243649 | ||
|
|
019afb5609 | ||
|
|
63062cdda5 | ||
|
|
8f5599a1c6 | ||
|
|
416b7196c2 | ||
|
|
0bdadcc772 | ||
|
|
aa08bf60b9 | ||
|
|
99ad7501bb | ||
|
|
6c973e2300 | ||
|
|
eb2d3e3260 | ||
|
|
f9c0f01aba | ||
|
|
83481b5291 | ||
|
|
a553958bb9 | ||
|
|
d16b22d2b5 | ||
|
|
3900d617fa | ||
|
|
c15e88de35 | ||
|
|
ff9fbc36e9 | ||
|
|
fe322c13c6 | ||
|
|
4d8b00609b | ||
|
|
6b7c011480 | ||
|
|
a488ff507e | ||
|
|
9685233553 | ||
|
|
837096ca28 | ||
|
|
bd550b9c2e | ||
|
|
5e8037d02b | ||
|
|
1af8079a8e | ||
|
|
030d344ea2 | ||
|
|
6a266449d5 | ||
|
|
895c9d7003 | ||
|
|
0edb1a431e | ||
|
|
8f32ccdd0f | ||
|
|
44e828320f | ||
|
|
b3c3760c3e | ||
|
|
b9ff8002bd | ||
|
|
ae4c3dba24 | ||
|
|
f008c16ea7 | ||
|
|
4d4a10aaae | ||
|
|
3e298b8c6e | ||
|
|
f7baaf88bb | ||
|
|
0606b93570 | ||
|
|
dd80145e4f | ||
|
|
c25c5d4fbc | ||
|
|
29b1b86286 | ||
|
|
33be42e3ca | ||
|
|
28c875e8cb | ||
|
|
09a50f8ff2 | ||
|
|
161888b037 | ||
|
|
3578e66a15 | ||
|
|
f759f98f97 | ||
|
|
36a5287b7b | ||
|
|
04692c33f2 | ||
|
|
da3f4b16c9 | ||
|
|
f00364bd14 | ||
|
|
c25eb536c7 | ||
|
|
34ff056ca9 | ||
|
|
4187e7dc79 | ||
|
|
de9549b466 | ||
|
|
140dea3993 | ||
|
|
545775fc26 | ||
|
|
514fa3068a | ||
|
|
ded8b7dd55 | ||
|
|
8444fc370e | ||
|
|
2c5fca84bd | ||
|
|
7e1768c2e8 | ||
|
|
39332d0b93 | ||
|
|
fc193d0932 | ||
|
|
0154648aa2 | ||
|
|
74b3c8cd3d | ||
|
|
e24f9314be | ||
|
|
8211ff8cb1 | ||
|
|
88dedeed43 | ||
|
|
997e384973 | ||
|
|
31371189f5 | ||
|
|
70ea6c2e7c | ||
|
|
03adfdd0de | ||
|
|
84569d13e6 | ||
|
|
519d7b8d3c | ||
|
|
5f036c8f54 | ||
|
|
83abba4e25 | ||
|
|
c718010a69 | ||
|
|
2f985536c4 | ||
|
|
bf0445ebd6 | ||
|
|
7b945ea4fe | ||
|
|
5fbc8ba5df | ||
|
|
d474f1b1b1 | ||
|
|
b36431f210 | ||
|
|
d4771720b9 | ||
|
|
2e2298d66e | ||
|
|
51a50faf69 | ||
|
|
193c10d4b7 | ||
|
|
def6be08e7 | ||
|
|
0633c60b9d | ||
|
|
7cbdcdf29e | ||
|
|
cf1a4eefda | ||
|
|
5256b661a3 | ||
|
|
bcccf523bf | ||
|
|
1a4a0c47ef | ||
|
|
286d7347a1 | ||
|
|
5a93bdeb28 | ||
|
|
770cbaf098 | ||
|
|
dec3e25472 | ||
|
|
8b919d2d5a | ||
|
|
d52e892559 | ||
|
|
df4161265b | ||
|
|
80bffb58c9 | ||
|
|
f8acce2614 | ||
|
|
daadd97f84 | ||
|
|
a3be101ce3 | ||
|
|
4bf32f23ab | ||
|
|
f14aa9755b | ||
|
|
8f5f975d83 | ||
|
|
7f7b03384f | ||
|
|
6e138c6f7f | ||
|
|
e6a72813e4 | ||
|
|
350a2d2bfb | ||
|
|
cb1ddc2109 | ||
|
|
376d53de40 | ||
|
|
fd592ba5de | ||
|
|
4b37d36212 | ||
|
|
0a10efefa6 | ||
|
|
5e559efc6f | ||
|
|
149f56ce10 | ||
|
|
c6f6441907 | ||
|
|
cc956d3038 | ||
|
|
d409b5b718 | ||
|
|
6ef87a3226 | ||
|
|
9a190a28d6 | ||
|
|
e7a215cd15 | ||
|
|
651f4e61c5 | ||
|
|
ffad8bd6b8 | ||
|
|
01aaf10e2d | ||
|
|
8db2a2db86 | ||
|
|
3389c9549a | ||
|
|
7ab8cf15ca | ||
|
|
56a25ab5d7 | ||
|
|
76649ad698 | ||
|
|
487d8bc474 | ||
|
|
529535cffd | ||
|
|
aff4cf1c16 | ||
|
|
3db1205a8d | ||
|
|
bc5be541c3 | ||
|
|
38c55c3de2 | ||
|
|
25b8ebcad7 | ||
|
|
d08951d46a | ||
|
|
9eba317de0 | ||
|
|
a9bb49a7e7 | ||
|
|
46c71eb619 | ||
|
|
8918af8d6e | ||
|
|
48447f7a47 | ||
|
|
943c5324ca | ||
|
|
a8f417b4f9 | ||
|
|
ff75079327 | ||
|
|
6f79daefc4 | ||
|
|
66b6bc0c2a | ||
|
|
b952075a1e | ||
|
|
49c72848e6 | ||
|
|
c19919a868 | ||
|
|
4c4ed6416d | ||
|
|
a457d04a0a | ||
|
|
019ab8c2cc | ||
|
|
cb0efa12ce | ||
|
|
bb454d5a97 | ||
|
|
af4021b29f | ||
|
|
a7a6951115 | ||
|
|
00bab4b69a | ||
|
|
6c304304f0 | ||
|
|
60942af93f | ||
|
|
f4dd65c66e | ||
|
|
3fbc290fd5 | ||
|
|
1b6779f89c | ||
|
|
dc933fc69b | ||
|
|
4d322a94ac | ||
|
|
17ecbc617d | ||
|
|
c892c70860 | ||
|
|
376f0b98dd | ||
|
|
728129a4b2 | ||
|
|
fe8649b09a | ||
|
|
64183b0ab9 | ||
|
|
8c7a7815a7 | ||
|
|
0dc6c5099f | ||
|
|
236385e480 | ||
|
|
ddf0133da5 | ||
|
|
f4a849bbac | ||
|
|
0c2344098b | ||
|
|
ee44a2dafd | ||
|
|
ff014f6658 | ||
|
|
c8dc59750b | ||
|
|
35f2c457dd | ||
|
|
5a60d574d8 | ||
|
|
08f6dd7347 | ||
|
|
0228bdf429 | ||
|
|
1d99b54374 | ||
|
|
dd32d23249 | ||
|
|
47927a91f1 | ||
|
|
954d89f101 | ||
|
|
5bb1b0c6a1 | ||
|
|
706e74c46d | ||
|
|
094b6c34f0 | ||
|
|
aeb2cde784 | ||
|
|
f1520fbd75 | ||
|
|
0321f764aa | ||
|
|
e0a1d1b5dc | ||
|
|
cfea52cf7d | ||
|
|
94736a53ef | ||
|
|
1960c4cd8c | ||
|
|
b66f5dd800 | ||
|
|
3d2938d896 | ||
|
|
903ac787b1 | ||
|
|
34cf3a330a | ||
|
|
c8c5d21e5a | ||
|
|
ec82189d47 | ||
|
|
ebf8173845 | ||
|
|
360d0d4f46 | ||
|
|
fb73fe70cd | ||
|
|
a0699349c7 | ||
|
|
2e6f19e597 | ||
|
|
d5f6a154ca | ||
|
|
afda3b0a5d | ||
|
|
4c4b3d47f3 | ||
|
|
7413340c75 | ||
|
|
b0dd745a03 | ||
|
|
be3012a65a | ||
|
|
55dd99e155 | ||
|
|
845ab268ce | ||
|
|
1df53af9bf | ||
|
|
427155bcb0 | ||
|
|
e8ee5981ac | ||
|
|
6d701af3c7 | ||
|
|
d6f9fe97f5 | ||
|
|
d450750818 | ||
|
|
1c5dac74ca | ||
|
|
d4f5e0b22d | ||
|
|
b0964fcb80 | ||
|
|
a285756040 | ||
|
|
20d38c571c | ||
|
|
e22dcd6d02 | ||
|
|
a2bc222fc8 | ||
|
|
198ba7a198 | ||
|
|
cc6704830e | ||
|
|
072f1690e6 | ||
|
|
5ca3254788 | ||
|
|
398dc5161a | ||
|
|
798c3c10d5 | ||
|
|
af2f4263cc | ||
|
|
fa76fb0f13 | ||
|
|
d6c1e48667 | ||
|
|
aa065f21e3 | ||
|
|
61f5059492 | ||
|
|
bd6f0ab825 | ||
|
|
0ab5a56493 | ||
|
|
c9ab34040d |
Submodule .claude/worktrees/scripts-dedup deleted from e578443914
@@ -13,17 +13,9 @@ jobs:
|
||||
steps:
|
||||
- name: Check out repository
|
||||
uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Install Nix
|
||||
uses: DeterminateSystems/nix-installer-action@v19
|
||||
|
||||
# Scoped to files changed since the PR base / previous push -- see
|
||||
# scripts/codex-maintenance.sh. CI never passes --full-check: that
|
||||
# full sweep is for local/manual use, since it's slow enough to time
|
||||
# out this runner.
|
||||
- name: Run maintenance checks (secrets, fmt, lint, eval -- changed files only)
|
||||
env:
|
||||
MAINT_BASE_SHA: ${{ github.event.pull_request.base.sha || github.event.before }}
|
||||
- name: Run maintenance checks (secrets, fmt, lint, eval)
|
||||
run: bash scripts/codex-maintenance.sh
|
||||
|
||||
@@ -13,17 +13,9 @@ jobs:
|
||||
steps:
|
||||
- name: Check out repository
|
||||
uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Install Nix
|
||||
uses: DeterminateSystems/nix-installer-action@v19
|
||||
|
||||
# Scoped to files changed since the PR base / previous push -- see
|
||||
# scripts/codex-maintenance.sh. CI never passes --full-check: that
|
||||
# full sweep is for local/manual use, since it's slow enough to time
|
||||
# out this runner.
|
||||
- name: Run maintenance checks (secrets, fmt, lint, eval -- changed files only)
|
||||
env:
|
||||
MAINT_BASE_SHA: ${{ github.event.pull_request.base.sha || github.event.before }}
|
||||
- name: Run maintenance checks (secrets, fmt, lint, eval)
|
||||
run: bash scripts/codex-maintenance.sh
|
||||
|
||||
+17
-195
@@ -1,29 +1,12 @@
|
||||
keys:
|
||||
- &admin age1njap586hc0q43kr03g6c8eqhdsmk8zcafkl3f83xwlc2gqhlmfgs4tmwad
|
||||
- &proxmox-minimal age19m0m7vdfg86yqy8l5mmle5jdd0unrn3f55t232w8h5ey42cqw34sfpt32n
|
||||
- &lxc-gui age1rrxqea6q6pn39sw8y5te63h2py8jgjl9v0jyper86w3ggtn67upqg3ah39
|
||||
- &baremetal-gui age1adur9g330gua4l6ndk8cqjg35qc8yxwgme6wrl2hpylcc7vxm38q05ejuy
|
||||
- &linode-docker age17e89ty6p0fw24daanen57wg8uald9s025t3wwxsw269svwpmgvrshfvfvt
|
||||
- &linode-gui age1hrx8qj02fj2ea6d4g9vqhyj9hl7fppkjqfdx2l37py3h6pdkr95s8n8rvs
|
||||
- &linode-minimal age1e7l8dusgmgfzd2cxrrzwepzjxt69hzqj4epee0cs27u6yg4kxcuqm34ncx
|
||||
- &linode-nix-cache age1jcx3yajjhghn8qh8za3yeu8nxykzlg3p4nrv03vnfvzl0mzayg2qmg940e
|
||||
- &linode-server age1sweerhrga9yf8x6sv0apz4ed4g48rnlcq34rpv20t0rcelwgpgeqwvndzz
|
||||
- &linode-tailscale-router age1f7usptjx9rv4rxauasve200gxtdt9jkqhhdqstlf20wvlm7u75rsjfw50m
|
||||
- &lxc-docker age17jqc66x9yeshfgd9v78mj483r4zzarqdtuxtrkxe4x5mw679gphshd94th
|
||||
- &lxc-minimal age1px0h5l9zp2dww0m8fncrc82kfdmzplsfv2ltat7sna28xpg09pqqcl3s2k
|
||||
- &lxc-nix-cache age1ufg390ydrmma849t9xfkxxl5xvdkk6mngnlzhmy7mvuaje8sgcmsmnq6l7
|
||||
- &lxc-pxe-boot age16j42pdc5dr6wnj7xayhkqdj2rny9u68fcqejs50hqq42scssh4gsnrrnlt
|
||||
- &lxc-server age1nruncs4l0ufk7yuc4des8p99c0alfndl0lhsws8tycl5pplfp56s30af5f
|
||||
- &lxc-tailscale-router age1k7d2du5mejsmv5rzavm4xwgpthqvcfsehduquv28nzs53zppa3kqngfxq2
|
||||
- &lxc-tor-relay age16kqfmvz4e23hmdlqresnyw69ej604s320mmd49h4hm3fhqchtgyqrws0k2
|
||||
- &proxmox-docker age1arhf2q45zw6wf2uevju4savp575x3m2tfvved5zzq3ay92ynua9s3cm92c
|
||||
- &proxmox-gui age19mn8zrxl8zpps9yvrh4euquvygpp4fp8queg7xc6qhtnl4ng8c9qx02qwn
|
||||
- &proxmox-nix-cache age1jlltcv5jcnm40z5k0q6hv053k2rqpqvemtuecdwn527uw8uqz4es3x7m68
|
||||
- &proxmox-pxe-boot age1ug787sgt6st6k82fgkrug2lzltw4qsukrrqqs3w27ewwqj8rg4hsxcmylz
|
||||
- &proxmox-server age1529taqdwr6t0w7cvzmty0d5y5593wffl0krt48j6uc4u39k56g2qf6ywtp
|
||||
- &proxmox-tailscale-router age1zhfyuzlq40reuqlr34gf77852nhs3t6mqfzrqmas8z6sxk7tcfhsungrm0
|
||||
- &proxmox-ha-server-1 age1nxlnrevqs2msdatze562vz6ym4pgydt2zndye83lwqauy6flggtqrevyw5
|
||||
- &proxmox-ha-server-2 age1scfc8p53q5aq2a87tcmsazmj8sfeft0s8kxg0et4nm3ucxyp3c3s6te82y
|
||||
- &admin age10nd382a9klsn2mrs60emdtsxe43pht3a0m9p29phfrhy0wfyt3vsq9r667
|
||||
- &docker age19gfn2yedg76dmztm4hncr7vf3r3c9j0qpt4rap7y7gersjk4m3ks2lhd0e
|
||||
- &server age1ll6hj5ggruetgjwjfnplpn5xtq35uhlcdflksx3xmnjm6s3uad9sz70jkf
|
||||
- &nix-cache age120le4a5l8dh3lyfgvmj3d9ksmej6ajs5mer5y7r0vfg3x9fn69dqf8xgzu
|
||||
- &lxc-minimal age1qz9d4ka4xgexujyd247s7lp737sulp5fhxl5d65fj2ykvc4j4edqrsdks8
|
||||
- &nix-minimal age120whqj96g26lsgy4udvgsn8dc9lumh8jeu3a564fx79rjr5lxffqmrljuu
|
||||
- &lxc-nix-cache age164px2a8e48ptsf9ngtan38aa6jls4jdl26mzrgzf6sn3vcvt49hqjrgr8w
|
||||
- &proxmox-minimal age10at8862478urh0eeuwh8hzln6ck78jgwtztgxatwqlzwagg77y5snm4xzg
|
||||
|
||||
creation_rules:
|
||||
# Shared across every currently-deployed host: root/nixos password hash,
|
||||
@@ -34,190 +17,29 @@ creation_rules:
|
||||
key_groups:
|
||||
- age:
|
||||
- *admin
|
||||
- *proxmox-minimal
|
||||
- *lxc-gui
|
||||
- *baremetal-gui
|
||||
- *linode-docker
|
||||
- *linode-gui
|
||||
- *linode-minimal
|
||||
- *linode-nix-cache
|
||||
- *linode-server
|
||||
- *linode-tailscale-router
|
||||
- *lxc-docker
|
||||
- *docker
|
||||
- *server
|
||||
- *nix-cache
|
||||
- *lxc-minimal
|
||||
- *nix-minimal
|
||||
- *lxc-nix-cache
|
||||
- *lxc-pxe-boot
|
||||
- *lxc-server
|
||||
- *lxc-tailscale-router
|
||||
- *lxc-tor-relay
|
||||
- *proxmox-docker
|
||||
- *proxmox-gui
|
||||
- *proxmox-nix-cache
|
||||
- *proxmox-pxe-boot
|
||||
- *proxmox-server
|
||||
- *proxmox-tailscale-router
|
||||
- *proxmox-ha-server-1
|
||||
- *proxmox-ha-server-2
|
||||
- *proxmox-minimal
|
||||
|
||||
- path_regex: secrets/nix-cache\.yaml$
|
||||
key_groups:
|
||||
- age:
|
||||
- *admin
|
||||
- *linode-nix-cache
|
||||
- *nix-cache
|
||||
- *lxc-nix-cache
|
||||
- *proxmox-nix-cache
|
||||
|
||||
- path_regex: secrets/server\.yaml$
|
||||
key_groups:
|
||||
- age:
|
||||
- *admin
|
||||
- *linode-server
|
||||
- *lxc-server
|
||||
- *proxmox-server
|
||||
- *server
|
||||
|
||||
- path_regex: secrets/tor-relay\.yaml$
|
||||
- path_regex: secrets/docker\.yaml$
|
||||
key_groups:
|
||||
- age:
|
||||
- *admin
|
||||
- *lxc-tor-relay
|
||||
|
||||
- path_regex: secrets/tailscale-router\.yaml$
|
||||
key_groups:
|
||||
- age:
|
||||
- *admin
|
||||
- *linode-tailscale-router
|
||||
- *lxc-tailscale-router
|
||||
- *proxmox-tailscale-router
|
||||
|
||||
# HA file server per-node secrets (beszel-token).
|
||||
# proxmox-ha-server-1 / proxmox-ha-server-2 keys are added automatically
|
||||
# by scripts/secrets/sync-host-keys.sh once the hosts are provisioned;
|
||||
# until then only the admin key can decrypt these files.
|
||||
- path_regex: secrets/ha-server-1\.yaml$
|
||||
key_groups:
|
||||
- age:
|
||||
- *admin
|
||||
- *proxmox-ha-server-1
|
||||
# proxmox-ha-server-1 added by sync-host-keys.sh
|
||||
|
||||
- path_regex: secrets/ha-server-2\.yaml$
|
||||
key_groups:
|
||||
- age:
|
||||
- *admin
|
||||
- *proxmox-ha-server-2
|
||||
# proxmox-ha-server-2 added by sync-host-keys.sh
|
||||
|
||||
# Shared HA cluster corosync authkey (binary sops file).
|
||||
# Encrypted for both HA nodes so either can decrypt on boot.
|
||||
# Both host keys added by sync-host-keys.sh; admin key allows initial creation.
|
||||
- path_regex: secrets/ha-corosync-authkey$
|
||||
key_groups:
|
||||
- age:
|
||||
- *admin
|
||||
- *proxmox-ha-server-1
|
||||
- *proxmox-ha-server-2
|
||||
# proxmox-ha-server-1 added by sync-host-keys.sh
|
||||
# proxmox-ha-server-2 added by sync-host-keys.sh
|
||||
|
||||
# gui-host-specific secrets (currently: wifi-password, see
|
||||
# modules/networking/wifi.nix). Only *lxc-gui has a registered key today
|
||||
# -- proxmox-gui/linode-gui/baremetal-gui haven't been provisioned via
|
||||
# scripts/secrets/sync-host-keys.sh yet, so whichever variant is actually
|
||||
# deployed next needs its recipient added here (and `sops updatekeys` rerun)
|
||||
# before it can decrypt this.
|
||||
- path_regex: secrets/gui\.yaml$
|
||||
key_groups:
|
||||
- age:
|
||||
- *admin
|
||||
- *lxc-gui
|
||||
- *baremetal-gui
|
||||
- *linode-gui
|
||||
- *proxmox-gui
|
||||
|
||||
# IPA host keytabs (binary sops files).
|
||||
# Each keytab is encrypted for all platform variants of that host so any
|
||||
# deployed variant can decrypt it at boot. Run
|
||||
# scripts/ipa/create-nixos-ipa-host-account.sh <hostname> to enroll a new
|
||||
# host and produce the keytab; this section is updated by that script.
|
||||
|
||||
- path_regex: secrets/nix-cache\.keytab$
|
||||
key_groups:
|
||||
- age:
|
||||
- *admin
|
||||
- *linode-nix-cache
|
||||
- *lxc-nix-cache
|
||||
- *proxmox-nix-cache
|
||||
|
||||
- path_regex: secrets/tailscale-router\.keytab$
|
||||
key_groups:
|
||||
- age:
|
||||
- *admin
|
||||
- *linode-tailscale-router
|
||||
- *lxc-tailscale-router
|
||||
- *proxmox-tailscale-router
|
||||
|
||||
- path_regex: secrets/pxe-boot\.keytab$
|
||||
key_groups:
|
||||
- age:
|
||||
- *admin
|
||||
- *lxc-pxe-boot
|
||||
- *proxmox-pxe-boot
|
||||
|
||||
# nixos = the workstation (hosts/nixos/host.nix). All gui platform variants
|
||||
# share the hostname "nixos" and must be able to decrypt at boot.
|
||||
- path_regex: secrets/nixos\.keytab$
|
||||
key_groups:
|
||||
- age:
|
||||
- *admin
|
||||
- *baremetal-gui
|
||||
- *lxc-gui
|
||||
- *proxmox-gui
|
||||
- *linode-gui
|
||||
|
||||
- path_regex: secrets/server\.keytab$
|
||||
key_groups:
|
||||
- age:
|
||||
- *admin
|
||||
- *linode-server
|
||||
- *lxc-server
|
||||
- *proxmox-server
|
||||
|
||||
- path_regex: secrets/docker\.keytab$
|
||||
key_groups:
|
||||
- age:
|
||||
- *admin
|
||||
- *linode-docker
|
||||
- *lxc-docker
|
||||
- *proxmox-docker
|
||||
|
||||
- path_regex: secrets/tor-relay\.keytab$
|
||||
key_groups:
|
||||
- age:
|
||||
- *admin
|
||||
- *lxc-tor-relay
|
||||
|
||||
- path_regex: secrets/nix-minimal\.keytab$
|
||||
key_groups:
|
||||
- age:
|
||||
- *admin
|
||||
- *lxc-minimal
|
||||
- *proxmox-minimal
|
||||
- *linode-minimal
|
||||
|
||||
# Host keytab for ha-server-1 FreeIPA enrollment (binary sops file).
|
||||
# Generated by scripts/ipa/create-nixos-ipa-host-account.sh.
|
||||
- path_regex: secrets/ha-server-1\.keytab$
|
||||
key_groups:
|
||||
- age:
|
||||
- *admin
|
||||
- *proxmox-ha-server-1
|
||||
# proxmox-ha-server-1 added by sync-host-keys.sh
|
||||
|
||||
# Host keytab for ha-server-2 FreeIPA enrollment (binary sops file).
|
||||
# Generated by scripts/ipa/create-nixos-ipa-host-account.sh.
|
||||
- path_regex: secrets/ha-server-2\.keytab$
|
||||
key_groups:
|
||||
- age:
|
||||
- *admin
|
||||
- *proxmox-ha-server-2
|
||||
# proxmox-ha-server-2 added by sync-host-keys.sh
|
||||
- *docker
|
||||
|
||||
@@ -6,14 +6,12 @@ This repository contains flake-based NixOS configurations for Wayne's LAN
|
||||
servers and workstation.
|
||||
|
||||
The flake exposes NixOS configurations named `<platform>-<buildtype>`
|
||||
(platforms: `linode`, `proxmox`, `lxc`, `baremetal`; build types: `minimal`,
|
||||
`nix-cache`, `server`, `docker`, `gui`, `pxe-boot`, `tailscale-router`,
|
||||
`tor-relay`, `ha-server`), generated from `modules/platforms/*` and
|
||||
`modules/build-types/*` by the `mkTarget` function in `flake.nix`. Not every
|
||||
combination is built — `pxe-boot` has no `linode` variant, `ha-server` only
|
||||
exists on `proxmox`, and `tor-relay` only exists on `lxc`. See `README.md`
|
||||
for the full current target list; treat `flake.nix` as the source of truth
|
||||
since this list can drift.
|
||||
(platforms: `linode`, `proxmox`, `lxc`; build types: `minimal`, `nix-cache`,
|
||||
`server`, `docker`, `gui`, `pxe-boot`), generated from `modules/platforms/*`
|
||||
and `modules/build-types/*` by the `mkTarget` function in `flake.nix`. Not
|
||||
every combination is built — `pxe-boot` has no `linode` variant. See
|
||||
`README.md` for the full current target list; treat `flake.nix` as the
|
||||
source of truth since this list can drift.
|
||||
|
||||
Do not deploy, switch, reboot, repartition, format disks, or run destructive
|
||||
install commands from this repository unless explicitly asked.
|
||||
@@ -37,14 +35,9 @@ Use these commands when validating changes:
|
||||
```bash
|
||||
bash scripts/codex-setup.sh
|
||||
bash scripts/codex-maintenance.sh
|
||||
bash scripts/codex-maintenance.sh dry-run
|
||||
```
|
||||
|
||||
With no flags, `codex-maintenance.sh` scopes fmt-check/statix/eval to files
|
||||
changed against a base ref — this is what CI runs on every push/PR. For the
|
||||
full sweep (every host, every package — slow; CI never runs this), use
|
||||
`bash scripts/codex-maintenance.sh --full-check` (add `--dry-run` for build
|
||||
planning on top of whichever scope is active).
|
||||
|
||||
Host evaluation is safe when limited to drvPath checks:
|
||||
|
||||
```bash
|
||||
|
||||
-150
@@ -1,150 +0,0 @@
|
||||
# Flake End-to-End Audit Report
|
||||
|
||||
**Date:** 2026-07-21
|
||||
**Scope:** Full static lint/eval sweep + live build/deploy/interrogate/destroy testing of every `lxc-*` and `proxmox-*` flake target against `pve.sweet.home`, plus an audit of the operator's ability to manage the flake/secrets tooling.
|
||||
**Branch:** `worktree-flake-e2e-audit` (this session's isolated worktree)
|
||||
|
||||
## Executive Summary
|
||||
|
||||
The flake itself is in good shape: `nixpkgs-fmt`, `statix`, and a full eval + dry-run build of every host and package are all clean. Every `lxc-*`/`proxmox-*` target's NixOS configuration builds successfully — no target has a broken derivation graph.
|
||||
|
||||
The issues found are **operational, not code-level**:
|
||||
|
||||
1. **pve.sweet.home is critically low on disk space** (91-95% full during this session) and cannot currently build the two largest closures (`gui`, `pxe-boot`) to completion — this actively blocks deploying/redeploying those hosts via the documented workflow.
|
||||
2. **A real, reproducible secrets-decryption failure** was caught live: a stale cached container image (built before a same-day sops-key fix) boots with sshd never starting and every secret failing to decrypt. This is a **general hazard in `create-proxmox-resource.sh`'s "reuse the cached image if present" default**, not a one-off.
|
||||
3. **sops key/anchor drift**: `proxmox-minimal` has a `.sops.yaml` recipient anchor with no corresponding private key anywhere in this environment; several `lxc-*`/`proxmox-*` targets have no sops registration at all yet.
|
||||
4. One concrete script bug was found and **fixed in this session**: `create-proxmox-resource.sh` never enabled the QEMU guest agent channel on VMs it creates, despite the guest OS already running it.
|
||||
5. A management-surface audit (of the operator's ability to run this repo day to day) found 5 process gaps, detailed below.
|
||||
|
||||
Nothing here required or received a `nixos-rebuild switch/boot/test`, `nixos-install`, or any disk-formatting command — all validation was `nix build`/`nix eval`, plus disposable `pct`/`qm` create-then-destroy cycles via the repo's own `create-proxmox-resource.sh`.
|
||||
|
||||
---
|
||||
|
||||
## 1. Static Analysis Results — all clean
|
||||
|
||||
`bash scripts/codex-maintenance.sh --full-check --dry-run` (whole-tree sweep, not just changed files):
|
||||
|
||||
| Check | Result |
|
||||
|---|---|
|
||||
| Secret grep | Clean — only the documented exceptions (installer's own hashed passwords, `access-tokens` comment references) |
|
||||
| `nixpkgs-fmt --check` | 0/53 files would be reformatted |
|
||||
| `statix` | No lint warnings |
|
||||
| nix-cache host key drift check | Up to date |
|
||||
| Full eval of every host's `system.build.toplevel` | All 19 `nixosConfigurations` targets evaluate cleanly |
|
||||
| Dry-run build of every host + package | All succeed, no derivation errors |
|
||||
|
||||
No drift, no formatting issues, no lint findings anywhere in the tree.
|
||||
|
||||
---
|
||||
|
||||
## 2. Per-Target Test Results
|
||||
|
||||
Legend: **LIVE** = built on pve, `pct`/`qm` create → interrogated → destroyed. **BUILD-ONLY** = `nix build` validated the config (mostly `.config.system.build.toplevel`, occasionally `.tarball`), no resource created on pve.
|
||||
|
||||
| Target | Test type | Result | Notes |
|
||||
|---|---|---|---|
|
||||
| `lxc-docker` | BUILD-ONLY | ✅ PASS | Live redeploy skipped — CT105 is already running this identity in production; `--allow-duplicate-host` would have destroyed it. |
|
||||
| `lxc-minimal` | **LIVE** | ✅ PASS (after retry) | First attempt reused a stale cached tarball predating a same-day sops-key commit → activation failed, sshd never started (see Finding #2). Redeployed with `--force-rebuild`: clean boot, `systemctl is-system-running` = `running`, secrets decrypted, sshd listening, users correct. |
|
||||
| `lxc-nix-cache` | BUILD-ONLY | ✅ PASS (after retry) | Live redeploy skipped — CT101 is already running this identity. First local build attempt appeared to hang on a remote-builder handoff to nix-cache; killed and retried with `--builders ""` (local-only), succeeded. |
|
||||
| `lxc-gui` | **LIVE (attempted)** | ⚠️ BLOCKED by pve disk space | Registered a fresh sops key (no prior registration existed), built successfully through the full NixOS system closure, then **failed packaging the tarball**: `No space left on device` on pve's root filesystem. Not a flake defect. |
|
||||
| `lxc-pxe-boot` | **LIVE (attempted)** | ⚠️ BLOCKED by pve disk space | Same failure as `lxc-gui` — this target additionally builds a full nested installer/netboot image (`stage-installer-artifacts.nix`), making it similarly large. Failed with the same `No space left on device` error, immediately after the gui attempt had already consumed pve's remaining headroom. |
|
||||
| `lxc-server` | BUILD-ONLY | ✅ PASS | No sops key registered yet; live deploy also would have hit `boot.zfs.extraPools` trying to import a real ZFS pool that doesn't exist in an isolated test container — an expected limitation of testing this build type outside its real hardware, not a bug. |
|
||||
| `lxc-tailscale-exit-node` | BUILD-ONLY | ✅ PASS | No sops key registered yet. |
|
||||
| `lxc-tor-relay` | BUILD-ONLY | ✅ PASS | Live redeploy skipped — CT106 already holds this identity in production. |
|
||||
| `proxmox-docker` | BUILD-ONLY | ✅ PASS (after retry) | Live redeploy skipped — both CT105 *and* VM103 already hold `docker` identities. Combined `toplevel` + `diskoImagesScript` build crashed with a **Nix-internal assertion failure** (`worker.cc:360`) under this session's memory pressure (see Finding #6) — not a flake bug. Retried with `toplevel` alone: clean. |
|
||||
| `proxmox-minimal` | **LIVE (attempted)** | ⚠️ BLOCKED by key drift → BUILD-ONLY | `.sops.yaml` has a registered `&proxmox-minimal` anchor but **no corresponding private key exists anywhere in this environment** — the script correctly refused to generate a mismatched replacement. Fell back to `toplevel` build: ✅ PASS. |
|
||||
| `proxmox-nix-cache` | BUILD-ONLY | ✅ PASS | No sops key registered yet. |
|
||||
| `proxmox-gui` | BUILD-ONLY | ⚠️ Killed after ~40min (resource-limited) | This session's local build machine has only 2GB RAM; swap filled completely (2.0/2.0GB) and the build stalled, so it was killed rather than risk destabilizing the session further. **Not a flake defect** — the equivalent `gui` NixOS configuration already proved fully buildable during the `lxc-gui` live attempt above (it built the entire system closure successfully and only failed at the pve-side tarball-packaging step due to disk space, not the config). |
|
||||
| `proxmox-pxe-boot` | BUILD-ONLY | ⚠️ Killed after ~35min (resource-limited) | Was deep into building the nested installer's kernel initrd (this build type bundles a full netboot installer image via `stage-installer-artifacts.nix`) when killed to keep the audit moving. **Not a flake defect** — this target's own module logic was already effectively validated via the earlier *live* pve deploy attempt (`lxc-pxe-boot` above), which built the complete image and only failed at the final tarball-packaging step due to pve's disk space (Finding 1). |
|
||||
| `proxmox-server` | BUILD-ONLY | ✅ PASS | No sops key registered yet; same ZFS-pool caveat as `lxc-server` would apply to a live deploy. |
|
||||
| `proxmox-tailscale-exit-node` | BUILD-ONLY | ✅ PASS | No sops key registered yet. |
|
||||
|
||||
**Not tested at all:** `linode-*` targets (not deployable to Proxmox) and `installer` (not a normal host) — both were still covered by the static eval/dry-run-build sweep above.
|
||||
|
||||
---
|
||||
|
||||
## 3. Findings, Ranked by Severity
|
||||
|
||||
### Finding 1 — pve.sweet.home is critically low on disk space (blocks real deployments)
|
||||
|
||||
At session start: `/dev/mapper/pve-root` was **95% full, 5.3GB free** (of 94GB). After two failed large builds it recovered slightly to **91% full, 8.2GB free** (nix cleans up its own failed-build scratch space). `/nix/store` alone is 26GB; `nix-store --gc --print-dead` reports **zero** reclaimable garbage — everything currently in the store is a live GC root, so `nix-collect-garbage` won't help without first removing old roots.
|
||||
|
||||
**Why it matters:** `create-proxmox-resource.sh` builds every VM/CT image **directly on pve**, not on a build machine and transferred over. With <10GB headroom, any closure approaching a few GB (the `gui` build type: full Cinnamon desktop + Firefox + LibreOffice + GIMP + VS Code + xrdp; the `pxe-boot` build type: nginx/atftpd *plus* an entire nested installer/netboot image) cannot currently be built there at all. Both `lxc-gui` and `lxc-pxe-boot` failed live with `No space left on device` during this audit.
|
||||
|
||||
**Recommended action:** Expand `pve-root`'s LV, or free space by pruning old container templates in `/var/lib/vz/template/cache` (1.5GB) / old backups in `/var/lib/vz/dump` (306MB) / auditing what's pinning 26GB of `/nix/store` as live GC roots (likely `result-*` symlinks — see below). This is real production disk state; **not something this session touched or fixed** — it needs the operator's judgment on what's safe to remove.
|
||||
|
||||
**Secondary, smaller finding:** every `create-proxmox-resource.sh` run leaves a `result-<target>` symlink in the node's repo checkout as a permanent GC root (`ls /root/nixos/result-*` on pve showed 3 from this session alone: `lxc-docker`, `lxc-minimal`, `lxc-nix-cache`). These accumulate forever and pin their entire closures in the store. Consider having the script clean up its own `result-*` link after staging the built artifact (or use a temp `--out-link` under `/tmp`), so `nix-collect-garbage` can actually reclaim old build outputs.
|
||||
|
||||
### Finding 2 — Stale cached images can silently ship broken secrets (reproduced live)
|
||||
|
||||
`create-proxmox-resource.sh`'s default behavior is: if the node already has `<target>.tar.xz`/`.raw` staged, **reuse it** — only `--force-rebuild` forces a fresh build. This session hit exactly the failure mode `docs/auto-installer.md` already warns about: `lxc-minimal`'s cached tarball (built 2026-07-20T15:57Z) predated a same-day sops-key fix commit (2026-07-20T17:49Z, "clean up in ailse 3"). The deployed container booted with:
|
||||
|
||||
```
|
||||
sops-install-secrets: failed to decrypt '.../common.yaml': Error getting data key: 0 successful groups required, got 0
|
||||
Activation script snippet 'setupSecrets' failed (1)
|
||||
```
|
||||
|
||||
— every secret permanently failed to decrypt, `sshd` never started (though the container otherwise looked "running"). This was **not a code bug**: the currently-committed `secrets/common.yaml` decrypts fine for that host's key when checked independently; the *cached artifact on pve* simply reflected an older commit's ciphertext. Redeploying with `--force-rebuild` fixed it immediately.
|
||||
|
||||
**Why it matters:** this is silent and easy to trigger by accident — any operator who redeploys a host without remembering `--force-rebuild` after a secrets change gets a container that looks like it started (`pct start` succeeds, `pct status` = running) but is completely inaccessible.
|
||||
|
||||
**Recommended action:** Have `create-proxmox-resource.sh` compare the cached image's build timestamp (or embed the source commit hash in the staged filename) against current HEAD, and warn (or refuse without `--force-rebuild`) if they differ — rather than silently trusting presence alone.
|
||||
|
||||
### Finding 3 — sops key/anchor drift
|
||||
|
||||
Two concrete instances hit live during this session:
|
||||
|
||||
- **`proxmox-minimal`**: `.sops.yaml` already has a registered `&proxmox-minimal` age recipient, but this environment's `host-keys/` directory has no corresponding private key file. `sync-host-keys.sh` correctly refused to generate a replacement (it would silently mismatch whatever's already registered/deployed) — but this means **no environment currently has this host's private key**, unless it exists on some other machine that was never backed up here.
|
||||
- **`lxc-gui`**, and by the same logic `lxc-server`/`lxc-tailscale-exit-node`/most `proxmox-*` targets, have **no sops registration at all yet** — expected for undeployed hosts per `docs/auto-installer.md`, but this session's live-testing needed to register `lxc-gui`'s key on the fly, which immediately hit **Finding 3b**: registering a key locally does nothing for pve's build until it's pushed to `origin/main` (pve builds via `git pull`, not from this uncommitted worktree). This is exactly gap #4 the management-surface audit (below) already flagged in the abstract — this session hit it concretely.
|
||||
|
||||
**Recommended action:** for `proxmox-minimal`, decide whether to regenerate its key (destroying old-key decrypt access, if anything still holds it) or track down wherever the original private key lives and back it up here. For the general pattern, see the management-surface audit's recommendation to pre-flight-check key registration before building.
|
||||
|
||||
### Finding 4 — QEMU guest agent never wired up (found and fixed this session)
|
||||
|
||||
`modules/common/configuration.nix:44` sets `services.qemuGuest.enable = true` on every host — the guest-side agent daemon is correctly enabled everywhere. But `scripts/proxmox/create-proxmox-resource.sh`'s `qm create` call never passed `--agent 1`, so **Proxmox never created the virtio-serial channel** the agent needs. Every `proxmox-*` VM this script ever created was silently missing `qm guest exec`/IP-address reporting in the Proxmox UI, despite the guest daemon actually running.
|
||||
|
||||
**Status: fixed in this session's worktree** (`scripts/proxmox/create-proxmox-resource.sh`, `qm create` now includes `--agent enabled=1`) — see the diff, included in the PR from this session.
|
||||
|
||||
### Finding 5 — Orphaned container on pve (CT102)
|
||||
|
||||
`pve.sweet.home` has a stopped LXC container, **VMID 102**, with an essentially empty config (`lock: create` and nothing else — no hostname, no rootfs, no network) — the leftover of a `pct create` that started and never finished. It predates this session (not created by any of this audit's activity) and wasn't touched. **Recommend the operator confirm it's abandoned and remove it** (`pct destroy 102 --purge 1`) — left as-is it may be someone's genuine in-progress work, so it wasn't assumed safe to delete autonomously.
|
||||
|
||||
### Finding 6 — Nix-internal crash under memory pressure (tooling, not flake)
|
||||
|
||||
Building `proxmox-docker`'s `toplevel` and `diskoImagesScript` together crashed with a Nix-internal assertion failure (`Assertion '!awake.empty()' failed ... worker.cc:360`, a known class of bug in Nix's multi-goal build scheduler) while this session's 2GB-RAM build container was under heavy swap pressure (1.8-2.0/2GB swap in use) from a separate concurrent build. Retrying the same target alone (no concurrency) succeeded cleanly. **Not a flake defect** — purely an artifact of this session's constrained build environment; noted for completeness since it looked alarming in isolation.
|
||||
|
||||
### Finding 7 — Management-surface audit: 5 operability gaps
|
||||
|
||||
A focused audit of "can the operator actually run this repo day to day" (flake, home-manager, sops, related scripts) found:
|
||||
|
||||
1. **No documented recovery path if the `&admin` sops age key is lost without a backup.** `scripts/secrets/backup-admin-key.sh` exists and works but is referenced nowhere in `README.md`/`docs/` — no forcing function ensures a backup was ever taken. `rotate-admin-key.sh` requires the *old* key to re-key; there's no bootstrap-from-nothing path documented (the real fallback — deriving an age identity from any still-live host's own SSH key — isn't written down anywhere).
|
||||
2. **home-manager has no standalone iteration path.** It's wired only inside `nixosConfigurations` (`flake.nix`) — no `homeConfigurations` output. The fastest real shortcut (`nix build .#nixosConfigurations.<target>.config.home-manager.users.nixos.home.activationPackage`) isn't documented anywhere, so the practical workflow is a full host rebuild to test one HM tweak.
|
||||
3. **Gitea's flake-lock-update workflow pushes straight to `main` with no pre-merge validation.** `.gitea/workflows/update-flake-lock.yml` commits and pushes `nix flake update`'s result directly; `codex-maintenance.sh` only runs *after*, on the resulting push — a genuinely broken lockfile bump lands on `main` before anything catches it. (The GitHub-side workflow is safer — PR-based — but has the opposite gap: nothing alerts if the PR sits unmerged.)
|
||||
4. **No pre-flight check that a build target has a registered sops key before building it.** `docs/auto-installer.md` documents the failure mode (silent, total secrets-decrypt failure) but nothing in `create-proxmox-resource.sh` refuses to proceed when it's about to build a target with no `.sops.yaml` anchor — it's on the operator to remember. This session's `lxc-gui` test hit close to this exact gap (needed the key added on the fly, mid-session).
|
||||
5. **`vars.remoteBuilderAuthorizedKeys` has the same drift risk as `vars.nixCacheHostKey`, but no checker script.** `sync-nix-cache-host-key.sh --check` guards the latter; the former (and `vars.pxeServerIp`/`vars.pbsIp`) has no equivalent — a rotated/revoked client key just silently stops working with no diagnostic pointing back here.
|
||||
|
||||
---
|
||||
|
||||
## 4. Action Plan (priority order)
|
||||
|
||||
1. **Free up disk space on pve.sweet.home** (or expand `pve-root`). Blocking: `lxc-gui`, `proxmox-gui`, `lxc-pxe-boot`, `proxmox-pxe-boot` cannot currently be built/redeployed on this node at all.
|
||||
2. **Decide on `proxmox-minimal`'s orphaned sops key**: locate the original private key and back it up here, or accept regenerating it (breaks decrypt access for whoever/whatever currently holds the old one).
|
||||
3. **Merge this session's PR** (see below) to get the `--agent 1` fix and `lxc-gui`'s new sops registration onto `main` — required before `lxc-gui` can be live-redeployed with working secrets.
|
||||
4. **Add a staleness guard to `create-proxmox-resource.sh`'s cache-reuse path** (Finding 2) — highest-leverage fix, since it silently produces a broken-but-"running" host.
|
||||
5. **Add a pre-flight sops-anchor check to `create-proxmox-resource.sh`** (management-surface gap #4) — same root cause class as #4 above, catch it before building instead of at first boot.
|
||||
6. Investigate/clean up **CT102** on pve (Finding 5) — confirm abandoned, then remove.
|
||||
7. Document `backup-admin-key.sh` in `README.md`'s Security Notes and add the live-host-key bootstrap-recovery procedure to `docs/` (management-surface gap #1).
|
||||
8. Add pre-push validation to the Gitea flake-lock-update workflow (management-surface gap #3).
|
||||
9. Lower-priority: document the home-manager `activationPackage` shortcut (gap #2); extend `sync-nix-cache-host-key.sh`'s drift-check pattern to `remoteBuilderAuthorizedKeys` (gap #5).
|
||||
10. Follow-up session: finish build-validating `proxmox-gui` and `proxmox-pxe-boot` (both killed here after 35-40min on this session's 2GB-RAM machine — not failures, just unfinished) once pve has headroom (item 1) — ideally from a machine with more RAM. `proxmox-server` and `proxmox-tailscale-exit-node` already passed build-only validation in this session, no follow-up needed.
|
||||
|
||||
---
|
||||
|
||||
## 5. Uncommitted Changes From This Session
|
||||
|
||||
This worktree (`worktree-flake-e2e-audit`) currently has:
|
||||
|
||||
- `scripts/proxmox/create-proxmox-resource.sh` — the `--agent enabled=1` fix (Finding 4).
|
||||
- `.sops.yaml` / `secrets/common.yaml` — `lxc-gui`'s new age key registered as a recipient (generated live during this session's testing).
|
||||
|
||||
Per this session's standard workflow, these will be committed, pushed, and opened as a draft PR rather than pushed to `main` directly — merging it is the operator's call, and is also **prerequisite to live-redeploying `lxc-gui` successfully** (its build will keep hitting the sops-staleness failure from Finding 2 on pve until this registration is on `origin/main`).
|
||||
@@ -27,81 +27,9 @@ machines when deployed.
|
||||
template for a *real* host — every other host uses sops-nix
|
||||
(`hashedPasswordFile`, see "Security Notes" in `README.md`). Flag any *new*
|
||||
secret-like string you encounter instead of committing it.
|
||||
- `host-keys/` is gitignored — used only by the auto-installer's own
|
||||
environment for pre-seeding non-LXC host keys before first boot (see
|
||||
`docs/auto-installer.md`). Never commit its contents; if `git status`
|
||||
ever shows it as trackable, something is wrong. All deployed hosts use
|
||||
clan vars (`vars/per-machine/<target>/openssh/`, committed and
|
||||
sops-encrypted) for their SSH host keys — those ARE tracked by git and
|
||||
belong in the repo.
|
||||
|
||||
### Two Proxmox nodes: `pve1.sweet.home` (production) and `pve-test.sweet.home` (sandbox)
|
||||
|
||||
There are two SSH-reachable Proxmox nodes on the LAN, both defined in
|
||||
`scripts/env.sh` (`PVE1_HOST` / `PVE_TEST_HOST`), individually targetable
|
||||
via `scripts/proxmox/create-proxmox-resource.sh --node <host>` or by
|
||||
overriding `PROXMOX_HOST`. `PROXMOX_HOST` itself still defaults to
|
||||
`PVE1_HOST` (production) — that default, and every other script behavior,
|
||||
is unchanged from before `pve-test` existed; the only thing new is that
|
||||
`pve-test` can now be reached at all. They are **not interchangeable** —
|
||||
one is real production infrastructure, the other exists specifically so
|
||||
there's somewhere safe to test. The restriction below is a policy for
|
||||
Claude specifically, not a change to the tooling's own default or
|
||||
anything the operator needs to opt into.
|
||||
|
||||
#### `pve1.sweet.home` (production — off-limits to Claude)
|
||||
|
||||
A real, live Proxmox node hosting production VMs/containers — not a
|
||||
sandbox, and not Claude's to touch by default.
|
||||
|
||||
- **Off-limits at all times unless the operator has given explicit,
|
||||
same-session instructions to act on this specific host.** That
|
||||
authorization is scoped to the task it was given for — don't carry it
|
||||
forward to unrelated later work in the same conversation, and never
|
||||
assume it from a previous session.
|
||||
- **Read-only for existing state is always fine, authorization or not.**
|
||||
You may SSH in (or use `pvesm`, `qm list`, `pct list`, `qm config`, `pct
|
||||
config`, the Proxmox API, etc.) to inspect the node's config, storage,
|
||||
and any existing VM/container — including ones this repo didn't create.
|
||||
- **Never** modify, stop, restart, delete, reconfigure, or create anything
|
||||
on this node (`qm set`, `pct set`, `qm destroy`, `pct destroy`, `qm
|
||||
stop`, `pct stop`, `qm create`, `pct create`, snapshot operations,
|
||||
storage changes, etc.) — including scratch/test resources — without
|
||||
that explicit go-ahead. Use `pve-test.sweet.home` for anything
|
||||
exploratory instead; it exists precisely so `pve1` never has to be the
|
||||
answer to "where do I test this."
|
||||
- **This is a Claude-specific policy, not something the scripts enforce.**
|
||||
`scripts/env.sh`/`create-proxmox-resource.sh` default to `pve1` exactly
|
||||
as they did before `pve-test` existed, with no extra flag or prompt
|
||||
required — that's deliberate, so the operator's own existing workflows
|
||||
don't change. Claude, however, must never rely on that default: every
|
||||
Proxmox action Claude takes on its own initiative — not explicitly
|
||||
pointed at `pve1` by the operator this session — targets `pve-test`
|
||||
instead (e.g. `--node "$PVE_TEST_HOST"`, or `PROXMOX_HOST=$PVE_TEST_HOST`).
|
||||
Claude's own default is `pve-test`, full stop, regardless of what the
|
||||
tooling's own unqualified default happens to be.
|
||||
|
||||
#### `pve-test.sweet.home` (sandbox — Claude's default target)
|
||||
|
||||
A separate Proxmox node set aside for testing. The *tooling's* default is
|
||||
still production (`PROXMOX_HOST` → `PVE1_HOST`, see above) — but
|
||||
**Claude's own default is this node**: absent an explicit, same-session
|
||||
instruction to use `pve1`, every Proxmox action Claude initiates targets
|
||||
`pve-test`. Once targeted, it's safe to create, interrogate, and destroy
|
||||
resources on without asking first.
|
||||
|
||||
- **Test VMs/containers are allowed, but must be torn down.** Create a
|
||||
scratch VM or container here (e.g. via
|
||||
`scripts/proxmox/create-proxmox-resource.sh` or raw `qm`/`pct create`)
|
||||
to validate something. Anything created this way must be destroyed
|
||||
again in the same session, before ending the task — never leave a test
|
||||
resource running. Use a VMID/name that's obviously scratch (and doesn't
|
||||
collide with a real flake target) so it's unambiguous what's safe to
|
||||
remove.
|
||||
- **Node-level config is still not yours to change.** Creating/destroying
|
||||
your own scratch guests is fine; Proxmox host config, storage pools, and
|
||||
networking on `pve-test` itself are still the operator's call to make
|
||||
manually, same as on `pve1`.
|
||||
- `host-keys/` is gitignored — locally-generated *private* SSH host keys for
|
||||
the auto-installer (see `docs/auto-installer.md`). Never commit its
|
||||
contents; if `git status` ever shows it as trackable, something is wrong.
|
||||
|
||||
## Commands
|
||||
|
||||
@@ -109,22 +37,11 @@ resources on without asking first.
|
||||
# One-time environment bootstrap (installs Nix if missing, prints hosts)
|
||||
bash scripts/codex-setup.sh
|
||||
|
||||
# Changed-files-only validation: secret grep (whole repo), nixpkgs-fmt --check
|
||||
# and statix on changed *.nix files, eval of the hosts/packages those changes
|
||||
# can affect. This is what CI runs on every push/PR.
|
||||
# Full validation: secret grep, nixpkgs-fmt --check, statix lint, eval all hosts
|
||||
bash scripts/codex-maintenance.sh
|
||||
|
||||
# Full sweep: nixpkgs-fmt --check/statix over the whole tree, eval every host
|
||||
# and package. Slow (minutes) -- CI never runs this; use it locally before a
|
||||
# release or after touching modules/common/*, flake.nix, or variables.nix for
|
||||
# extra confidence beyond the automatic full-fallback those paths already
|
||||
# trigger in the default mode (see below).
|
||||
bash scripts/codex-maintenance.sh --full-check
|
||||
|
||||
# Either mode, plus a dry-run build (no result symlink) of every host/package
|
||||
# in whichever scope is active
|
||||
bash scripts/codex-maintenance.sh --dry-run
|
||||
bash scripts/codex-maintenance.sh --full-check --dry-run
|
||||
# Same, plus a dry-run build (no result symlink) of every host's toplevel
|
||||
bash scripts/codex-maintenance.sh dry-run
|
||||
|
||||
# List the hosts the flake currently exposes
|
||||
nix eval --json .#nixosConfigurations --apply builtins.attrNames | jq -r '.[]'
|
||||
@@ -141,125 +58,35 @@ maintenance script pulls them via `nix run github:NixOS/nixpkgs/nixos-25.11#<too
|
||||
There is no test suite — "correctness" here means the flake evaluates and
|
||||
`nixpkgs-fmt`/`statix` are clean.
|
||||
|
||||
With no flags, `codex-maintenance.sh` diffs against a base ref (env
|
||||
`MAINT_BASE_SHA`, else the PR base SHA in CI, else `HEAD^` locally) and scopes
|
||||
fmt-check/statix to the changed `*.nix` files and eval to the hosts/packages
|
||||
those changes can affect — a `hosts/<name>/host.nix` edit only evals that
|
||||
host's targets, a `modules/platforms/<platform>.nix` edit only evals that
|
||||
platform's hosts, and so on. A change to `flake.nix`, `flake.lock`,
|
||||
`variables.nix`, `modules/common/*`, or any other `modules/*.nix` file outside
|
||||
`platforms/`/`build-types/` (whose blast radius isn't safely inferable from
|
||||
the path alone) falls back to evaluating every host and package, same as
|
||||
`--full-check` would, just without the whole-tree fmt/statix sweep. This
|
||||
exists because the whole-tree sweep is what was timing out CI; **CI always
|
||||
runs the plain, no-flag form and never passes `--full-check`.**
|
||||
|
||||
The default mode's diff is against the working tree (uncommitted and staged
|
||||
edits included, not just committed ones), so it's already the right tool for
|
||||
an interactive session too: after editing one or two hosts/modules, plain
|
||||
`bash scripts/codex-maintenance.sh` naturally scopes to just what you
|
||||
touched. Reserve `--full-check` for changes that plausibly affect every host
|
||||
(`modules/common/*`, `flake.nix`, `variables.nix` — though the default mode
|
||||
already falls back to evaluating everything for those paths, `--full-check`
|
||||
additionally re-checks fmt/statix over the whole tree) or as a final check
|
||||
before committing.
|
||||
**In an interactive agent session**, prefer targeted checks over full-repo
|
||||
sweeps: after editing one or two hosts/modules, evaluate just the
|
||||
`nixosConfigurations.<host>` you touched (plus any `config.system.build.tarball`
|
||||
/`diskoImagesScript`/package output affected) rather than looping over every
|
||||
host — `codex-maintenance.sh` evaluates 18 hosts plus every package/tarball/
|
||||
image variant now and is slow to run after each small change. Reserve a full
|
||||
`codex-maintenance.sh` run for changes that plausibly affect every host
|
||||
(`modules/common/*`, `flake.nix`, `variables.nix`) or as a final check before
|
||||
committing. This is a session-workflow preference only — it does not apply to
|
||||
CI, which should keep running the full script on every push/PR regardless of
|
||||
diff size; that's the point of it.
|
||||
|
||||
## Scripts
|
||||
|
||||
Beyond `codex-setup.sh`/`codex-maintenance.sh` above, `scripts/` is
|
||||
organized by purpose: `scripts/secrets/` (sops/age + SSH host-key
|
||||
management), `scripts/proxmox/` (Proxmox deployment), `scripts/installer/`
|
||||
(the auto-installer's own shell script, templated into the image — see
|
||||
below), `scripts/lib/` (shared helpers, sourced by the scripts below — not
|
||||
run directly), and a handful of repo-wide scripts left at the top level
|
||||
(`env.sh`, `bump-nixpkgs-release.sh`, plus `codex-setup.sh`/
|
||||
`codex-maintenance.sh` above). When adding a new script, put it in the
|
||||
matching subfolder rather than the top level, and if it duplicates logic
|
||||
another script already has, lift the shared part into `scripts/lib/`
|
||||
instead of copying it.
|
||||
Beyond `codex-setup.sh`/`codex-maintenance.sh` above, `scripts/` also has:
|
||||
|
||||
### `scripts/installer/`
|
||||
|
||||
- `scripts/installer/auto-install.sh` — the interactive install script
|
||||
baked into the auto-installer image (see `docs/auto-installer.md`), kept
|
||||
as a real, version-controlled shell file rather than inline in
|
||||
`modules/installer/common.nix`'s Nix. It sources `scripts/env.sh` itself
|
||||
for `LAN_DOMAIN` (`export LAN_DOMAIN`/`: "${LAN_DOMAIN:=...}"`, matching
|
||||
`variables.nix`'s `lanDomain` — manually kept in sync, same pattern as
|
||||
`NIX_CACHE_HOST` mirroring `nixCacheHost`), rather than Nix-level string
|
||||
substitution — that's what makes it work identically whether run
|
||||
straight from a git checkout or from inside the built installer image.
|
||||
`common.nix` bakes `scripts/env.sh` in alongside it at a matching
|
||||
relative path (`/etc/nixos-installer/env.sh` next to
|
||||
`/etc/nixos-installer/installer/auto-install.sh`) so the script's own
|
||||
`source "$(dirname ...)/../env.sh"` line resolves the same way in both
|
||||
contexts — this is also why it's invoked from
|
||||
`/etc/nixos-installer/installer/auto-install.sh` rather than a flat
|
||||
`/etc/auto-install.sh`. `#!/usr/bin/env bash`, not
|
||||
`#!/run/current-system/sw/bin/bash`: the latter only resolves on an
|
||||
already-activated NixOS system, breaking the checked-out-file case
|
||||
entirely (confirmed live: "cannot execute: required file not found" on
|
||||
a non-NixOS box); `/usr/bin/env` is reliably present on both NixOS
|
||||
(`environment.usrbinenv`'s own default) and any normal Linux distro.
|
||||
|
||||
### `scripts/secrets/`
|
||||
|
||||
- `scripts/secrets/sync-host-keys.sh` — generates/registers SSH host keys
|
||||
and their `.sops.yaml`/`secrets/*.yaml` recipients for flake targets,
|
||||
idempotently (`--all`, `<target>`, `--remove`, `--regenerate-all-keys`,
|
||||
all with `--dry-run`). Stores keys as clan vars
|
||||
(`vars/per-machine/<target>/openssh/`, committed and sops-encrypted) for
|
||||
all flake targets. The primary tool for provisioning a new host's
|
||||
secrets access — see "Creating a new machine" in
|
||||
`docs/auto-installer.md`.
|
||||
- `scripts/secrets/prepare-host-key.sh` — narrower predecessor: generates a
|
||||
key by an arbitrary name without touching `.sops.yaml`. Still useful to
|
||||
- `scripts/sync-host-keys.sh` — generates/registers SSH host keys and their
|
||||
`.sops.yaml`/`secrets/*.yaml` recipients for flake targets, idempotently
|
||||
(`--all`, `<target>`, `--remove`, `--regenerate-all-keys`, all with
|
||||
`--dry-run`). The primary tool for provisioning a new host's secrets
|
||||
access — see "Creating a new machine" in `docs/auto-installer.md`.
|
||||
- `scripts/prepare-host-key.sh` — narrower predecessor: generates a key by
|
||||
an arbitrary name without touching `.sops.yaml`. Still useful to
|
||||
pre-generate a key before its flake target exists yet, since
|
||||
`sync-host-keys.sh` can only act on targets `nixosConfigurations` already
|
||||
has.
|
||||
- `scripts/secrets/rotate-admin-key.sh <backup-admin-key> [--new-key-file
|
||||
<path>] [--dry-run]` — rotates `.sops.yaml`'s `&admin` age key: decrypts
|
||||
with a backed-up copy of the key currently trusted as `&admin` (verified
|
||||
by deriving its public key and comparing, not taken on faith), replaces
|
||||
the `&admin` line with a new key already present in the environment
|
||||
(defaults to wherever sops/age itself would look), and runs
|
||||
`sops updatekeys` on every `secrets/*.yaml`. One-way: the old key can no
|
||||
longer decrypt anything re-encrypted this way. This is the automation
|
||||
for the manual steps `sync-host-keys.sh`/`create-proxmox-resource.sh`
|
||||
print when they bootstrap a brand-new, not-yet-trusted key on a machine
|
||||
with no prior admin access.
|
||||
- `scripts/secrets/backup-admin-key.sh <dest-path> [--key-file <path>]
|
||||
[--force] [--dry-run]` — copies the local sops age key (source
|
||||
resolution matches sops/age itself: `$SOPS_AGE_KEY` inline, then
|
||||
`--key-file`, then `$SOPS_AGE_KEY_FILE`, then the XDG default) to an
|
||||
arbitrary destination path with `0600` permissions, validating it's a
|
||||
real age identity and round-tripping the public key before and after the
|
||||
write. Refuses to overwrite an existing `<dest-path>` without `--force`.
|
||||
Purely a local filesystem copy — never touches `.sops.yaml`/
|
||||
`secrets/*.yaml` or the repo at all. The resulting file is exactly what
|
||||
`rotate-admin-key.sh` expects as its backup-key argument.
|
||||
- `scripts/secrets/sync-nix-cache-host-key.sh [--check] [--dry-run]
|
||||
[--host <name>]` — detects drift between the ed25519 SSH host key
|
||||
nix-cache is actually serving right now (via `ssh-keyscan`) and
|
||||
`vars.nixCacheHostKey` (`variables.nix`), the value
|
||||
`modules/nix-cache/remote-builder-client.nix` bakes into every real
|
||||
client's declarative `programs.ssh.knownHosts` and
|
||||
`configure-nix-cache-client.sh` hardcodes as its own default for
|
||||
non-NixOS clients. That value has no automatic source of truth — it's
|
||||
set once from whatever nix-cache's host key happened to be at the time,
|
||||
and silently goes stale if the host is ever rebuilt/recreated with a new
|
||||
key, breaking every client's distributed-build SSH trust with no error
|
||||
that points back here. `--check` (used by `codex-maintenance.sh`, which
|
||||
treats an unreachable nix-cache — e.g. from a non-LAN CI runner — as a
|
||||
silent skip rather than a failure) only reports drift; the no-flags form
|
||||
updates both files in place. Declarative clients still need a rebuild to
|
||||
pick up the fix.
|
||||
|
||||
### `scripts/proxmox/`
|
||||
|
||||
- `scripts/proxmox/create-proxmox-resource.sh` — builds a `lxc-*`/
|
||||
`proxmox-*` target's tarball/disk image and creates it on a real Proxmox
|
||||
node (`pct create` against the tarball as a CT template / `qm create`+
|
||||
- `scripts/create-proxmox-resource.sh` — builds a `lxc-*`/`proxmox-*`
|
||||
target's tarball/disk image and creates it on a real Proxmox node
|
||||
(`pct create` against the tarball as a CT template / `qm create`+
|
||||
`importdisk`), or reconfigures an existing resource's cores/memory/disk
|
||||
size (`--modify`, always requires typing the VMID back to confirm).
|
||||
Checks for an already-uploaded image on the node before building
|
||||
@@ -269,71 +96,22 @@ instead of copying it.
|
||||
Refuses to create a target whose host identity already exists live on
|
||||
the node (checked directly via `qm`/`pct`, not any file in this repo)
|
||||
unless `--allow-duplicate-host` is passed. `--dry-run` throughout both
|
||||
modes. The first time it has to bootstrap build tooling on a node (i.e.
|
||||
`nix` wasn't already on its `PATH`), it also runs
|
||||
`scripts/proxmox/configure-nix-cache-client.sh` there (non-fatally — a
|
||||
failure just falls back to building from source / `cache.nixos.org`) so
|
||||
the node substitutes from and can offload builds to nix-cache on every
|
||||
subsequent run, not just this one.
|
||||
- `scripts/proxmox/configure-nix-cache-client.sh [--dry-run]
|
||||
[--no-remote-builder] [--no-restart]` — the non-NixOS equivalent of
|
||||
`modules/nix-cache/client.nix`/`remote-builder-client.nix`, for a plain
|
||||
Debian machine with the Nix package manager (not NixOS) already
|
||||
installed: run as root *on that machine* to add nix-cache as a
|
||||
substituter in `/etc/nix/nix.conf` (`https://cache.nixos.org/` kept as
|
||||
fallback) via `extra-substituters`/`extra-trusted-public-keys` so it
|
||||
layers on top of whatever's already there instead of clobbering it, and,
|
||||
if `/root/.ssh/nixremote` is already present (see docs/nix-cache.md
|
||||
"Remote builder SSH keys"), configures it as a distributed-build
|
||||
machine too and trusts nix-cache's SSH host key in
|
||||
`/etc/ssh/ssh_known_hosts`. Idempotent (re-running replaces its own
|
||||
marked block rather than duplicating it); restarts `nix-daemon` by
|
||||
default so the change takes effect immediately.
|
||||
|
||||
### `scripts/lib/`
|
||||
|
||||
Sourced by the scripts above, never run directly:
|
||||
|
||||
- `nix-bootstrap.sh` — `NIX_CONFIG`/`ensure_nix_profile`, shared by
|
||||
`codex-setup.sh`/`codex-maintenance.sh` and the remote build commands
|
||||
`create-proxmox-resource.sh` runs over SSH.
|
||||
- `nix-eval.sh` — `NIX_EVAL_FLAGS` plus `list_flake_targets`/
|
||||
`flake_target_hostname` flake-introspection helpers.
|
||||
- `ssh-host-keys.sh` — `generate_host_ed25519_key`/`ssh_pubkey_to_age`,
|
||||
shared by `sync-host-keys.sh` and `prepare-host-key.sh`.
|
||||
- `sops-age.sh` — `age_pubkey_from_identity_file`/`sops_yaml_admin_pubkey`/
|
||||
`sops_updatekeys` plus the shared sops/age default key-file resolution,
|
||||
shared by `backup-admin-key.sh`, `rotate-admin-key.sh`, and
|
||||
`sync-host-keys.sh`.
|
||||
- `confirm.sh` — `confirm_typed`, the "type X back to confirm" destructive-
|
||||
action prompt shared by `create-proxmox-resource.sh` and
|
||||
`sync-host-keys.sh`.
|
||||
- `sync-host-keys-edit-sops.py` — the `.sops.yaml` anchor/key_groups editor
|
||||
`sync-host-keys.sh` shells out to (see that script for why: precise,
|
||||
idempotent YAML edits are impractical in bash).
|
||||
|
||||
### Top level
|
||||
|
||||
modes.
|
||||
- `scripts/env.sh` — shared config (`PROXMOX_HOST`, storage pool, bridge,
|
||||
default cores/memory, `NIX_CACHE_HOST`, `LAN_DOMAIN`) sourced by
|
||||
`create-proxmox-resource.sh` and `scripts/installer/auto-install.sh`. Add
|
||||
new cross-script config here instead of duplicating it per-script.
|
||||
default cores/memory) sourced by `create-proxmox-resource.sh`. Add new
|
||||
cross-script config here instead of duplicating it per-script.
|
||||
- `scripts/bump-nixpkgs-release.sh` — bumps `flake.nix`'s `nixpkgs.url`/
|
||||
`home-manager.url` in place. Exists because flake input URLs can't
|
||||
reference `variables.nix` (confirmed empirically — `nix flake metadata`
|
||||
errors on it), so this is the closest equivalent to a single source of
|
||||
truth for the tracked release.
|
||||
|
||||
`sync-host-keys.sh`, `create-proxmox-resource.sh`, and
|
||||
`rotate-admin-key.sh` genuinely mutate real state when run for real (not
|
||||
`--dry-run`): real `secrets/*.yaml` recipients, real Proxmox VMs/
|
||||
containers, real revocation of decrypt access. They require the
|
||||
operator's own SSH/sops access, which an agent session doesn't have — but
|
||||
don't suggest running any of them non-dry-run without the operator's
|
||||
explicit go-ahead even if it becomes technically reachable.
|
||||
`backup-admin-key.sh` only writes a key copy to a path the operator gives
|
||||
it — lower-stakes than the others, but it still handles a real private
|
||||
key, so treat its destination path choice as the operator's call too.
|
||||
`sync-host-keys.sh` and `create-proxmox-resource.sh` genuinely mutate real
|
||||
state when run for real (not `--dry-run`): real `secrets/*.yaml`
|
||||
recipients, real Proxmox VMs/containers. They require the operator's own
|
||||
SSH/sops access, which an agent session doesn't have — but don't suggest
|
||||
running either non-dry-run without the operator's explicit go-ahead even
|
||||
if it becomes technically reachable.
|
||||
|
||||
## Architecture
|
||||
|
||||
@@ -355,14 +133,10 @@ nixosSystem {
|
||||
}
|
||||
```
|
||||
|
||||
Platforms: `linode`, `proxmox`, `lxc`, `baremetal`. Build types: `minimal`,
|
||||
`nix-cache`, `server`, `docker`, `gui`, `pxe-boot`, `tailscale-exit-node`,
|
||||
`tor-relay`. Not every combination is built — e.g. `pxe-boot` has no `linode`
|
||||
variant (PXE/DHCP/TFTP need LAN L2 adjacency a Linode VPS doesn't have),
|
||||
`tor-relay` currently only exists as `lxc-tor-relay`, and `baremetal`
|
||||
currently only exists as `baremetal-gui` (the real gui-host hardware —
|
||||
see `hosts/nixos/host.nix` and `modules/platforms/baremetal.nix`). Treat
|
||||
`flake.nix`'s
|
||||
Platforms: `linode`, `proxmox`, `lxc`. Build types: `minimal`, `nix-cache`,
|
||||
`server`, `docker`, `gui`, `pxe-boot`, `tailscale-exit-node`. Not every
|
||||
combination is built — e.g. `pxe-boot` has no `linode` variant (PXE/DHCP/TFTP
|
||||
need LAN L2 adjacency a Linode VPS doesn't have). Treat `flake.nix`'s
|
||||
`generatedTargets` as the source
|
||||
of truth for which hosts exist — `README.md`, `AGENTS.md`,
|
||||
`docs/flake-lock-automation.md`, and the CI eval workflows
|
||||
@@ -378,23 +152,19 @@ removing a host.
|
||||
of their own beyond narrow parameterized helpers (see
|
||||
`modules/beszel/host-token.nix` below) — all shared behavior comes from the
|
||||
platform/build-type modules composed in `flake.nix`, not from the host file.
|
||||
- `modules/platforms/{linode,proxmox,lxc,baremetal}.nix` — platform-specific
|
||||
config: boot method, guest tooling, and the hardware config, imported
|
||||
directly by the platform module itself — **not** wired in from
|
||||
`flake.nix`. VM platforms use `../hardware-configuration/vm/{proxmox,linode}.nix`;
|
||||
`baremetal.nix` uses `../hardware-configuration/baremetal.nix` (adapted
|
||||
from a real `nixos-generate-config` run on the actual hardware, not a
|
||||
vm/ file, since it isn't a VM) plus `hardware.enableRedistributableFirmware
|
||||
= true` for real wifi/GPU/microcode firmware that VMs never needed.
|
||||
`lxc.nix` has no hardware-configuration counterpart since containers
|
||||
share the host kernel; instead it imports nixpkgs' own
|
||||
- `modules/platforms/{linode,proxmox,lxc}.nix` — platform-specific config:
|
||||
boot method, guest tooling, and (for linode/proxmox) the hypervisor-specific
|
||||
hardware config, imported directly by the platform module itself
|
||||
(`../hardware-configuration/vm/{proxmox,linode}.nix`) — **not** wired in
|
||||
from `flake.nix`. `lxc.nix` has no hardware-configuration counterpart since
|
||||
containers share the host kernel; instead it imports nixpkgs' own
|
||||
`virtualisation/proxmox-lxc.nix`, which gives every `lxc-*` host a
|
||||
`config.system.build.tarball` output — a plain rootfs tarball, used as a
|
||||
`pct create ... vztmpl` CT template (**not** `pct restore`, which expects
|
||||
`vzdump` backup-archive metadata this doesn't have), no install step —
|
||||
see `docs/auto-installer.md`.
|
||||
- `modules/build-types/*.nix` — what a system is for:
|
||||
minimal/server/docker/gui/pxe-boot/nix-cache/tailscale-exit-node/tor-relay.
|
||||
minimal/server/docker/gui/pxe-boot/nix-cache.
|
||||
- `modules/common/configuration.nix` — base NixOS config imported by every
|
||||
host: locale, users, nix settings, git.
|
||||
- `modules/common/home.nix` / `hosts/nixos/home.nix` — Home Manager config for
|
||||
@@ -411,16 +181,6 @@ removing a host.
|
||||
boots, so this declares them with `destroy = false` (disko never wipes
|
||||
them) and a bare `filesystem`/`swap` content type instead of a partition
|
||||
table — idempotent against an already-provisioned disk, never destructive.
|
||||
- `modules/disko/baremetal.nix` — `baremetal-gui`'s disko config: a ZFS
|
||||
RAID0 (striped, no redundancy — disko's zpool `mode` defaults to `""`,
|
||||
which is a plain stripe rather than `"mirror"`/`"raidz"`) root pool
|
||||
across two disks, ESP + systemd-boot on the first. Device paths
|
||||
(`vars.guiRootDisk1`/`guiRootDisk2`) are placeholders — fill in stable
|
||||
`/dev/disk/by-id/...` paths before running disko for real.
|
||||
`modules/platforms/baremetal.nix` also imports
|
||||
`modules/services/zfs/enable-service.nix` for this (the `zfs_unstable`
|
||||
package, autoScrub/autoSnapshot/trim) — the only other importer today is
|
||||
`server`'s NFS data pool, an unrelated non-root ZFS use.
|
||||
- `modules/boot/efi.nix` — systemd-boot + EFI vars, paired with the disko module.
|
||||
- `modules/installer/` — the auto-installer environment (ISO, also served as
|
||||
PXE netboot): `common.nix` (shared config + the generated
|
||||
@@ -439,8 +199,7 @@ removing a host.
|
||||
and `environmentFile`; used by `hosts/server/host.nix` and
|
||||
`hosts/nix-cache/host.nix` to avoid duplicating that boilerplate.
|
||||
- `modules/tailscale/`, `modules/docker/`, `modules/networking/`,
|
||||
`modules/traefik/`, `modules/tor/`, `modules/services/*` — single-purpose,
|
||||
single-host
|
||||
`modules/traefik/`, `modules/services/*` — single-purpose, single-host
|
||||
feature modules (e.g. `docker/enable-service.nix`,
|
||||
`services/zfs/enable-service.nix`). Grep `modules/build-types/*.nix` for
|
||||
each build type's `imports` list to see which modules apply where.
|
||||
|
||||
@@ -8,15 +8,13 @@ workstation.
|
||||
Targets are named `<platform>-<buildtype>`, generated from two orthogonal
|
||||
pieces composed in `flake.nix`:
|
||||
|
||||
- **Platforms** (what it runs on): `linode`, `proxmox`, `lxc`, `baremetal`
|
||||
- **Platforms** (what it runs on): `linode`, `proxmox`, `lxc`
|
||||
- **Build types** (what it's for): `minimal`, `nix-cache`, `server`, `docker`,
|
||||
`gui`, `pxe-boot`, `tailscale-router`, `tor-relay`, `ha-server`
|
||||
`gui`, `pxe-boot`, `tailscale-exit-node`
|
||||
|
||||
Not every combination exists — `pxe-boot` has no `linode` variant, since
|
||||
PXE/DHCP/TFTP need LAN L2 adjacency that a Linode VPS doesn't have,
|
||||
`tor-relay` and `ha-server` currently only exist on `lxc`/`proxmox`, and
|
||||
`baremetal` currently only exists as `baremetal-gui` (the real gui-host
|
||||
hardware). The full list:
|
||||
PXE/DHCP/TFTP need LAN L2 adjacency that a Linode VPS doesn't have. The full
|
||||
list:
|
||||
|
||||
| Target | Purpose |
|
||||
| --- | --- |
|
||||
@@ -27,28 +25,21 @@ hardware). The full list:
|
||||
| `linode-server` / `proxmox-server` / `lxc-server` | Storage, NFS, backup, and monitoring exporter host — previously the flat `server` target |
|
||||
| `linode-docker` / `proxmox-docker` / `lxc-docker` | Docker host for the main container stack — previously the flat `docker` target |
|
||||
| `linode-gui` / `proxmox-gui` / `lxc-gui` | Cinnamon desktop workstation — previously the flat `nixos` target |
|
||||
| `baremetal-gui` | Same Cinnamon desktop workstation, on the real gui-host hardware — ZFS RAID0 root, systemd-boot |
|
||||
| `proxmox-pxe-boot` / `lxc-pxe-boot` | HTTP/iPXE boot asset host — previously the flat `pxe-boot` target |
|
||||
| `linode-tailscale-router` / `proxmox-tailscale-router` / `lxc-tailscale-router` | Tailscale subnet router + MagicDNS forwarder for the LAN |
|
||||
| `lxc-tor-relay` | Tor middle relay |
|
||||
| `proxmox-ha-server-1` / `proxmox-ha-server-2` | HA file-server cluster nodes — DRBD + XFS + iSCSI + NFS, managed by Corosync + Pacemaker |
|
||||
| `linode-tailscale-exit-node` / `proxmox-tailscale-exit-node` / `lxc-tailscale-exit-node` | Tailscale exit node |
|
||||
|
||||
Which variant of a given buildtype is actually deployed isn't tracked
|
||||
anywhere in this repo — that's live infrastructure state, not something a
|
||||
committed file can keep accurate, and it changes independently of the code.
|
||||
Check the Proxmox node itself, or `/etc/flake-target` on a running host (see
|
||||
below), if you need to know what's really out there right now.
|
||||
`scripts/proxmox/create-proxmox-resource.sh`'s duplicate-host guard works the same
|
||||
`scripts/create-proxmox-resource.sh`'s duplicate-host guard works the same
|
||||
way: it checks the Proxmox node directly rather than any file here.
|
||||
Real, production deployments live on `pve1.sweet.home`; there's a second
|
||||
node, `pve-test.sweet.home`, set aside purely for scratch/test resources —
|
||||
see `scripts/env.sh` (`PVE1_HOST` / `PVE_TEST_HOST`, and the
|
||||
`--node`/`PROXMOX_HOST` targeting they feed into) and CLAUDE.md's Proxmox
|
||||
section for which is which.
|
||||
|
||||
Each buildtype's `hosts/<name>/host.nix` carries the per-machine identity
|
||||
(hostname, hostId, per-machine secrets, `system.stateVersion`) that must stay
|
||||
fixed regardless of which platform it's built for. Every deployed host
|
||||
fixed regardless of which platform it's built for — see
|
||||
`flake-target-refactor-spec.md` for the full rationale. Every deployed host
|
||||
stamps its own active target name into `/etc/flake-target` at build time, so
|
||||
`nixos-rebuild switch --flake .#$(cat /etc/flake-target)` always picks up the
|
||||
right one even after a platform migration changes the flake attribute name.
|
||||
@@ -67,13 +58,12 @@ nix eval --json .#nixosConfigurations --apply builtins.attrNames | jq -r '.[]'
|
||||
| `variables.nix` | Single source of truth for shared values (LAN domain/CIDR, hostnames, timezone, primary username, storage root, NFS share subpaths/mountpoints, service ports, ...) — passed to every module and Home Manager config as the `vars` argument via `specialArgs`/`extraSpecialArgs` |
|
||||
| `hosts/<name>/host.nix` | Per-machine identity: hostname, hostId, per-machine secrets, `system.stateVersion` |
|
||||
| `hosts/nixos/home.nix` | Workstation-specific Home Manager config (used by the `gui` build type) |
|
||||
| `modules/platforms/` | Platform-specific config: virtualisation guest tools, boot method, hardware config (`linode.nix`, `proxmox.nix`, `lxc.nix`, `baremetal.nix`) |
|
||||
| `modules/platforms/` | Platform-specific config: virtualisation guest tools, boot method, hardware config (`linode.nix`, `proxmox.nix`, `lxc.nix`) |
|
||||
| `modules/build-types/` | Build-type-specific config: what makes a system minimal/server/docker/gui/pxe-boot/nix-cache |
|
||||
| `modules/common/` | Shared NixOS config, Home Manager, aliases imported by every host |
|
||||
| `modules/nix-cache/` | Binary cache and remote builder client/server modules |
|
||||
| `modules/installer/` | Auto-installer environment (ISO, also served as PXE netboot) — see `docs/auto-installer.md` |
|
||||
| `host-keys/` | Gitignored; only used by the auto-installer environment for pre-seeding SSH host keys before first boot — see `docs/auto-installer.md`. All deployed hosts use clan vars (`vars/per-machine/<target>/openssh/`) instead |
|
||||
| `vars/per-machine/` | Clan vars: committed, sops-encrypted SSH host keys for all deployed hosts; read by `create-proxmox-resource.sh` at deploy time |
|
||||
| `host-keys/` | Gitignored, locally-generated SSH host keys for the auto-installer — see `docs/auto-installer.md` |
|
||||
| `docs/` | Operational notes for cache, builders, lock updates, boot services, the auto-installer, and Proxmox image builds |
|
||||
| `scripts/` | Codex setup, validation, host-key, release-bump, and Proxmox resource helpers |
|
||||
|
||||
@@ -83,19 +73,10 @@ Safe validation commands for Codex and local review:
|
||||
|
||||
```bash
|
||||
bash scripts/codex-setup.sh
|
||||
bash scripts/codex-maintenance.sh dry-run
|
||||
bash scripts/codex-maintenance.sh
|
||||
```
|
||||
|
||||
`codex-maintenance.sh` with no flags (what CI runs on every push/PR) scopes
|
||||
fmt-check/statix/eval to files changed against a base ref — fast, but only
|
||||
as thorough as the diff. For the full sweep (every host, every package,
|
||||
fmt-check and statix over the whole tree — slow, CI never runs this):
|
||||
|
||||
```bash
|
||||
bash scripts/codex-maintenance.sh --full-check
|
||||
bash scripts/codex-maintenance.sh --full-check --dry-run
|
||||
```
|
||||
|
||||
For individual host evaluation:
|
||||
|
||||
```bash
|
||||
@@ -132,11 +113,10 @@ Three different paths depending on target, none of them involving a manual
|
||||
disk image and attached to a new VM with no install step — see
|
||||
`docs/proxmox-images.md`.
|
||||
|
||||
`scripts/proxmox/create-proxmox-resource.sh --type lxc|vm --host <name>` automates
|
||||
either of the last two end to end (host-key registration, building the
|
||||
image directly on the Proxmox node itself, `pct create`/`qm create`), with
|
||||
`--dry-run` and a guard against duplicating an already-deployed host's
|
||||
identity. See its `--help`.
|
||||
`scripts/create-proxmox-resource.sh --type lxc|vm --host <name>` automates
|
||||
either of the last two end to end (build, host-key registration, upload,
|
||||
`pct create`/`qm create`), with `--dry-run` and a guard against duplicating
|
||||
an already-deployed host's identity. See its `--help`.
|
||||
|
||||
## Security Notes
|
||||
|
||||
@@ -163,10 +143,9 @@ sops-nix-everywhere: it has a hardcoded login password instead (no stable
|
||||
per-boot host key for sops-nix to derive from on ephemeral media) — see
|
||||
"Host keys" in `docs/auto-installer.md` for why, and how the private keys it
|
||||
*does* pre-seed for target hosts stay out of git via the gitignored
|
||||
`host-keys/` directory. All deployed hosts use clan vars
|
||||
(`vars/per-machine/<target>/openssh/`, committed and sops-encrypted) for
|
||||
their SSH host keys.
|
||||
`host-keys/` directory.
|
||||
|
||||
This repository's git *history* still contains secrets committed before the
|
||||
sops-nix migration — those are being scrubbed and rotated separately; don't
|
||||
treat the repo as safe to make public until that's finished.
|
||||
This repository's git *history* still contains secrets committed before this
|
||||
migration (see `remove-sensetive-info-refactor.md`) — those are being
|
||||
scrubbed and rotated separately; don't treat the repo as safe to make public
|
||||
until that's finished.
|
||||
|
||||
@@ -1,25 +0,0 @@
|
||||
-----BEGIN CERTIFICATE-----
|
||||
MIIESDCCArCgAwIBAgIBATANBgkqhkiG9w0BAQsFADA1MRMwEQYDVQQKDApTV0VF
|
||||
VC5IT01FMR4wHAYDVQQDDBVDZXJ0aWZpY2F0ZSBBdXRob3JpdHkwHhcNMjYwNzI2
|
||||
MjExMzQxWhcNNDYwNzI2MjExMzQxWjA1MRMwEQYDVQQKDApTV0VFVC5IT01FMR4w
|
||||
HAYDVQQDDBVDZXJ0aWZpY2F0ZSBBdXRob3JpdHkwggGiMA0GCSqGSIb3DQEBAQUA
|
||||
A4IBjwAwggGKAoIBgQCzljYktbHdMGVJ6Wq0XQJuHLN6dkCSOgtoIzQtriPQkkNI
|
||||
uo28LwobaiQQ8sX4kGRH/BTKnH8QlId/jug4Uc+sDHnABYu++AiOhPbBX8gCpRQ0
|
||||
hebBjZiktHSBUEJR31siWOVdBoKBDJEoxehx7XUXvcxIJcaRN+LHYjO86nJN55HB
|
||||
VwFU2JcYDk98c+144dFJxXdr++MjWe4Z/oVVU8JHIOtNtKhVhvij6oOSWxcYoJO/
|
||||
S80LRj1vx/o6o/3G6bYug7PjY7JjZk/Oj61whijZkcsoO1MXSYI6UywJZGflv+ZB
|
||||
7HyufdYAsK3WhE8O2FX3/kq64Ol83HNtoR8Dt68rTg1xpW6K45jS6iDPKueYGkb0
|
||||
oSx7e++90VAW2PDhj6QQ3JJ4O5VQwrrecekJzUrAean0FOEbmgyi4PsEp1Vk6LDQ
|
||||
SsIn1x0euyxVivQMlzNX2XrZL3urn1BNPqAdntXQMkR0Wl8sbUiJPe0kxG52CGXs
|
||||
6yfNEXbPmVGcC0TBdGECAwEAAaNjMGEwHQYDVR0OBBYEFLh5QbI1UWMH0WR4z8bG
|
||||
lhrOX3X5MB8GA1UdIwQYMBaAFLh5QbI1UWMH0WR4z8bGlhrOX3X5MA8GA1UdEwEB
|
||||
/wQFMAMBAf8wDgYDVR0PAQH/BAQDAgHGMA0GCSqGSIb3DQEBCwUAA4IBgQCVodVN
|
||||
owwo53OQe02QhtEbIur2PL7zIfvhvCTRD4J8gwpbMIqT7JQK0tV6Mvsg2L8yTb2O
|
||||
KjrWeLKHGWaZZlhGSPTbkMFdb/Ls8M9FSnkc2bwcdWW3Z1lOiCjBYYqwLCG6JhvB
|
||||
5SXVwWNJwXeasL2m7oFTSwhsqPpARJ2t25u2N35o+tqIoCjijKwkmEOT66N9EAbu
|
||||
2VQjtYZWPkBtP4YCe0Ey6u4oy7sy8ThNAjOylZok+J4JW7QEFjK4Q/emhA4aQq5H
|
||||
gg9qgMuG+5oi6D1g2Wy+fMTRBaukJtLYZbBpQMQhMYWg44uPp/2bbNPTID/nV1KB
|
||||
GcPyHaskcVxPdYWxAPMwk3AeJXWyOq7atAPTF5sbk0kQQf2m+vyOqcli5CxRMUgV
|
||||
rcyi9l6+dZW4U+38Q0ET5M3OuxNI4hA7kVY2cfTakXWNqh97+TIHnstblDhAxECK
|
||||
6ZLMJQYUy7LqJTX84H27CBWLexEMjXwdr5HCV88Fj6mAK0fRufnIw5FeneA=
|
||||
-----END CERTIFICATE-----
|
||||
+17
-44
@@ -8,13 +8,10 @@ lives here.
|
||||
The installer provides a small NixOS install environment (ISO, or the same
|
||||
image netbooted via PXE) with SSH access, Git support, and an interactive
|
||||
installation script.
|
||||
Logging in as any user (root or `nixos`) runs
|
||||
`/etc/nixos-installer/installer/auto-install.sh` (the same file as
|
||||
`scripts/installer/auto-install.sh` in this repo — see "Installer process"
|
||||
below for why it's baked in at that path rather than a flat
|
||||
`/etc/auto-install.sh`), discovers available hosts from this same flake,
|
||||
lets the operator choose a target, applies that host's Disko storage
|
||||
configuration, installs NixOS, and reboots.
|
||||
Logging in as any user (root or `nixos`) runs `/etc/auto-install.sh`,
|
||||
discovers available hosts from this same flake, lets the operator choose a
|
||||
target, applies that host's Disko storage configuration, installs NixOS, and
|
||||
reboots.
|
||||
|
||||
**This applies to every `nixosConfigurations` target except `lxc-*` hosts —
|
||||
see "LXC hosts" immediately below for why those are different.**
|
||||
@@ -22,7 +19,7 @@ see "LXC hosts" immediately below for why those are different.**
|
||||
## LXC hosts
|
||||
|
||||
`lxc-*` targets (`lxc-minimal`, `lxc-nix-cache`, `lxc-server`, `lxc-docker`,
|
||||
`lxc-gui`, `lxc-pxe-boot`, `lxc-tailscale-router`, `lxc-tor-relay`) are **not** installed via `auto-install.sh` — the
|
||||
`lxc-gui`, `lxc-pxe-boot`) are **not** installed via `auto-install.sh` — the
|
||||
interactive menu deliberately excludes them. Don't try to select one there;
|
||||
`nixos-install` would bind-mount `/` onto `/mnt` (LXC containers have no raw
|
||||
disk to partition) and then refuse to touch the filesystem it's currently
|
||||
@@ -76,9 +73,9 @@ booting one:
|
||||
|
||||
First boot runs `boot.postBootCommands` (registers the Nix store DB and
|
||||
system profile) — there's no separate activation step to run yourself.
|
||||
`scripts/proxmox/create-proxmox-resource.sh --type lxc --host <name>` automates all
|
||||
of this (host-key handling, building the tarball directly on the Proxmox
|
||||
node itself, `pct create` with the flags above) — see its `--help`.
|
||||
`scripts/create-proxmox-resource.sh --type lxc --host <name>` automates all
|
||||
of this (build, host-key handling, upload, `pct create` with the flags
|
||||
above) — see its `--help`.
|
||||
|
||||
Host keys still need pre-seeding the same way as any other host — the
|
||||
sops-nix activation-vs-first-boot race is identical regardless of how the
|
||||
@@ -105,7 +102,7 @@ groups required, got 0`, and *every* secret (including this host's own
|
||||
login) permanently fails to decrypt, silently — no error in the boot log
|
||||
at all, since the activation step that would install secrets only runs on
|
||||
a from-scratch first activation and skips silently once `/run/current-system`
|
||||
already exists. `scripts/proxmox/create-proxmox-resource.sh` always builds with
|
||||
already exists. `scripts/create-proxmox-resource.sh` always builds with
|
||||
`NIXOS_HOST_KEYS_DIR` set for this reason.
|
||||
|
||||
## Layout
|
||||
@@ -119,10 +116,10 @@ already exists. `scripts/proxmox/create-proxmox-resource.sh` always builds with
|
||||
`docs/pxe-boot.md`).
|
||||
- `modules/installer/host-keys.nix` — optionally bakes pre-generated SSH
|
||||
host keys into the image; see "Host keys" below.
|
||||
- `scripts/secrets/sync-host-keys.sh` — admin-workstation tool that generates,
|
||||
- `scripts/sync-host-keys.sh` — admin-workstation tool that generates,
|
||||
registers, and (via `--remove`/`--regenerate-all-keys`) retires host
|
||||
keys; see "Creating a New Machine" below.
|
||||
- `scripts/secrets/prepare-host-key.sh` — narrower predecessor: generates a single
|
||||
- `scripts/prepare-host-key.sh` — narrower predecessor: generates a single
|
||||
key by an arbitrary name without touching `.sops.yaml`. Still useful for
|
||||
pre-generating a key *before* its flake target exists (`sync-host-keys.sh`
|
||||
can only act on targets `nixosConfigurations` already has); otherwise
|
||||
@@ -133,15 +130,13 @@ Flake outputs:
|
||||
```nix
|
||||
nixosConfigurations.installer # ISO/netboot installer image
|
||||
|
||||
packages.x86_64-linux.iso # installer ISO/netboot image
|
||||
packages.x86_64-linux.pxe # auto-installer netboot bundle (kernel + initrd + ipxe script)
|
||||
packages.x86_64-linux.pxe-minimal # vanilla NixOS minimal netboot bundle (no installer wiring)
|
||||
packages.x86_64-linux.iso # installer ISO/netboot image
|
||||
packages.x86_64-linux.pxe # netboot-ipxe + netboot-initrd + netboot-kernel, bundled
|
||||
```
|
||||
|
||||
```sh
|
||||
nix build .#iso
|
||||
nix build .#pxe
|
||||
nix build .#pxe-minimal
|
||||
```
|
||||
|
||||
There's no `nixosConfigurations.proxmox-lxc` (installer-boots-as-an-LXC-
|
||||
@@ -155,11 +150,7 @@ use case.
|
||||
|
||||
The `pxe` variant is also built automatically as part of the `pxe-boot` host
|
||||
itself (`modules/pxe-boot/stage-installer-artifacts.nix`) and served over
|
||||
iPXE as the menu's "NixOS Auto-Installer" entry — see `docs/pxe-boot.md`.
|
||||
That same host also builds and serves `packages.x86_64-linux.pxe-minimal`,
|
||||
a vanilla NixOS minimal netboot image with none of this auto-installer's
|
||||
wiring, as a separate "NixOS Minimal" menu entry — also documented in
|
||||
`docs/pxe-boot.md`, not covered further here since it's not this installer.
|
||||
iPXE — see `docs/pxe-boot.md`.
|
||||
|
||||
## Host keys
|
||||
|
||||
@@ -197,10 +188,6 @@ default.
|
||||
`auto-install.sh` still supports the older manual path as a fallback: if a
|
||||
host's key isn't baked in (`/etc/host-keys`), it checks `/root/host-keys`
|
||||
next, where you can `scp` a key in after boot, same as before this migration.
|
||||
If neither has it and the script is running interactively (an actual
|
||||
operator at the other end of stdin, not an unattended run), it prompts for
|
||||
an arbitrary directory to check (a mounted USB stick, another filesystem,
|
||||
etc.) and copies the key pair into `/root/host-keys` from there if found.
|
||||
|
||||
## Storage
|
||||
|
||||
@@ -226,21 +213,7 @@ entirely (see "LXC hosts" above), so it never reaches this code path.
|
||||
|
||||
## Installer process
|
||||
|
||||
`scripts/installer/auto-install.sh` is a real, version-controlled shell
|
||||
script — not an inline Nix string. It sources `scripts/env.sh` for
|
||||
`LAN_DOMAIN` itself (same as every other script in `scripts/`), so it
|
||||
behaves identically whether it's run straight from a git checkout (e.g.
|
||||
manually, from a stock NixOS ISO that isn't this repo's own installer
|
||||
image) or from inside the built installer image. That's also why it's
|
||||
baked in at `/etc/nixos-installer/installer/auto-install.sh` rather than a
|
||||
flat `/etc/auto-install.sh` — `modules/installer/common.nix` bakes
|
||||
`scripts/env.sh` in alongside it at `/etc/nixos-installer/env.sh`,
|
||||
preserving the same relative layout (`installer/auto-install.sh` ->
|
||||
`../env.sh`) the checked-out repo has, so the script's own
|
||||
`source ".../env.sh"` line resolves correctly in both places without any
|
||||
Nix-level templating.
|
||||
|
||||
Once running, it:
|
||||
`/etc/auto-install.sh`:
|
||||
|
||||
1. Queries `nixosConfigurations` from this flake over the network (`git+https://<lanDomain>/beatzaplenty/nixos.git`) — this happens at *install* time, not build time, so a generic installer image always sees whatever hosts are currently committed, without needing a rebuild.
|
||||
2. Presents them as a menu; confirms the choice.
|
||||
@@ -265,7 +238,7 @@ GitHub token behind sops-nix for all of them).
|
||||
2. **On your admin workstation, generate and register its host key:**
|
||||
|
||||
```sh
|
||||
./scripts/secrets/sync-host-keys.sh <flake-target>
|
||||
./scripts/sync-host-keys.sh <flake-target>
|
||||
```
|
||||
|
||||
This generates `host-keys/<flake-target>_ssh_host_ed25519_key(.pub)`,
|
||||
@@ -277,7 +250,7 @@ GitHub token behind sops-nix for all of them).
|
||||
|
||||
Doing this for every host that needs one at once — after adding several
|
||||
new targets, or just to catch up any that were missed — is
|
||||
`./scripts/secrets/sync-host-keys.sh --all`. See `scripts/secrets/sync-host-keys.sh --help`
|
||||
`./scripts/sync-host-keys.sh --all`. See `scripts/sync-host-keys.sh --help`
|
||||
for its other modes (`--remove`, `--regenerate-all-keys`).
|
||||
|
||||
3. **Commit and push.** The flake build the installer uses has to see the
|
||||
|
||||
@@ -8,14 +8,9 @@ and to verify that declared NixOS hosts still evaluate after dependency updates.
|
||||
- A scheduled workflow runs `nix flake update` once per week.
|
||||
- On GitHub, any resulting `flake.lock` change is proposed through a pull request.
|
||||
- On Gitea, the workflow can commit and push `flake.lock` directly when PR automation is not configured.
|
||||
- A separate CI workflow runs `scripts/codex-maintenance.sh` before merge.
|
||||
Its default mode scopes eval to the hosts/packages a change can affect,
|
||||
determined from a git diff against the PR base — but a `flake.lock` change
|
||||
is treated as repo-wide and always falls back to evaluating every host, so
|
||||
a lock-file update PR still gets full coverage. Hosts are still listed
|
||||
dynamically via
|
||||
`nix eval --json .#nixosConfigurations --apply builtins.attrNames` rather
|
||||
than hand-enumerated, so that fallback can't drift as `<platform>-<buildtype>`
|
||||
- A separate CI workflow evaluates every configured host before merge, listed
|
||||
dynamically via `nix eval --json .#nixosConfigurations --apply builtins.attrNames`
|
||||
rather than hand-enumerated, so it can't drift as `<platform>-<buildtype>`
|
||||
targets are added or removed. See `README.md` for the current target list.
|
||||
|
||||
## Why hosts should stop using `--upgrade-all`
|
||||
|
||||
@@ -1,128 +0,0 @@
|
||||
# IP Addressing Scheme
|
||||
|
||||
## Subnets
|
||||
|
||||
| Subnet | CIDR | Purpose | Routed? |
|
||||
|---|---|---|---|
|
||||
| LAN | `192.168.2.0/24` | General LAN — clients and infrastructure | Yes (gateway .254) |
|
||||
| Storage | `192.168.4.0/29` | HA file server DRBD replication | No — internal `vmbr1` only, no uplink |
|
||||
|
||||
The storage subnet never leaves pve1. `vmbr1` is a Proxmox Linux bridge with no physical port
|
||||
attached; traffic between the two HA file server VMs stays in-kernel.
|
||||
|
||||
The host octet is consistent across subnets for any host that has multiple interfaces — e.g.
|
||||
ha-node1 is always `.228` (LAN: `192.168.2.228`, storage: `192.168.4.228`).
|
||||
|
||||
---
|
||||
|
||||
## LAN — 192.168.2.0/24
|
||||
|
||||
### Address map
|
||||
|
||||
| Range | Purpose |
|
||||
|---|---|
|
||||
| .1–.9 | Reserved, never assign |
|
||||
| .10–.59 | Client DHCP pool (router-assigned) |
|
||||
| .60–.219 | Unallocated buffer |
|
||||
| .220–.229 | Virtual nodes (VMs / LXC containers) |
|
||||
| .230–.239 | Expansion buffer (reserved, unallocated) |
|
||||
| .240–.249 | Physical nodes (bare-metal hosts) |
|
||||
| .250–.253 | Network services |
|
||||
| .254 | Router / gateway |
|
||||
|
||||
### Network services (.250–.253)
|
||||
|
||||
| IP | Hostname | Role |
|
||||
|---|---|---|
|
||||
| `192.168.2.254` | router | Gateway (TP-Link) |
|
||||
| `192.168.2.253` | domain-controller | FreeIPA — authoritative DNS for `sweet.home`, Kerberos, LDAP |
|
||||
| `192.168.2.250`–`.252` | — | Reserved for future network services |
|
||||
|
||||
### Physical nodes (.240–.249)
|
||||
|
||||
| IP | Hostname | Role |
|
||||
|---|---|---|
|
||||
| `192.168.2.245` | pve1 | Proxmox VE hypervisor |
|
||||
| `192.168.2.244` | pbs | Proxmox Backup Server |
|
||||
| `192.168.2.243` | nixos | Bare-metal workstation (`baremetal-gui`) |
|
||||
| `192.168.2.246`–`.249` | — | Reserved — second Proxmox node and associated services |
|
||||
| `192.168.2.240`–`.242` | — | Reserved |
|
||||
|
||||
pve1 sits mid-range deliberately so a second Proxmox node can slot in on either side.
|
||||
|
||||
### Virtual nodes (.220–.229)
|
||||
|
||||
All VMs and LXC containers run on pve1.
|
||||
|
||||
| IP | Hostname | Role | Status |
|
||||
|---|---|---|---|
|
||||
| `192.168.2.229` | ha-vip | HA file server iSCSI floating VIP (Pacemaker) | Future |
|
||||
| `192.168.2.228` | ha-node1 | HA file server node 1 (DRBD + XFS + iSCSI) | Future |
|
||||
| `192.168.2.227` | ha-node2 | HA file server node 2 (DRBD + XFS + iSCSI) | Future |
|
||||
| `192.168.2.226` | server | Current NFS/ZFS file server — retires when HA is live | Retiring |
|
||||
| `192.168.2.225` | docker | Docker / Traefik stack | Active |
|
||||
| `192.168.2.224` | nix-cache | Nix binary cache + remote builder | Active |
|
||||
| `192.168.2.223` | pxe-boot | PXE / TFTP / HTTP netboot server | Active |
|
||||
| `192.168.2.222` | tailscale-router | Tailscale exit node / router | Active |
|
||||
| `192.168.2.221` | tor-relay | Tor relay | Active |
|
||||
| `192.168.2.220` | pdm | Proxmox Deploy Manager | Active |
|
||||
|
||||
### Client DHCP pool (.10–.59)
|
||||
|
||||
Assigned by the router. DNS option points to `192.168.2.253` (domain-controller).
|
||||
|
||||
Devices in this range: phones, laptops, IoT, Canon printer, any non-infrastructure host.
|
||||
No static reservations for infrastructure hosts — all infra uses static IP configuration
|
||||
on the guest itself (not DHCP reservations), so IPs survive VM recreation regardless of
|
||||
MAC address churn.
|
||||
|
||||
---
|
||||
|
||||
## Storage network — 192.168.4.0/29
|
||||
|
||||
Internal to pve1 only. Proxmox bridge `vmbr1`, no physical NIC attached.
|
||||
|
||||
| IP | Hostname | Interface role |
|
||||
|---|---|---|
|
||||
| `192.168.4.228` | ha-node1 | DRBD replication NIC |
|
||||
| `192.168.4.227` | ha-node2 | DRBD replication NIC |
|
||||
| — | no gateway | Isolated — not routed to LAN or internet |
|
||||
|
||||
---
|
||||
|
||||
## Migration reference
|
||||
|
||||
Current → target IP for every host being renumbered.
|
||||
|
||||
| Host | Current IP | New IP | Config location |
|
||||
|---|---|---|---|
|
||||
| router | `192.168.2.254` | `192.168.2.254` | unchanged |
|
||||
| domain-controller | `192.168.2.138` | `192.168.2.253` | `/etc/sysconfig/network-scripts/ifcfg-eth0` on guest |
|
||||
| pve1 | `192.168.2.250` | `192.168.2.245` | `/etc/network/interfaces` on Proxmox host |
|
||||
| pbs | `192.168.2.108` | `192.168.2.244` | static config on PBS host |
|
||||
| nixos workstation | `192.168.2.119` | `192.168.2.243` | `networking.interfaces` / NetworkManager on guest |
|
||||
| ha-node1 | — | `192.168.2.228` | future |
|
||||
| ha-node2 | — | `192.168.2.227` | future |
|
||||
| ha-vip | — | `192.168.2.229` | future (Pacemaker resource) |
|
||||
| server | `192.168.2.252` | `192.168.2.226` | static config on guest |
|
||||
| docker | `192.168.2.249` | `192.168.2.225` | static config on guest |
|
||||
| nix-cache | `192.168.2.120` | `192.168.2.224` | static config on guest |
|
||||
| pxe-boot | `192.168.2.247` | `192.168.2.223` | static config on guest; update `vars.pxeServerIp` in `variables.nix` ✓ |
|
||||
| tailscale-router | `192.168.2.121` | `192.168.2.222` | static config on guest |
|
||||
| tor-relay | `192.168.2.107` | `192.168.2.221` | static config on guest |
|
||||
| pdm | `192.168.2.248` | `192.168.2.220` | static config on guest |
|
||||
|
||||
### Cutover notes
|
||||
|
||||
- **Do domain-controller first** — it becomes the DNS server; everything else depends on it
|
||||
having its new IP and FreeIPA DNS configured before Pi-hole is retired.
|
||||
- **pve1 last among physical hosts** — changing the Proxmox management IP drops the web UI
|
||||
briefly; all guests keep running.
|
||||
- **Update Pi-hole custom.list / FreeIPA DNS A records** to new IPs before flipping any host,
|
||||
so name resolution stays valid throughout the migration.
|
||||
- **variables.nix already updated** for `pxeServerIp` (.247→.223), `pbsIp` (.108→.244), and
|
||||
new `domainControllerIp` (.253). Rebuild affected hosts after renumbering.
|
||||
- **Router DHCP**: once domain-controller is at .253 and FreeIPA DNS is serving `sweet.home`,
|
||||
switch router DHCP on with pool .10–.59 and DNS option pointing to .253; retire Pi-hole CT.
|
||||
- **Pi-hole's iPXE dnsmasq config** (`99-ipxe-chainload.conf`) moves to the pxe-boot CT as a
|
||||
dnsmasq proxy-mode config before Pi-hole is decommissioned.
|
||||
@@ -1,366 +0,0 @@
|
||||
# Network Cutover Plan
|
||||
|
||||
Moves the LAN from the current flat/Pi-hole-managed state to the new IP scheme
|
||||
defined in `docs/ip-addressing.md`. Works in five independent stages — each
|
||||
stage is safe to pause after and resume later. Rollback steps are given at
|
||||
every point where something can break.
|
||||
|
||||
**Before starting anything:** confirm you have
|
||||
- SSH access to `192.168.2.138` (domain-controller, current IP)
|
||||
- SSH access to `192.168.2.250` (pve1)
|
||||
- Browser access to Pi-hole admin at `http://192.168.2.253`
|
||||
- Browser access to router admin at `http://192.168.2.254`
|
||||
- The FreeIPA `admin` password to hand
|
||||
|
||||
---
|
||||
|
||||
## Stage 1 — Prepare FreeIPA DNS (zero downtime)
|
||||
|
||||
Everything here is additive. Pi-hole keeps running. Nothing breaks if you stop
|
||||
mid-stage.
|
||||
|
||||
### 1a. Add NextDNS forwarders
|
||||
|
||||
```bash
|
||||
ssh wayne@192.168.2.138
|
||||
kinit admin # enter FreeIPA admin password when prompted
|
||||
ipa dnsconfig-mod \
|
||||
--forwarder=45.90.28.142 \
|
||||
--forwarder=45.90.30.142 \
|
||||
--forward-policy=only
|
||||
```
|
||||
|
||||
**Verify external resolution works through FreeIPA before continuing:**
|
||||
```bash
|
||||
dig @127.0.0.1 google.com +short # must return an IP, not SERVFAIL
|
||||
```
|
||||
|
||||
### 1b. Add A records for every host at their CURRENT IPs
|
||||
|
||||
These represent the live state now. You'll update each record to the new IP
|
||||
when you renumber that host in Stage 5.
|
||||
|
||||
```bash
|
||||
ipa dnsrecord-add sweet.home pve1 --a-rec 192.168.2.250
|
||||
ipa dnsrecord-add sweet.home pbs --a-rec 192.168.2.108
|
||||
ipa dnsrecord-add sweet.home nixos --a-rec 192.168.2.119
|
||||
ipa dnsrecord-add sweet.home server --a-rec 192.168.2.252
|
||||
ipa dnsrecord-add sweet.home docker --a-rec 192.168.2.249
|
||||
ipa dnsrecord-add sweet.home nix-cache --a-rec 192.168.2.120
|
||||
ipa dnsrecord-add sweet.home pxe-boot --a-rec 192.168.2.247
|
||||
ipa dnsrecord-add sweet.home tailscale-router --a-rec 192.168.2.121
|
||||
ipa dnsrecord-add sweet.home tor-relay --a-rec 192.168.2.107
|
||||
ipa dnsrecord-add sweet.home pdm --a-rec 192.168.2.248
|
||||
ipa dnsrecord-add sweet.home router --a-rec 192.168.2.254
|
||||
```
|
||||
|
||||
### 1c. Clean up stale reverse-zone PTR records
|
||||
|
||||
FreeIPA already has PTR records from an earlier import but some are wrong.
|
||||
Fix them now so reverse DNS is accurate from day one.
|
||||
|
||||
```bash
|
||||
# Remove stale "win11" entry at .250 (should be pve1)
|
||||
ipa dnsrecord-del 2.168.192.in-addr.arpa 250 --ptr-rec win11.
|
||||
ipa dnsrecord-add 2.168.192.in-addr.arpa 250 --ptr-rec pve1.sweet.home.
|
||||
|
||||
# Fix unqualified PTR records (missing .sweet.home. suffix)
|
||||
ipa dnsrecord-mod 2.168.192.in-addr.arpa 108 --ptr-rec pbs.sweet.home.
|
||||
ipa dnsrecord-mod 2.168.192.in-addr.arpa 248 --ptr-rec pdm.sweet.home.
|
||||
ipa dnsrecord-mod 2.168.192.in-addr.arpa 249 --ptr-rec docker.sweet.home.
|
||||
ipa dnsrecord-mod 2.168.192.in-addr.arpa 252 --ptr-rec server.sweet.home.
|
||||
|
||||
# Add any missing PTR records
|
||||
ipa dnsrecord-add 2.168.192.in-addr.arpa 119 --ptr-rec nixos.sweet.home.
|
||||
ipa dnsrecord-add 2.168.192.in-addr.arpa 120 --ptr-rec nix-cache.sweet.home.
|
||||
ipa dnsrecord-add 2.168.192.in-addr.arpa 121 --ptr-rec tailscale-router.sweet.home.
|
||||
ipa dnsrecord-add 2.168.192.in-addr.arpa 247 --ptr-rec pxe-boot.sweet.home.
|
||||
ipa dnsrecord-add 2.168.192.in-addr.arpa 254 --ptr-rec router.sweet.home.
|
||||
```
|
||||
|
||||
### 1d. Point domain-controller's own DNS at itself
|
||||
|
||||
```bash
|
||||
sudo nmcli connection modify "System eth0" ipv4.dns "127.0.0.1"
|
||||
sudo nmcli connection up "System eth0"
|
||||
```
|
||||
|
||||
**Verify:**
|
||||
```bash
|
||||
dig pve1.sweet.home +short # must return 192.168.2.250
|
||||
dig google.com +short # must return an IP (NextDNS forwarding)
|
||||
```
|
||||
|
||||
**Rollback 1d:** `sudo nmcli connection modify "System eth0" ipv4.dns "192.168.2.253" && sudo nmcli connection up "System eth0"`
|
||||
|
||||
---
|
||||
|
||||
## Stage 2 — Move pxe-boot DHCP options off Pi-hole (zero downtime)
|
||||
|
||||
Pi-hole's dnsmasq currently serves the iPXE boot options via
|
||||
`99-ipxe-chainload.conf`. Before Pi-hole is retired, that config must move to
|
||||
the pxe-boot CT running dnsmasq in proxy mode so PXE boot keeps working.
|
||||
|
||||
### 2a. Add dnsmasq proxy config to the pxe-boot NixOS module
|
||||
|
||||
In `modules/build-types/pxe-boot.nix`, add:
|
||||
|
||||
```nix
|
||||
services.dnsmasq = {
|
||||
enable = true;
|
||||
settings = {
|
||||
# Proxy mode: respond only to PXE DHCP requests, leave normal leases to router
|
||||
dhcp-range = [ "192.168.2.0,proxy" ];
|
||||
# iPXE client detection
|
||||
dhcp-match = [
|
||||
"set:ipxe,175"
|
||||
"set:efi64,option:client-arch,7"
|
||||
"set:efi64,option:client-arch,9"
|
||||
];
|
||||
dhcp-userclass = "set:ipxe,iPXE";
|
||||
# Boot file selection
|
||||
dhcp-boot = [
|
||||
"tag:ipxe,tag:efi64,http://${vars.pxeServerIp}/boot.ipxe"
|
||||
"tag:ipxe,http://${vars.pxeServerIp}/boot.ipxe"
|
||||
"tag:efi64,ipxe.efi,,${vars.pxeServerIp}"
|
||||
"undionly.kpxe,,${vars.pxeServerIp}"
|
||||
];
|
||||
};
|
||||
};
|
||||
```
|
||||
|
||||
### 2b. Rebuild and deploy the pxe-boot CT
|
||||
|
||||
```bash
|
||||
# On pve1 — build the new tarball
|
||||
nix build .#lxc-pxe-boot.config.system.build.tarball
|
||||
|
||||
# Verify dnsmasq starts correctly in the CT after deploy
|
||||
ssh nixos@192.168.2.247 systemctl status dnsmasq
|
||||
```
|
||||
|
||||
### 2c. Remove the iPXE config from Pi-hole
|
||||
|
||||
In the Pi-hole CT, remove `/etc/dnsmasq.d/99-ipxe-chainload.conf` and
|
||||
restart the FTL service:
|
||||
|
||||
```bash
|
||||
ssh wayne@pve1.sweet.home \
|
||||
"sudo pct exec 100 -- bash -c 'rm /etc/dnsmasq.d/99-ipxe-chainload.conf && systemctl restart pihole-FTL'"
|
||||
```
|
||||
|
||||
**Verify:** PXE boot a test machine — it should still get an iPXE response and
|
||||
reach the boot menu.
|
||||
|
||||
**Rollback 2c:** restore the file from the Pi-hole config backup at
|
||||
`/etc/pihole/config_backups/` and restart pihole-FTL.
|
||||
|
||||
---
|
||||
|
||||
## Stage 3 — DHCP migration: Pi-hole → router (brief maintenance window)
|
||||
|
||||
**Do this in the evening.** Existing DHCP leases stay valid during the
|
||||
switchover so connected devices don't drop — only new lease requests fail
|
||||
during the gap, which is under 60 seconds if you follow the steps in order.
|
||||
|
||||
The key: configure the router's DHCP DNS option to point at `.253` (Pi-hole's
|
||||
current IP). This way, all new leases issued by the router still get the same
|
||||
DNS server address — clients never need to change their DNS config. When Pi-hole
|
||||
is retired and the DC takes `.253` in Stage 4, `.253` just starts answering
|
||||
differently. No client reconfiguration.
|
||||
|
||||
### 3a. Pre-configure router DHCP (do not enable yet)
|
||||
|
||||
Log into `http://192.168.2.254`, find the DHCP settings and fill in — but
|
||||
leave DHCP **disabled** until step 3b:
|
||||
|
||||
| Setting | Value |
|
||||
|---|---|
|
||||
| Start IP | 192.168.2.10 |
|
||||
| End IP | 192.168.2.59 |
|
||||
| Subnet mask | 255.255.255.0 |
|
||||
| Gateway | 192.168.2.254 |
|
||||
| Primary DNS | 192.168.2.253 |
|
||||
| Secondary DNS | *(leave blank)* |
|
||||
| Lease time | 24h |
|
||||
|
||||
Save without enabling.
|
||||
|
||||
### 3b. Switchover (do steps in quick succession)
|
||||
|
||||
1. **Disable Pi-hole DHCP:** Pi-hole admin UI → Settings → DHCP → uncheck
|
||||
"DHCP server enabled" → Save
|
||||
2. **Enable router DHCP** immediately after step 1
|
||||
|
||||
### 3c. Verify router DHCP is working
|
||||
|
||||
On a phone or laptop, disconnect from WiFi and reconnect (or run
|
||||
`sudo dhclient -r && sudo dhclient` on a Linux host):
|
||||
|
||||
```bash
|
||||
ip addr show # IP should be in 192.168.2.10–59 range
|
||||
dig google.com # should resolve (Pi-hole DNS still running at .253)
|
||||
dig pve1.sweet.home # should resolve via FreeIPA at .138 (relayed via Pi-hole)
|
||||
```
|
||||
|
||||
Wait 10–15 minutes for the most active devices to renew their leases. There's
|
||||
no need to wait for all leases to expire before proceeding.
|
||||
|
||||
**Rollback 3b:** Re-enable Pi-hole DHCP. Disable router DHCP. Done — existing
|
||||
leases remain valid so most devices are unaffected.
|
||||
|
||||
---
|
||||
|
||||
## Stage 4 — Move domain-controller from .138 to .253
|
||||
|
||||
Pi-hole lives at `.253`. The DC must take `.253` the moment Pi-hole stops so
|
||||
clients that still have `.253` as their DNS server don't notice the change.
|
||||
Script these commands in advance and run them in rapid succession.
|
||||
|
||||
**Pre-stage: have this SSH command ready before running step 4a:**
|
||||
```bash
|
||||
ssh wayne@192.168.2.138 "
|
||||
sudo nmcli connection modify 'System eth0' \
|
||||
ipv4.addresses '192.168.2.253/24' \
|
||||
ipv4.gateway '192.168.2.254' \
|
||||
ipv4.dns '127.0.0.1' \
|
||||
ipv4.method manual && \
|
||||
sudo nmcli connection up 'System eth0'
|
||||
"
|
||||
```
|
||||
|
||||
**Also update the Proxmox VM config to match (run from pve1):**
|
||||
```bash
|
||||
sudo qm set 108 \
|
||||
--ipconfig0 ip=192.168.2.253/24,gw=192.168.2.254 \
|
||||
--nameserver 192.168.2.253
|
||||
```
|
||||
|
||||
### 4a. Stop Pi-hole
|
||||
|
||||
```bash
|
||||
ssh wayne@pve1.sweet.home "sudo pct stop 100"
|
||||
```
|
||||
|
||||
### 4b. Immediately: change DC's IP to .253
|
||||
|
||||
Run the pre-staged SSH command from above. You have ~30 seconds before any
|
||||
client notices Pi-hole is gone. If SSH to `.138` refuses (the IP is already
|
||||
changing), open a Proxmox console to VM 108 and run the `nmcli` commands
|
||||
there.
|
||||
|
||||
### 4c. Update Proxmox VM config
|
||||
|
||||
Run the pre-staged `qm set 108` command from above.
|
||||
|
||||
### 4d. Verify
|
||||
|
||||
```bash
|
||||
ssh wayne@192.168.2.253 # must connect (new DC IP)
|
||||
dig @192.168.2.253 pve1.sweet.home +short # must return 192.168.2.250
|
||||
dig @192.168.2.253 google.com +short # must return an IP
|
||||
```
|
||||
|
||||
From a client device that renewed its DHCP lease in Stage 3:
|
||||
```bash
|
||||
cat /etc/resolv.conf # should show 192.168.2.253
|
||||
dig pve1.sweet.home # should resolve
|
||||
```
|
||||
|
||||
**Rollback 4:** `ssh wayne@pve1.sweet.home "sudo pct start 100"`. Change DC IP
|
||||
back to .138 via Proxmox console. This restores full Pi-hole DNS/DHCP service.
|
||||
Leave Pi-hole CT stopped-but-intact for 48 hours before deleting it.
|
||||
|
||||
---
|
||||
|
||||
## Stage 5 — Host renumbering (one at a time, any order)
|
||||
|
||||
For each host:
|
||||
1. Update FreeIPA DNS A record and PTR record to the new IP
|
||||
2. Change the static IP on the host itself
|
||||
3. Verify SSH to new IP
|
||||
4. Update `variables.nix` if that host has an IP variable (pxe-boot, pbs — already done in this PR)
|
||||
|
||||
**FreeIPA record update template** (run as admin on domain-controller):
|
||||
```bash
|
||||
ipa dnsrecord-mod sweet.home <hostname> --a-rec <new-ip>
|
||||
ipa dnsrecord-del 2.168.192.in-addr.arpa <old-last-octet> --ptr-rec <hostname>.sweet.home.
|
||||
ipa dnsrecord-add 2.168.192.in-addr.arpa <new-last-octet> --ptr-rec <hostname>.sweet.home.
|
||||
```
|
||||
|
||||
### Renumbering order
|
||||
|
||||
| # | Host | Old IP | New IP | How to change IP |
|
||||
|---|---|---|---|---|
|
||||
| 1 | nixos workstation | .119 | .243 | NetworkManager on guest; or `nmcli connection modify` |
|
||||
| 2 | nix-cache | .120 | .224 | `pct set 102 --net0 name=eth0,bridge=vmbr0,ip=192.168.2.224/24,gw=192.168.2.254` then `pct reboot 102` |
|
||||
| 3 | tailscale-router | .121 | .222 | Static config on guest; check Tailscale ACLs if IP is referenced there |
|
||||
| 4 | tor-relay | .107 | .221 | `pct set 104 --net0 name=eth0,bridge=vmbr0,ip=192.168.2.221/24,gw=192.168.2.254` then `pct reboot 104` |
|
||||
| 5 | pdm | .248 | .220 | `pct set 106 --net0 name=eth0,bridge=vmbr0,ip=192.168.2.220/24,gw=192.168.2.254` then `pct reboot 106` |
|
||||
| 6 | pxe-boot | .247 | .223 | `pct set 103 --net0 name=eth0,bridge=vmbr0,ip=192.168.2.223/24,gw=192.168.2.254` then rebuild NixOS (already updated in variables.nix) |
|
||||
| 7 | server | .252 | .226 | Static config on guest; NFS clients (docker) lose mounts briefly — they remount automatically |
|
||||
| 8 | docker | .249 | .225 | Static config on guest; do this after server is at .226 |
|
||||
| 9 | pbs | .108 | .244 | Static config on PBS host itself; update in `pbsIp` already done in variables.nix |
|
||||
| 10 | pve1 | .250 | .245 | Edit `/etc/network/interfaces` on the Proxmox host — see below |
|
||||
|
||||
### pve1 renumber (step 10 — do last)
|
||||
|
||||
All guests keep running; only the Proxmox web UI is briefly unreachable.
|
||||
|
||||
```bash
|
||||
ssh wayne@pve1.sweet.home
|
||||
|
||||
# Edit /etc/network/interfaces: change address from .250 to .245
|
||||
sudo nano /etc/network/interfaces
|
||||
# Change: address 192.168.2.250/24
|
||||
# To: address 192.168.2.245/24
|
||||
|
||||
sudo systemctl restart networking
|
||||
# SSH will drop here — reconnect to new IP
|
||||
```
|
||||
|
||||
```bash
|
||||
ssh wayne@192.168.2.245 # verify
|
||||
```
|
||||
|
||||
Update FreeIPA DNS:
|
||||
```bash
|
||||
ipa dnsrecord-mod sweet.home pve1 --a-rec 192.168.2.245
|
||||
ipa dnsrecord-del 2.168.192.in-addr.arpa 250 --ptr-rec pve1.sweet.home.
|
||||
ipa dnsrecord-add 2.168.192.in-addr.arpa 245 --ptr-rec pve1.sweet.home.
|
||||
```
|
||||
|
||||
**Rollback any step 5 host:** change the IP back on the guest and update the
|
||||
FreeIPA record back to the old IP. The old IP is unoccupied so you can
|
||||
temporarily use either.
|
||||
|
||||
---
|
||||
|
||||
## Stage 6 — Final cleanup
|
||||
|
||||
Once all hosts are at their new IPs and verified:
|
||||
|
||||
```bash
|
||||
# Delete the Pi-hole CT (already stopped since Stage 4)
|
||||
ssh wayne@pve1.sweet.home "sudo pct destroy 100"
|
||||
|
||||
# Remove stale FreeIPA records for retired addresses
|
||||
ipa dnsrecord-del sweet.home pihole --del-all
|
||||
ipa dnsrecord-del 2.168.192.in-addr.arpa 253 --ptr-rec pihole.sweet.home.
|
||||
|
||||
# Rebuild any NixOS hosts that reference pbsIp or pxeServerIp to pick up
|
||||
# the updated variables.nix values (pxe-boot mandatory; others as convenient)
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Rollback summary
|
||||
|
||||
| What broke | How to roll back |
|
||||
|---|---|
|
||||
| FreeIPA DNS not resolving | Check `systemctl status named` on DC; restart if failed |
|
||||
| FreeIPA DNS unreachable | `pct start 100` on pve1 (restores Pi-hole) |
|
||||
| Router DHCP not handing out leases | Re-enable Pi-hole DHCP; disable router DHCP |
|
||||
| DC unreachable after IP change | Proxmox console on VM 108 → `nmcli connection up "System eth0"` with old IP |
|
||||
| Host unreachable after renumber | Proxmox console → revert IP; or `pct set <id> --net0 ...` old IP and reboot CT |
|
||||
| pve1 web UI gone after renumber | SSH to .245 and check `/etc/network/interfaces`; if wrong, fix and restart networking |
|
||||
+9
-17
@@ -46,26 +46,18 @@ the new key up automatically on next activation — no more manual
|
||||
|
||||
## Remote builder SSH keys
|
||||
|
||||
Each client authenticates as `nixremote` using its **own default root SSH
|
||||
identity** (`/root/.ssh/id_ed25519`) — not a separately-named or shared
|
||||
keypair. If a client doesn't have one yet:
|
||||
On each client, install the private key used to authenticate as `nixremote`:
|
||||
|
||||
```bash
|
||||
sudo ssh-keygen -t ed25519 -N '' -f /root/.ssh/id_ed25519
|
||||
sudo install -d -m 0700 /root/.ssh
|
||||
sudo install -m 0600 ./nixremote /root/.ssh/nixremote
|
||||
sudo ssh -i /root/.ssh/nixremote nixremote@nix-cache nix-store --version
|
||||
```
|
||||
|
||||
Then add its `.pub` contents as a new entry in `vars.remoteBuilderAuthorizedKeys`
|
||||
(`variables.nix`) and rebuild `nix-cache` to pick it up (that list is
|
||||
declarative — an imperative `ssh-copy-id nixremote@nix-cache` won't stick;
|
||||
it gets overwritten on every rebuild). Verify with:
|
||||
On `nix-cache`, install the matching public key used by `nixremote` authorized keys.
|
||||
|
||||
```bash
|
||||
sudo ssh -i /root/.ssh/id_ed25519 nixremote@nix-cache nix-store --version
|
||||
```
|
||||
|
||||
The committed `remoteBuilderAuthorizedKeys` entries are public SSH keys
|
||||
only. Keep the matching private keys on client hosts and out of the
|
||||
repository.
|
||||
The committed `nixremote` authorized keys are public SSH keys only. Keep the
|
||||
matching private keys on client hosts and out of the repository.
|
||||
|
||||
nix-cache's own SSH *host* key is trusted declaratively via
|
||||
`programs.ssh.knownHosts` in `modules/nix-cache/remote-builder-client.nix`,
|
||||
@@ -84,8 +76,8 @@ After deployment:
|
||||
curl http://nix-cache/nix-cache-info
|
||||
nix store ping --store http://nix-cache
|
||||
nix show-config | grep -E 'substituters|trusted-public-keys|builders-use-substitutes'
|
||||
sudo ssh -i /root/.ssh/id_ed25519 nixremote@nix-cache nix-store --version
|
||||
nix build nixpkgs#hello --builders 'ssh://nixremote@nix-cache x86_64-linux /root/.ssh/id_ed25519 4 2 big-parallel,kvm,nixos-test,benchmark' -L
|
||||
sudo ssh -i /root/.ssh/nixremote nixremote@nix-cache nix-store --version
|
||||
nix build nixpkgs#hello --builders 'ssh://nixremote@nix-cache x86_64-linux /root/.ssh/nixremote 4 2 big-parallel,kvm,nixos-test,benchmark' -L
|
||||
nix path-info -r nixpkgs#hello
|
||||
curl -I "http://nix-cache/$(basename "$(nix path-info nixpkgs#hello)").narinfo"
|
||||
```
|
||||
|
||||
@@ -7,13 +7,11 @@ config (`modules/disko/proxmox.nix`) already used to format a real disk on
|
||||
install, so there's nothing host-specific to write; it's available for every
|
||||
`proxmox-*` target automatically.
|
||||
|
||||
`scripts/proxmox/create-proxmox-resource.sh --type vm --host <name>` automates the
|
||||
`scripts/create-proxmox-resource.sh --type vm --host <name>` automates the
|
||||
whole walkthrough below (and the equivalent LXC one) end to end, including
|
||||
host-key handling and building the image directly on the Proxmox node
|
||||
itself (no local build, no image transfer) — see its `--help`. The steps
|
||||
here are what it runs under the hood, useful for doing any of it by hand
|
||||
or understanding what it does before you trust it against real
|
||||
infrastructure.
|
||||
host-key handling and upload — see its `--help`. The steps here are what it
|
||||
runs under the hood, useful for doing any of it by hand or understanding
|
||||
what it does before you trust it against real infrastructure.
|
||||
|
||||
## Building
|
||||
|
||||
@@ -51,7 +49,7 @@ sudo ./result \
|
||||
--build-memory 2048
|
||||
```
|
||||
|
||||
Generate the key first with `scripts/secrets/sync-host-keys.sh <hostname>`, same
|
||||
Generate the key first with `scripts/sync-host-keys.sh <hostname>`, same
|
||||
as any other host — see `docs/auto-installer.md` for the full walkthrough
|
||||
(it registers the new key in `.sops.yaml` and re-encrypts the affected
|
||||
`secrets/*.yaml` files too, no manual editing needed).
|
||||
|
||||
+22
-131
@@ -1,10 +1,9 @@
|
||||
# pxe-boot
|
||||
|
||||
The `pxe-boot` host serves HTTP boot assets for iPXE clients — including
|
||||
self-staged copies of both this flake's own auto-installer netboot image
|
||||
(see `docs/auto-installer.md` for what that image actually is and does once
|
||||
booted) and a vanilla, unmodified NixOS minimal netboot image for plain
|
||||
rescue/inspection use.
|
||||
The `pxe-boot` host serves HTTP boot assets for iPXE clients — including a
|
||||
self-staged copy of this flake's own auto-installer netboot image, see
|
||||
`docs/auto-installer.md` for what that image actually is and does once
|
||||
booted.
|
||||
|
||||
## Host Role
|
||||
|
||||
@@ -15,7 +14,6 @@ rescue/inspection use.
|
||||
- TFTP root for first-stage bootloaders: `/srv/pxe/tftp`
|
||||
- iPXE entry script: `/srv/pxe/http/boot.ipxe`
|
||||
- Generated iPXE menu: `/srv/pxe/http/menu.ipxe`
|
||||
- Debian Minimal iPXE script: `/srv/pxe/http/debian.ipxe`
|
||||
- SystemRescue iPXE script: `/srv/pxe/http/systemrescue.ipxe`
|
||||
- TFTP fallback script: `/srv/pxe/tftp/autoexec.ipxe`
|
||||
- Boot binaries copied from the Nix `ipxe` package:
|
||||
@@ -29,140 +27,43 @@ The host creates these directories with systemd tmpfiles:
|
||||
```text
|
||||
/srv/pxe
|
||||
/srv/pxe/http
|
||||
/srv/pxe/http/images -> /mnt/pxe-images (symlink to NFS share)
|
||||
/srv/pxe/http/auto-installer
|
||||
/srv/pxe/http/nixos-minimal
|
||||
/srv/pxe/http/debian
|
||||
/srv/pxe/http/images
|
||||
/srv/pxe/http/nixos
|
||||
/srv/pxe/http/systemrescue
|
||||
/srv/pxe/http/ubuntu
|
||||
/srv/pxe/http/rescue
|
||||
/srv/pxe/tftp
|
||||
```
|
||||
|
||||
`/srv/pxe/http/images` is a symlink to `/mnt/pxe-images`, which is an NFS
|
||||
mount of `server.sweet.home:/tank/pxe-boot/images`
|
||||
(`modules/pxe-boot/mount-pxe-images.nix`). Place large images there (ISOs,
|
||||
disk images) rather than on the pxe-boot host's own root disk. For an LXC
|
||||
pxe-boot container the mount uses NFSv3+nolock with `nofail` (eager,
|
||||
non-blocking on server unavailability); for a Proxmox VM it uses NFSv4.2
|
||||
with `x-systemd.automount` (lazy, triggered on first access).
|
||||
|
||||
When running as `lxc-pxe-boot`, the Proxmox container must have
|
||||
`features: nesting=1,mount=nfs` (at minimum) in its Proxmox config. `nesting=1`
|
||||
is required by systemd 260+ for credential isolation (user namespace creation
|
||||
and internal move-mounts); without it, AppArmor denies both, and every
|
||||
systemd service that uses `PrivateUsers`, `PrivateDevices`, or credential
|
||||
passing fails on boot. `mount=nfs` allows the NFSv3 mount. Both are set
|
||||
automatically by `scripts/proxmox/create-proxmox-resource.sh` (via
|
||||
`PROXMOX_DEFAULT_LXC_FEATURES` in `scripts/env.sh` which defaults to
|
||||
`nesting=1,keyctl=1,mount=nfs;nfs4`). If you ever change these features
|
||||
manually via `pct set`, be sure to include both — `pct set` replaces the
|
||||
entire features string, it does not append to it.
|
||||
Mount shared image storage under `/srv/pxe/http`, preferably
|
||||
`/srv/pxe/http/images` unless a menu entry expects files in a specific
|
||||
directory such as `/srv/pxe/http/nixos`.
|
||||
|
||||
The HTTP iPXE chain is:
|
||||
|
||||
```text
|
||||
undionly.kpxe or ipxe.efi
|
||||
-> autoexec.ipxe from the TFTP root, when iPXE requests it
|
||||
-> http://192.168.2.223/boot.ipxe
|
||||
-> http://192.168.2.223/menu.ipxe
|
||||
-> http://192.168.2.247/boot.ipxe
|
||||
-> http://192.168.2.247/menu.ipxe
|
||||
```
|
||||
|
||||
The generated menu currently exposes entries for:
|
||||
|
||||
- NixOS Auto-Installer
|
||||
- NixOS Minimal
|
||||
- Debian Minimal
|
||||
- FreeIPA Server (Rocky Linux 9)
|
||||
- NixOS installer
|
||||
- SystemRescue environment
|
||||
- iPXE shell
|
||||
- Reboot
|
||||
|
||||
Both NixOS entries chain-load a `netboot.ipxe` staged into their own
|
||||
directory (`/srv/pxe/http/auto-installer/netboot.ipxe` and
|
||||
`/srv/pxe/http/nixos-minimal/netboot.ipxe`), each nixpkgs' own generated
|
||||
netboot iPXE script (correct `init=`/`initrd=` kernel parameters included)
|
||||
rather than a hand-rolled boot line — that script in turn expects its
|
||||
kernel/initrd siblings in the same directory. Each directory's three files
|
||||
(`bzImage`, `initrd`, `netboot.ipxe`) are built from source and staged
|
||||
automatically by `modules/pxe-boot/stage-installer-artifacts.nix` via
|
||||
`systemd.tmpfiles.rules` — no manual operator step required:
|
||||
|
||||
- `auto-installer` is this flake's own `netbootSystem` (`flake.nix`) — the
|
||||
same auto-installer image `nix build .#pxe` produces. See
|
||||
`docs/auto-installer.md`.
|
||||
- `nixos-minimal` is `netbootMinimalSystem` (`flake.nix`) — nixpkgs'
|
||||
`netboot-minimal.nix` composed on its own, with none of this flake's
|
||||
auto-installer wiring (no `common.nix`, no `auto-install.sh`, no baked
|
||||
host keys or custom users). Same `nix build .#pxe-minimal` mechanism as
|
||||
the auto-installer image, just a different module composition. Useful
|
||||
as a plain rescue/inspection shell that doesn't assume anything about
|
||||
this flake.
|
||||
|
||||
Both images set `networking.hostName` to match their menu entry/staged
|
||||
directory name (`auto-installer` / `nixos-minimal`), so each one's
|
||||
generated system name (`nixos-system-<name>-*`) is self-describing rather
|
||||
than the nixpkgs default of `nixos-system-nixos-*` for both.
|
||||
|
||||
The Debian Minimal entry chains `http://<pxeServerIp>/debian.ipxe`, which loads
|
||||
the Debian bookworm netboot kernel and initrd from `/srv/pxe/http/debian/`. The
|
||||
`fetch-debian-netboot.service` oneshot downloads these files from
|
||||
`deb.debian.org` on first boot (idempotent — skips if files are already
|
||||
present):
|
||||
|
||||
```text
|
||||
/srv/pxe/http/debian/linux (Debian bookworm netboot kernel)
|
||||
/srv/pxe/http/debian/initrd.gz (Debian bookworm netboot initrd)
|
||||
```
|
||||
|
||||
The service requires outbound internet access on the pxe-boot host. To
|
||||
re-download (e.g. after a Debian point release), delete the files and restart
|
||||
the service:
|
||||
|
||||
```bash
|
||||
rm /srv/pxe/http/debian/linux /srv/pxe/http/debian/initrd.gz
|
||||
systemctl restart fetch-debian-netboot.service
|
||||
```
|
||||
|
||||
To update to a different Debian release, change `debianRelease` in
|
||||
`modules/build-types/pxe-boot.nix` and redeploy.
|
||||
|
||||
The **FreeIPA Server (Rocky Linux 9)** entry chains
|
||||
`http://<pxeServerIp>/rocky-freeipa.ipxe`, which boots the Rocky Linux 9
|
||||
Anaconda installer with a Kickstart file (`rocky-freeipa.ks`) hosted on the
|
||||
same server. The `fetch-rocky-pxeboot.service` oneshot downloads the pxeboot
|
||||
kernel and initrd from the Rocky Linux mirror on first boot (idempotent):
|
||||
|
||||
```text
|
||||
/srv/pxe/http/rocky/vmlinuz (Rocky Linux 9 Anaconda pxeboot kernel)
|
||||
/srv/pxe/http/rocky/initrd.img (Rocky Linux 9 Anaconda pxeboot initrd)
|
||||
```
|
||||
|
||||
The Kickstart file is generated from the NixOS module and staged at
|
||||
`/srv/pxe/http/rocky-freeipa.ks`. It performs a fully unattended install:
|
||||
|
||||
1. Installs Rocky Linux 9 with `ipa-server` + `ipa-server-dns` packages
|
||||
2. Configures static IP `192.168.2.138`, hostname `domain-controller.sweet.home`
|
||||
3. Creates user `wayne` with the `adminSshKey` from `variables.nix`
|
||||
4. Generates random IPA passwords and writes them to `/root/ipa-credentials.txt`
|
||||
5. Creates a `freeipa-first-boot.service` oneshot that runs `ipa-server-install`
|
||||
on first reboot (~20 minutes)
|
||||
|
||||
After the install completes:
|
||||
- SSH in as `wayne@domain-controller` using the admin key
|
||||
- Monitor FreeIPA install progress: `sudo tail -f /root/freeipa-install.log`
|
||||
- Retrieve credentials: `sudo cat /root/ipa-credentials.txt` (save to password manager)
|
||||
- Configure Pi-hole: `server=/sweet.home/192.168.2.138` in dnsmasq
|
||||
|
||||
To refresh the pxeboot files (e.g. after a Rocky point release):
|
||||
|
||||
```bash
|
||||
rm /srv/pxe/http/rocky/vmlinuz /srv/pxe/http/rocky/initrd.img
|
||||
systemctl restart fetch-rocky-pxeboot.service
|
||||
```
|
||||
|
||||
To update to a different Rocky release, change `rockyRelease` in
|
||||
`modules/build-types/pxe-boot.nix` and redeploy.
|
||||
The NixOS installer entry chain-loads `/srv/pxe/http/nixos/netboot.ipxe`,
|
||||
which is nixpkgs' own generated netboot iPXE script (correct `init=`/`initrd=`
|
||||
kernel parameters included) rather than a hand-rolled boot line — that script
|
||||
in turn expects its kernel/initrd siblings in the same directory. All three
|
||||
files (`bzImage`, `initrd`, `netboot.ipxe`) are built from this flake's own
|
||||
`modules/installer/iso.nix` netboot image (the same one `nix build .#pxe`
|
||||
produces) and staged automatically by
|
||||
`modules/pxe-boot/stage-installer-artifacts.nix` via `systemd.tmpfiles.rules`
|
||||
— no manual operator step required.
|
||||
|
||||
The SystemRescue entry expects the source ISO at:
|
||||
|
||||
@@ -170,16 +71,13 @@ The SystemRescue entry expects the source ISO at:
|
||||
/srv/pxe/http/images/systemrescue.iso
|
||||
```
|
||||
|
||||
Since `/srv/pxe/http/images` is the NFS-backed symlink, place the ISO on the
|
||||
NFS share at `server.sweet.home:/tank/pxe-boot/images/systemrescue.iso`.
|
||||
|
||||
The `stage-systemrescue.service` oneshot extracts that ISO into:
|
||||
|
||||
```text
|
||||
/srv/pxe/http/systemrescue
|
||||
```
|
||||
|
||||
The rescue menu entry then chains `http://192.168.2.223/systemrescue.ipxe`,
|
||||
The rescue menu entry then chains `http://192.168.2.247/systemrescue.ipxe`,
|
||||
which loads the SystemRescue kernel and initramfs from the extracted tree and
|
||||
uses `archiso_http_srv` to fetch the squashfs payload over HTTP.
|
||||
|
||||
@@ -196,13 +94,6 @@ After deployment by an operator, basic service checks are:
|
||||
```bash
|
||||
curl http://pxe-boot/boot.ipxe
|
||||
curl http://pxe-boot/menu.ipxe
|
||||
curl http://pxe-boot/debian.ipxe
|
||||
curl -I http://pxe-boot/debian/linux
|
||||
curl -I http://pxe-boot/debian/initrd.gz
|
||||
curl http://pxe-boot/rocky-freeipa.ipxe
|
||||
curl http://pxe-boot/rocky-freeipa.ks
|
||||
curl -I http://pxe-boot/rocky/vmlinuz
|
||||
curl -I http://pxe-boot/rocky/initrd.img
|
||||
curl http://pxe-boot/systemrescue.ipxe
|
||||
curl -I http://pxe-boot/systemrescue/sysresccd/boot/x86_64/vmlinuz
|
||||
curl -I http://pxe-boot/systemrescue/sysresccd/boot/x86_64/sysresccd.img
|
||||
|
||||
Executable
+143
@@ -0,0 +1,143 @@
|
||||
# Spec: Refactor Flake Targets into Platform × Build-Type Matrix
|
||||
|
||||
## Context
|
||||
|
||||
The flake at `~/nixos` currently defines these output targets (flat, ad-hoc naming):
|
||||
|
||||
- `docker`
|
||||
- `linode-minimal`
|
||||
- `nix-cache`
|
||||
- `nix-minimal`
|
||||
- `nixos`
|
||||
- `server`
|
||||
- `pxe-boot`
|
||||
|
||||
Some already follow a `platform-buildtype` convention (`linode-minimal`), most don't.
|
||||
`~/nix-auto-installer` is a related repo and should be checked for any coupling to
|
||||
these target names (scripts, docs, CI, or install automation that reference them by
|
||||
name) before renaming anything.
|
||||
|
||||
## Goal
|
||||
|
||||
Restructure the flake so targets are generated from two orthogonal concepts:
|
||||
|
||||
**Build types** (what the system is for):
|
||||
- `minimal`
|
||||
- `nix-cache`
|
||||
- `server`
|
||||
- `docker`
|
||||
- `pxe-boot`
|
||||
- `gui`
|
||||
|
||||
**Platforms** (what it's deployed on):
|
||||
- `linode` (Linode VM)
|
||||
- `proxmox` (Proxmox VM)
|
||||
- `lxc` (Proxmox LXC container)
|
||||
|
||||
Final targets should be named consistently as `<platform>-<buildtype>`, e.g.:
|
||||
|
||||
```
|
||||
linode-minimal proxmox-minimal lxc-minimal
|
||||
linode-nix-cache proxmox-nix-cache lxc-nix-cache
|
||||
linode-server proxmox-server lxc-server
|
||||
linode-docker proxmox-docker lxc-docker
|
||||
linode-pxe-boot proxmox-pxe-boot lxc-pxe-boot
|
||||
linode-gui proxmox-gui lxc-gui
|
||||
```
|
||||
|
||||
That's the full matrix (18 targets) if every build type applies to every platform.
|
||||
See **Open Questions** below — some combinations may not make sense and should be
|
||||
confirmed with me before being built out, not silently included or dropped.
|
||||
|
||||
## Migration mapping (old → new)
|
||||
|
||||
| Old target | New target | Notes |
|
||||
|--------------------|------------------------------------------------------|-------|
|
||||
| `linode-minimal` | `linode-minimal` | Already correct, keep as-is |
|
||||
| `nix-minimal` | likely `proxmox-minimal` or a platform-less base module | Ambiguous — see Open Questions |
|
||||
| `nix-cache` | base module consumed by `linode-nix-cache`, `proxmox-nix-cache`, `lxc-nix-cache` | Currently platform-less; needs to become a build-type module, not a standalone target |
|
||||
| `server` | base module consumed by `linode-server`, `proxmox-server`, `lxc-server` | Same as above |
|
||||
| `docker` | base module consumed by `linode-docker`, `proxmox-docker`, `lxc-docker` | Confirm docker actually makes sense as an LXC/VM guest build vs. a standalone container image — see Open Questions |
|
||||
| `pxe-boot` | TBD — may stay a single target rather than a per-platform one | See Open Questions |
|
||||
| `nixos` | TBD — unclear what this maps to in the new scheme | See Open Questions |
|
||||
|
||||
## Open Questions (Claude Code: raise these with me before implementing, don't guess)
|
||||
|
||||
1. **`nixos` target** — what is this currently used for (bare metal install, dev
|
||||
shell, template)? It doesn't obviously map to any of the six build types.
|
||||
2. **`nix-minimal` vs `linode-minimal`** — are these two different things, or is
|
||||
`nix-minimal` a leftover/duplicate?
|
||||
3. **`pxe-boot` and `gui` across all three platforms** — does PXE boot make sense
|
||||
for an LXC container or a cloud VM (Linode), or is it inherently bare-metal/
|
||||
network-boot only and should remain a single non-platform target? Does `gui`
|
||||
make sense inside an LXC container?
|
||||
4. **`docker` as a build type** — is this "a NixOS host configured to run Docker"
|
||||
(which would sensibly have linode/proxmox/lxc variants), or "a Docker container
|
||||
image built by the flake" (which wouldn't take a platform prefix at all, since
|
||||
it doesn't run on Linode/Proxmox/LXC as a guest OS)? These are structurally
|
||||
different and change how it should be wired in.
|
||||
5. Confirm whether all 18 combinations should actually exist, or whether this is
|
||||
meant to produce only the combinations that are genuinely useful (e.g. maybe no
|
||||
one needs `lxc-pxe-boot`).
|
||||
|
||||
## Implementation approach
|
||||
|
||||
1. **Inventory first.** Read the current `flake.nix` and any `nixosConfigurations`/
|
||||
`modules` structure. Map every existing target to what module(s) it actually
|
||||
pulls in. Don't assume — confirm against the real file contents.
|
||||
2. **Separate build-type and platform into their own module directories**, e.g.:
|
||||
```
|
||||
modules/build-types/minimal.nix
|
||||
modules/build-types/nix-cache.nix
|
||||
modules/build-types/server.nix
|
||||
modules/build-types/docker.nix
|
||||
modules/build-types/pxe-boot.nix
|
||||
modules/build-types/gui.nix
|
||||
|
||||
modules/platforms/linode.nix
|
||||
modules/platforms/proxmox.nix
|
||||
modules/platforms/lxc.nix
|
||||
```
|
||||
Build-type modules should contain only what makes a system "minimal" vs
|
||||
"server" vs "gui", etc. Platform modules should contain only what's specific
|
||||
to running as a Linode VM vs Proxmox VM vs LXC container (virtualisation
|
||||
guest tools, boot method, filesystem/image format, LXC-specific constraints
|
||||
like no kernel modules, etc).
|
||||
3. **Generate the target matrix programmatically** in `flake.nix` rather than
|
||||
hand-writing 18 near-identical `nixosConfigurations` entries — e.g. a small
|
||||
function that takes a platform name and build-type name, composes the two
|
||||
modules plus any shared base module, and produces the named output. This
|
||||
keeps future build types/platforms a one-line addition rather than a copy-paste
|
||||
job.
|
||||
4. **Only build combinations we've confirmed make sense** (see Open Questions) —
|
||||
don't emit all 18 by default if some are structurally invalid.
|
||||
5. **Preserve existing working configs during the transition.** Don't delete the
|
||||
old target names until their replacements build successfully — rename/alias
|
||||
at the end, not the start, so there's no window where the flake is broken.
|
||||
|
||||
## Verification
|
||||
|
||||
For every new target produced:
|
||||
```bash
|
||||
nix flake check
|
||||
nix build .#nixosConfigurations.<target>.config.system.build.toplevel
|
||||
```
|
||||
Confirm each builds without evaluation errors before considering it done. If a
|
||||
target fails to build, report which one and why rather than silently skipping it.
|
||||
|
||||
## Deliverables
|
||||
|
||||
- Refactored `flake.nix` using the composed module + generated-matrix approach.
|
||||
- New `modules/build-types/*.nix` and `modules/platforms/*.nix` files.
|
||||
- Old flat target names removed only after their replacements are verified.
|
||||
- A short `README.md` (or section in existing docs) listing the final target
|
||||
names and what each one is for.
|
||||
- A summary at the end of what changed, what was removed, and any of the Open
|
||||
Questions above that got resolved differently than expected.
|
||||
|
||||
## Out of scope
|
||||
|
||||
- Don't touch `~/nix-auto-installer` contents beyond checking it for references
|
||||
to the old target names — if changes there are needed, flag them, don't make
|
||||
them without confirming.
|
||||
- Don't add new build types or platforms beyond the ones listed here.
|
||||
Generated
+7
-157
@@ -1,62 +1,5 @@
|
||||
{
|
||||
"nodes": {
|
||||
"clan-core": {
|
||||
"inputs": {
|
||||
"data-mesher": "data-mesher",
|
||||
"disko": [
|
||||
"disko"
|
||||
],
|
||||
"flake-parts": "flake-parts",
|
||||
"nix-darwin": "nix-darwin",
|
||||
"nix-select": "nix-select",
|
||||
"nixpkgs": [
|
||||
"nixpkgs"
|
||||
],
|
||||
"sops-nix": [
|
||||
"sops-nix"
|
||||
],
|
||||
"systems": "systems",
|
||||
"treefmt-nix": "treefmt-nix"
|
||||
},
|
||||
"locked": {
|
||||
"lastModified": 1783497933,
|
||||
"narHash": "sha256-TxmwEews6URFPqOWEHNychtXbFDgLZjbOfEXtvtOm6U=",
|
||||
"rev": "3dc0221ca09033599fe98055e9bbc81bdf32732a",
|
||||
"type": "tarball",
|
||||
"url": "https://git.clan.lol/api/v1/repos/clan/clan-core/archive/3dc0221ca09033599fe98055e9bbc81bdf32732a.tar.gz"
|
||||
},
|
||||
"original": {
|
||||
"type": "tarball",
|
||||
"url": "https://git.clan.lol/clan/clan-core/archive/26.05.tar.gz"
|
||||
}
|
||||
},
|
||||
"data-mesher": {
|
||||
"inputs": {
|
||||
"flake-parts": [
|
||||
"clan-core",
|
||||
"flake-parts"
|
||||
],
|
||||
"nixpkgs": [
|
||||
"clan-core",
|
||||
"nixpkgs"
|
||||
],
|
||||
"treefmt-nix": [
|
||||
"clan-core",
|
||||
"treefmt-nix"
|
||||
]
|
||||
},
|
||||
"locked": {
|
||||
"lastModified": 1778718524,
|
||||
"narHash": "sha256-pXLoI6Ax0EnUK6r34UM1vibVC7CfTu6j72R2692ZzPs=",
|
||||
"rev": "12c552ad547d87254f33f33bddd1a2cdbeac754d",
|
||||
"type": "tarball",
|
||||
"url": "https://git.clan.lol/api/v1/repos/clan/data-mesher/archive/12c552ad547d87254f33f33bddd1a2cdbeac754d.tar.gz"
|
||||
},
|
||||
"original": {
|
||||
"type": "tarball",
|
||||
"url": "https://git.clan.lol/clan/data-mesher/archive/main.tar.gz"
|
||||
}
|
||||
},
|
||||
"disko": {
|
||||
"inputs": {
|
||||
"nixpkgs": [
|
||||
@@ -108,30 +51,9 @@
|
||||
"type": "github"
|
||||
}
|
||||
},
|
||||
"flake-parts": {
|
||||
"inputs": {
|
||||
"nixpkgs-lib": [
|
||||
"clan-core",
|
||||
"nixpkgs"
|
||||
]
|
||||
},
|
||||
"locked": {
|
||||
"lastModified": 1778716662,
|
||||
"narHash": "sha256-m1Yf0wZ8j1OHjTc2UwHwyQRSnNeSgLJOd7q5Y45hzi4=",
|
||||
"owner": "hercules-ci",
|
||||
"repo": "flake-parts",
|
||||
"rev": "f7c1a2d347e4c52d5fb8d10cb4d94b5884e546fb",
|
||||
"type": "github"
|
||||
},
|
||||
"original": {
|
||||
"owner": "hercules-ci",
|
||||
"repo": "flake-parts",
|
||||
"type": "github"
|
||||
}
|
||||
},
|
||||
"flake-utils": {
|
||||
"inputs": {
|
||||
"systems": "systems_2"
|
||||
"systems": "systems"
|
||||
},
|
||||
"locked": {
|
||||
"lastModified": 1694529238,
|
||||
@@ -173,11 +95,11 @@
|
||||
]
|
||||
},
|
||||
"locked": {
|
||||
"lastModified": 1785119570,
|
||||
"narHash": "sha256-Rgs2xKnGLFWQscxUaXX07oyZeuMDOHEbqDOsgliLFGM=",
|
||||
"lastModified": 1783740085,
|
||||
"narHash": "sha256-qajyHfZY29G2oEQk+uHxmsJcRoBUBXP9maTpFlwP/dI=",
|
||||
"owner": "nix-community",
|
||||
"repo": "home-manager",
|
||||
"rev": "d4fd24667c8cbef124bb70a20380cab75ec8474d",
|
||||
"rev": "3cd22efe6471dc7365c822bd9ad73a21e55f38fb",
|
||||
"type": "github"
|
||||
},
|
||||
"original": {
|
||||
@@ -187,40 +109,6 @@
|
||||
"type": "github"
|
||||
}
|
||||
},
|
||||
"nix-darwin": {
|
||||
"inputs": {
|
||||
"nixpkgs": [
|
||||
"clan-core",
|
||||
"nixpkgs"
|
||||
]
|
||||
},
|
||||
"locked": {
|
||||
"lastModified": 1779036909,
|
||||
"narHash": "sha256-zXcwYQGCT6pzinK+1dBB2ekTVtfxGZAapb3Evdcu4fY=",
|
||||
"owner": "nix-darwin",
|
||||
"repo": "nix-darwin",
|
||||
"rev": "56c666e108467d87d13508936aade6d567f2a501",
|
||||
"type": "github"
|
||||
},
|
||||
"original": {
|
||||
"owner": "nix-darwin",
|
||||
"repo": "nix-darwin",
|
||||
"type": "github"
|
||||
}
|
||||
},
|
||||
"nix-select": {
|
||||
"locked": {
|
||||
"lastModified": 1763303120,
|
||||
"narHash": "sha256-yxcNOha7Cfv2nhVpz9ZXSNKk0R7wt4AiBklJ8D24rVg=",
|
||||
"rev": "3d1e3860bef36857a01a2ddecba7cdb0a14c35a9",
|
||||
"type": "tarball",
|
||||
"url": "https://git.clan.lol/api/v1/repos/clan/nix-select/archive/3d1e3860bef36857a01a2ddecba7cdb0a14c35a9.tar.gz"
|
||||
},
|
||||
"original": {
|
||||
"type": "tarball",
|
||||
"url": "https://git.clan.lol/clan/nix-select/archive/main.tar.gz"
|
||||
}
|
||||
},
|
||||
"nixos-conf-editor": {
|
||||
"inputs": {
|
||||
"flake-compat": "flake-compat",
|
||||
@@ -259,11 +147,11 @@
|
||||
},
|
||||
"nixpkgs_2": {
|
||||
"locked": {
|
||||
"lastModified": 1785104993,
|
||||
"narHash": "sha256-eKbrvPoAOFutbYMdbB3r5EQVmFxKv24iKqHPPUXA0gM=",
|
||||
"lastModified": 1784011430,
|
||||
"narHash": "sha256-lDebytrYdd47IBLwvNOD+6AGeoqZ78CIKlp70hzW280=",
|
||||
"owner": "NixOS",
|
||||
"repo": "nixpkgs",
|
||||
"rev": "8623c4c20aa4ca2f5fb81510d2944066c3fb0d96",
|
||||
"rev": "8eeec934ae0dbeca3d7868c059568a65c08b2fc3",
|
||||
"type": "github"
|
||||
},
|
||||
"original": {
|
||||
@@ -275,7 +163,6 @@
|
||||
},
|
||||
"root": {
|
||||
"inputs": {
|
||||
"clan-core": "clan-core",
|
||||
"disko": "disko",
|
||||
"home-manager": "home-manager",
|
||||
"nixos-conf-editor": "nixos-conf-editor",
|
||||
@@ -327,22 +214,6 @@
|
||||
}
|
||||
},
|
||||
"systems": {
|
||||
"locked": {
|
||||
"lastModified": 1774449309,
|
||||
"narHash": "sha256-brhZ8DmuGtzkCYHJg4HEd602amKm89Y9ytsFZ5uWD1w=",
|
||||
"owner": "nix-systems",
|
||||
"repo": "default",
|
||||
"rev": "c29398b59d2048c4ab79345812849c9bd15e9150",
|
||||
"type": "github"
|
||||
},
|
||||
"original": {
|
||||
"owner": "nix-systems",
|
||||
"ref": "future-26.11",
|
||||
"repo": "default",
|
||||
"type": "github"
|
||||
}
|
||||
},
|
||||
"systems_2": {
|
||||
"locked": {
|
||||
"lastModified": 1681028828,
|
||||
"narHash": "sha256-Vy1rq5AaRuLzOxct8nz4T6wlgyUR7zLU309k9mBC768=",
|
||||
@@ -356,27 +227,6 @@
|
||||
"repo": "default",
|
||||
"type": "github"
|
||||
}
|
||||
},
|
||||
"treefmt-nix": {
|
||||
"inputs": {
|
||||
"nixpkgs": [
|
||||
"clan-core",
|
||||
"nixpkgs"
|
||||
]
|
||||
},
|
||||
"locked": {
|
||||
"lastModified": 1780220602,
|
||||
"narHash": "sha256-eynAfOmbmxJnkp7YewvCEbShNnnYJ9gLLqkzsYtBPeM=",
|
||||
"owner": "numtide",
|
||||
"repo": "treefmt-nix",
|
||||
"rev": "db947814a175b7ca6ded66e21383d938df01c227",
|
||||
"type": "github"
|
||||
},
|
||||
"original": {
|
||||
"owner": "numtide",
|
||||
"repo": "treefmt-nix",
|
||||
"type": "github"
|
||||
}
|
||||
}
|
||||
},
|
||||
"root": "root",
|
||||
|
||||
@@ -16,19 +16,6 @@
|
||||
url = "github:Mic92/sops-nix";
|
||||
inputs.nixpkgs.follows = "nixpkgs";
|
||||
};
|
||||
clan-core = {
|
||||
url = "https://git.clan.lol/clan/clan-core/archive/26.05.tar.gz";
|
||||
# Deduplicate modules: clan-core bundles its own disko and sops-nix
|
||||
# (both imported by nixosModules.clanCore). Without follows, we'd get
|
||||
# two different versions of each, and disko's _module.args.diskoLib
|
||||
# unique option would conflict. With follows, clan-core uses the same
|
||||
# store paths as us, so NixOS deduplicates the imports.
|
||||
inputs = {
|
||||
nixpkgs.follows = "nixpkgs";
|
||||
disko.follows = "disko";
|
||||
sops-nix.follows = "sops-nix";
|
||||
};
|
||||
};
|
||||
};
|
||||
|
||||
outputs = { self, nixpkgs, nixos-conf-editor, home-manager, sops-nix, ... } @ inputs:
|
||||
@@ -54,22 +41,6 @@
|
||||
modules = [
|
||||
inputs.disko.nixosModules.disko
|
||||
sops-nix.nixosModules.sops
|
||||
inputs.clan-core.nixosModules.clanCore
|
||||
{
|
||||
# Required clan settings. directory is the flake root (where
|
||||
# vars/ and sops/ directories live); machine.name is the flake
|
||||
# target name (matches what clan vars generate uses as the key
|
||||
# under vars/per-machine/). enableRecommendedDefaults = false
|
||||
# is mandatory: without it, clan unconditionally enables
|
||||
# networking.useNetworkd, adds packages, and tweaks nix settings
|
||||
# -- none of which belong here.
|
||||
clan.core = {
|
||||
settings.directory = self;
|
||||
settings.machine.name = flakeTarget;
|
||||
enableRecommendedDefaults = false;
|
||||
};
|
||||
}
|
||||
./modules/clan/ssh-host-key.nix
|
||||
./modules/common/configuration.nix
|
||||
./modules/platforms/${platform}.nix
|
||||
./modules/build-types/${buildType}.nix
|
||||
@@ -94,7 +65,7 @@
|
||||
# file without a same-option circular dependency (a module
|
||||
# contributing to environment.etc can't read the merged
|
||||
# environment.etc it's itself contributing to).
|
||||
specialArgs = { inherit inputs vars netbootSystem netbootMinimalSystem flakeTarget; };
|
||||
specialArgs = { inherit inputs vars netbootSystem flakeTarget; };
|
||||
};
|
||||
|
||||
# Generated platform x build-type matrix. pxe-boot has no linode
|
||||
@@ -120,19 +91,13 @@
|
||||
linode-gui = mkTarget { platform = "linode"; buildType = "gui"; hostPath = ./hosts/nixos/host.nix; homeFile = ./hosts/nixos/home.nix; };
|
||||
proxmox-gui = mkTarget { platform = "proxmox"; buildType = "gui"; hostPath = ./hosts/nixos/host.nix; homeFile = ./hosts/nixos/home.nix; };
|
||||
lxc-gui = mkTarget { platform = "lxc"; buildType = "gui"; hostPath = ./hosts/nixos/host.nix; homeFile = ./hosts/nixos/home.nix; };
|
||||
baremetal-gui = mkTarget { platform = "baremetal"; buildType = "gui"; hostPath = ./hosts/nixos/host.nix; homeFile = ./hosts/nixos/home.nix; };
|
||||
|
||||
proxmox-pxe-boot = mkTarget { platform = "proxmox"; buildType = "pxe-boot"; hostPath = ./hosts/pxe-boot/host.nix; };
|
||||
lxc-pxe-boot = mkTarget { platform = "lxc"; buildType = "pxe-boot"; hostPath = ./hosts/pxe-boot/host.nix; };
|
||||
|
||||
linode-tailscale-router = mkTarget { platform = "linode"; buildType = "tailscale-router"; hostPath = ./hosts/tailscale-router/host.nix; };
|
||||
proxmox-tailscale-router = mkTarget { platform = "proxmox"; buildType = "tailscale-router"; hostPath = ./hosts/tailscale-router/host.nix; };
|
||||
lxc-tailscale-router = mkTarget { platform = "lxc"; buildType = "tailscale-router"; hostPath = ./hosts/tailscale-router/host.nix; };
|
||||
|
||||
lxc-tor-relay = mkTarget { platform = "lxc"; buildType = "tor-relay"; hostPath = ./hosts/tor-relay/host.nix; };
|
||||
|
||||
proxmox-ha-server-1 = mkTarget { platform = "proxmox"; buildType = "ha-server"; hostPath = ./hosts/ha-server-1/host.nix; };
|
||||
proxmox-ha-server-2 = mkTarget { platform = "proxmox"; buildType = "ha-server"; hostPath = ./hosts/ha-server-2/host.nix; };
|
||||
linode-tailscale-exit-node = mkTarget { platform = "linode"; buildType = "tailscale-exit-node"; hostPath = ./hosts/tailscale-exit-node/host.nix; };
|
||||
proxmox-tailscale-exit-node = mkTarget { platform = "proxmox"; buildType = "tailscale-exit-node"; hostPath = ./hosts/tailscale-exit-node/host.nix; };
|
||||
lxc-tailscale-exit-node = mkTarget { platform = "lxc"; buildType = "tailscale-exit-node"; hostPath = ./hosts/tailscale-exit-node/host.nix; };
|
||||
};
|
||||
|
||||
# Auto-install environments (migrated from the former nix-auto-installer
|
||||
@@ -152,61 +117,19 @@
|
||||
|
||||
# Same installer environment, built as netboot (kernel + initrd +
|
||||
# iPXE script) instead of an ISO — this is what packages.pxe bundles.
|
||||
#
|
||||
# Deliberately imports common.nix directly, NOT ./modules/installer/iso.nix
|
||||
# (which pulls in nixpkgs' installation-cd-minimal.nix) -- confirmed live
|
||||
# that composing the ISO module together with netboot-minimal.nix hangs
|
||||
# every boot waiting for a device that can never exist on a netboot
|
||||
# client ("A start job is running for /dev/disk/by-label/nixos-minimal-...").
|
||||
# Both installation-cd-base.nix and netboot.nix set fileSystems."/" via
|
||||
# the identical lib.mkImageMediaOverride (mkOverride 60) priority --
|
||||
# genuinely conflicting root-filesystem strategies (ISO-by-label vs.
|
||||
# netboot-tmpfs) at the same priority, and the ISO one was winning.
|
||||
# netboot-minimal.nix's own chain (netboot-base.nix) already imports
|
||||
# profiles/installation-device.nix independently, so common.nix's
|
||||
# initialHashedPassword override (which assumes that profile is
|
||||
# present) still applies correctly without iso.nix in the mix.
|
||||
#
|
||||
# networking.hostName is set explicitly (rather than left at nixpkgs'
|
||||
# own "nixos" default) so this image's generated system name
|
||||
# (nixos-system-auto-installer-*) matches its iPXE menu entry —
|
||||
# see modules/build-types/pxe-boot.nix's :auto-installer item — and
|
||||
# its staged directory, /srv/pxe/http/auto-installer.
|
||||
netbootSystem = nixpkgs.lib.nixosSystem {
|
||||
inherit system;
|
||||
modules = [
|
||||
./modules/installer/common.nix
|
||||
./modules/installer/iso.nix
|
||||
({ modulesPath, ... }: {
|
||||
imports = [
|
||||
(modulesPath + "/installer/netboot/netboot-minimal.nix")
|
||||
];
|
||||
})
|
||||
{ networking.hostName = "auto-installer"; }
|
||||
];
|
||||
specialArgs = { inherit vars; };
|
||||
};
|
||||
|
||||
# A genuinely vanilla NixOS minimal netboot image: nixpkgs'
|
||||
# netboot-minimal.nix on its own, with none of this flake's
|
||||
# auto-installer wiring (no common.nix — no auto-install.sh, no
|
||||
# baked host keys, no custom users/passwords). Built from source via
|
||||
# the same nixosSystem + netboot-minimal.nix path as netbootSystem
|
||||
# above, so both go through an identical build mechanism; the only
|
||||
# difference is what's composed in. hostName again matches this
|
||||
# image's iPXE menu entry (:nixos-minimal) and staged directory
|
||||
# (/srv/pxe/http/nixos-minimal).
|
||||
netbootMinimalSystem = nixpkgs.lib.nixosSystem {
|
||||
inherit system;
|
||||
modules = [
|
||||
({ modulesPath, ... }: {
|
||||
imports = [
|
||||
(modulesPath + "/installer/netboot/netboot-minimal.nix")
|
||||
];
|
||||
})
|
||||
{ networking.hostName = "nixos-minimal"; }
|
||||
];
|
||||
};
|
||||
|
||||
in
|
||||
{
|
||||
|
||||
@@ -228,15 +151,6 @@
|
||||
{ name = "initrd"; path = netbootSystem.config.system.build.netbootRamdisk; }
|
||||
{ name = "kernel"; path = netbootSystem.config.system.build.kernel; }
|
||||
];
|
||||
|
||||
# Vanilla NixOS minimal netboot bundle — see netbootMinimalSystem
|
||||
# above. Staged onto the pxe-boot host alongside packages.pxe by
|
||||
# modules/pxe-boot/stage-installer-artifacts.nix.
|
||||
pxe-minimal = pkgs.linkFarm "pxe-minimal" [
|
||||
{ name = "netboot.ipxe"; path = netbootMinimalSystem.config.system.build.netbootIpxeScript; }
|
||||
{ name = "initrd"; path = netbootMinimalSystem.config.system.build.netbootRamdisk; }
|
||||
{ name = "kernel"; path = netbootMinimalSystem.config.system.build.kernel; }
|
||||
];
|
||||
};
|
||||
};
|
||||
}
|
||||
|
||||
+3
-17
@@ -1,24 +1,10 @@
|
||||
{ vars, ... }:
|
||||
_:
|
||||
|
||||
{
|
||||
networking = {
|
||||
hostName = "docker";
|
||||
hostId = "007f0200";
|
||||
useDHCP = false;
|
||||
interfaces.${vars.vmLanInterface}.ipv4.addresses = [{
|
||||
address = vars.dockerIp;
|
||||
prefixLength = vars.lanPrefixLength;
|
||||
}];
|
||||
defaultGateway = { address = vars.lanGateway; interface = vars.vmLanInterface; };
|
||||
nameservers = [ vars.domainControllerIp ];
|
||||
};
|
||||
networking.hostName = "docker";
|
||||
networking.hostId = "007f0200";
|
||||
boot.zfs.forceImportRoot = false;
|
||||
|
||||
# Only advertise the LAN interface to IPA DNS. Without this, SSSD registers
|
||||
# every Docker bridge (172.x.x.x) as an A record for docker.sweet.home —
|
||||
# the default dyndns.interface = "*" catches them all.
|
||||
security.ipa.dyndns.interface = vars.lxcLanInterface; # eth0
|
||||
|
||||
# Preserved from the pre-refactor `docker` target — stateVersion must never
|
||||
# be bumped on an already-installed machine.
|
||||
system.stateVersion = "25.05";
|
||||
|
||||
@@ -1,30 +0,0 @@
|
||||
{ vars, ... }:
|
||||
{
|
||||
imports = [
|
||||
(import ../../modules/beszel/host-token.nix {
|
||||
name = "ha-server-1";
|
||||
sopsFile = ../../secrets/ha-server-1.yaml;
|
||||
})
|
||||
];
|
||||
|
||||
networking = {
|
||||
hostName = vars.haServer1Host;
|
||||
hostId = "3a4b5c6d";
|
||||
useDHCP = false;
|
||||
interfaces.${vars.vmLanInterface}.ipv4.addresses = [{
|
||||
address = vars.haServer1Ip;
|
||||
prefixLength = vars.lanPrefixLength;
|
||||
}];
|
||||
interfaces.${vars.vmStorageInterface}.ipv4.addresses = [{
|
||||
address = vars.haServer1StorageIp;
|
||||
prefixLength = vars.haStoragePrefixLength;
|
||||
}];
|
||||
defaultGateway = { address = vars.lanGateway; interface = vars.vmLanInterface; };
|
||||
nameservers = [ vars.domainControllerIp ];
|
||||
};
|
||||
|
||||
# Set KEY after pairing this host with the beszel hub; the token is sops-managed.
|
||||
services.beszel.agent.environment.KEY = "";
|
||||
|
||||
system.stateVersion = "26.05";
|
||||
}
|
||||
@@ -1,30 +0,0 @@
|
||||
{ vars, ... }:
|
||||
{
|
||||
imports = [
|
||||
(import ../../modules/beszel/host-token.nix {
|
||||
name = "ha-server-2";
|
||||
sopsFile = ../../secrets/ha-server-2.yaml;
|
||||
})
|
||||
];
|
||||
|
||||
networking = {
|
||||
hostName = vars.haServer2Host;
|
||||
hostId = "7e8f9a0b";
|
||||
useDHCP = false;
|
||||
interfaces.${vars.vmLanInterface}.ipv4.addresses = [{
|
||||
address = vars.haServer2Ip;
|
||||
prefixLength = vars.lanPrefixLength;
|
||||
}];
|
||||
interfaces.${vars.vmStorageInterface}.ipv4.addresses = [{
|
||||
address = vars.haServer2StorageIp;
|
||||
prefixLength = vars.haStoragePrefixLength;
|
||||
}];
|
||||
defaultGateway = { address = vars.lanGateway; interface = vars.vmLanInterface; };
|
||||
nameservers = [ vars.domainControllerIp ];
|
||||
};
|
||||
|
||||
# Set KEY after pairing this host with the beszel hub; the token is sops-managed.
|
||||
services.beszel.agent.environment.KEY = "";
|
||||
|
||||
system.stateVersion = "26.05";
|
||||
}
|
||||
@@ -8,18 +8,10 @@
|
||||
})
|
||||
];
|
||||
|
||||
networking = {
|
||||
hostName = vars.nixCacheHost;
|
||||
useDHCP = false;
|
||||
interfaces.${vars.lxcLanInterface}.ipv4.addresses = [{
|
||||
address = vars.nixCacheIp;
|
||||
prefixLength = vars.lanPrefixLength;
|
||||
}];
|
||||
defaultGateway = { address = vars.lanGateway; interface = vars.lxcLanInterface; };
|
||||
nameservers = [ vars.domainControllerIp ];
|
||||
};
|
||||
networking.hostName = vars.nixCacheHost;
|
||||
|
||||
services.beszel.agent.environment = {
|
||||
#DOCKER_HOST = "tcp://docker-socket-proxy:2375";
|
||||
KEY = "ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIFPR9kwtC4TAeTRu46A7+opZsYpxqkRJ+x/ZyB2GWCeG";
|
||||
};
|
||||
|
||||
|
||||
@@ -19,15 +19,11 @@
|
||||
nextcloud-client
|
||||
# vscode
|
||||
chromium
|
||||
claude-code
|
||||
fish
|
||||
sops
|
||||
];
|
||||
|
||||
# Optional: set environment vars
|
||||
sessionVariables = {
|
||||
EDITOR = "vim";
|
||||
SOPS_AGE_KEY_FILE = "${config.home.homeDirectory}/.config/sops/age/keys.txt";
|
||||
};
|
||||
|
||||
file = {
|
||||
|
||||
@@ -1,17 +1,8 @@
|
||||
_:
|
||||
|
||||
{
|
||||
imports = [
|
||||
../../modules/networking/wifi.nix
|
||||
];
|
||||
|
||||
networking.hostName = "nixos";
|
||||
|
||||
# Only needed now that baremetal-gui exists (ZFS root) -- harmless on the
|
||||
# ext4-rooted linode/proxmox/lxc-gui variants, so set unconditionally
|
||||
# rather than only on the baremetal platform.
|
||||
networking.hostId = "de6a9ffc";
|
||||
|
||||
# Preserved from the pre-refactor `nixos` target — stateVersion must never
|
||||
# be bumped on an already-installed machine.
|
||||
system.stateVersion = "25.05";
|
||||
|
||||
+2
-11
@@ -1,16 +1,7 @@
|
||||
{ vars, ... }:
|
||||
_:
|
||||
|
||||
{
|
||||
networking = {
|
||||
hostName = "pxe-boot";
|
||||
useDHCP = false;
|
||||
interfaces.${vars.lxcLanInterface}.ipv4.addresses = [{
|
||||
address = vars.pxeServerIp;
|
||||
prefixLength = vars.lanPrefixLength;
|
||||
}];
|
||||
defaultGateway = { address = vars.lanGateway; interface = vars.lxcLanInterface; };
|
||||
nameservers = [ vars.domainControllerIp ];
|
||||
};
|
||||
networking.hostName = "pxe-boot";
|
||||
|
||||
# Preserved from the pre-refactor `pxe-boot` target — stateVersion must
|
||||
# never be bumped on an already-installed machine.
|
||||
|
||||
+3
-11
@@ -8,19 +8,11 @@
|
||||
})
|
||||
];
|
||||
|
||||
networking = {
|
||||
hostName = vars.nfsServerHost;
|
||||
hostId = "6689f93e";
|
||||
useDHCP = false;
|
||||
interfaces.${vars.vmLanInterface}.ipv4.addresses = [{
|
||||
address = vars.serverIp;
|
||||
prefixLength = vars.lanPrefixLength;
|
||||
}];
|
||||
defaultGateway = { address = vars.lanGateway; interface = vars.vmLanInterface; };
|
||||
nameservers = [ vars.domainControllerIp ];
|
||||
};
|
||||
networking.hostName = vars.nfsServerHost;
|
||||
networking.hostId = "6689f93e";
|
||||
|
||||
services.beszel.agent.environment = {
|
||||
#DOCKER_HOST = "tcp://docker-socket-proxy:2375";
|
||||
KEY = "ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIFPR9kwtC4TAeTRu46A7+opZsYpxqkRJ+x/ZyB2GWCeG";
|
||||
EXTRA_FILESYSTEMS = "${vars.storageRoot}/${vars.nfsShares.dockerVolumes.subpath}";
|
||||
LOG_LEVEL = "debug";
|
||||
|
||||
@@ -0,0 +1,12 @@
|
||||
_:
|
||||
|
||||
{
|
||||
networking.hostName = "exit-node";
|
||||
|
||||
# No networking.hostId: only ZFS-touching hosts (server, docker) need one
|
||||
# for pool-import safety, and this host does neither.
|
||||
|
||||
# A genuinely new host (not a pre-refactor carry-over), so it tracks the
|
||||
# flake's current nixpkgs release rather than being pinned to an older one.
|
||||
system.stateVersion = "26.05";
|
||||
}
|
||||
@@ -1,30 +0,0 @@
|
||||
{ vars, ... }:
|
||||
|
||||
{
|
||||
imports = [
|
||||
(import ../../modules/beszel/host-token.nix {
|
||||
name = "tailscale-router";
|
||||
sopsFile = ../../secrets/tailscale-router.yaml;
|
||||
})
|
||||
];
|
||||
|
||||
networking = {
|
||||
hostName = "tailscale-router";
|
||||
useDHCP = false;
|
||||
interfaces.${vars.lxcLanInterface}.ipv4.addresses = [{
|
||||
address = vars.tailscaleRouterIp;
|
||||
prefixLength = vars.lanPrefixLength;
|
||||
}];
|
||||
defaultGateway = { address = vars.lanGateway; interface = vars.lxcLanInterface; };
|
||||
nameservers = [ vars.domainControllerIp ];
|
||||
};
|
||||
|
||||
services.beszel.agent.environment = {
|
||||
KEY = "ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIFPR9kwtC4TAeTRu46A7+opZsYpxqkRJ+x/ZyB2GWCeG";
|
||||
};
|
||||
|
||||
# No networking.hostId: only ZFS-touching hosts (server, docker) need one
|
||||
# for pool-import safety, and this host does neither.
|
||||
|
||||
system.stateVersion = "26.05";
|
||||
}
|
||||
@@ -1,32 +0,0 @@
|
||||
{ vars, ... }:
|
||||
|
||||
{
|
||||
imports = [
|
||||
(import ../../modules/beszel/host-token.nix {
|
||||
name = "tor-relay";
|
||||
sopsFile = ../../secrets/tor-relay.yaml;
|
||||
})
|
||||
];
|
||||
|
||||
networking = {
|
||||
hostName = "tor-relay";
|
||||
useDHCP = false;
|
||||
interfaces.${vars.lxcLanInterface}.ipv4.addresses = [{
|
||||
address = vars.torRelayIp;
|
||||
prefixLength = vars.lanPrefixLength;
|
||||
}];
|
||||
defaultGateway = { address = vars.lanGateway; interface = vars.lxcLanInterface; };
|
||||
nameservers = [ vars.domainControllerIp ];
|
||||
};
|
||||
|
||||
# No networking.hostId: only ZFS-touching hosts need one for pool-import
|
||||
# safety, and this host does neither.
|
||||
|
||||
services.beszel.agent.environment = {
|
||||
KEY = "ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIFPR9kwtC4TAeTRu46A7+opZsYpxqkRJ+x/ZyB2GWCeG";
|
||||
};
|
||||
|
||||
# A genuinely new host (not a pre-refactor carry-over), so it tracks the
|
||||
# flake's current nixpkgs release rather than being pinned to an older one.
|
||||
system.stateVersion = "26.05";
|
||||
}
|
||||
@@ -15,6 +15,7 @@
|
||||
../docker/enable-service.nix
|
||||
../docker/nextcloud-cron-job.nix
|
||||
../docker/docker-health-to-gotify.nix
|
||||
../tailscale/enable-service.nix
|
||||
../traefik/rotate-logs.nix
|
||||
../raspi/mount-data.nix
|
||||
../services/enable-rpcbind.nix
|
||||
|
||||
@@ -1,17 +1,6 @@
|
||||
{ config, pkgs, lib, inputs, vars, ... }:
|
||||
|
||||
{
|
||||
imports = [
|
||||
../docker/enable-service.nix
|
||||
];
|
||||
|
||||
nixpkgs.overlays = [
|
||||
(final: prev: {
|
||||
docker = prev.docker_29;
|
||||
docker_cli = prev.docker_29;
|
||||
})
|
||||
];
|
||||
|
||||
environment.systemPackages = with pkgs; [
|
||||
inputs.nixos-conf-editor.packages.${pkgs.stdenv.hostPlatform.system}.nixos-conf-editor
|
||||
nodejs
|
||||
@@ -29,7 +18,7 @@
|
||||
];
|
||||
|
||||
boot.loader.grub.useOSProber = true;
|
||||
programs.direnv.enable = true;
|
||||
|
||||
services = {
|
||||
xserver = {
|
||||
enable = true;
|
||||
|
||||
@@ -1,48 +0,0 @@
|
||||
# HA file server build type: DRBD + XFS + LIO iSCSI + NFS, managed by
|
||||
# Corosync + Pacemaker. Both ha-server-1 and ha-server-2 use this type.
|
||||
#
|
||||
# NFS start/stop:
|
||||
# services.nfs.server.enable = true configures /etc/exports, wires up
|
||||
# rpcbind, and loads kernel modules — but nfs-server.service.wantedBy is
|
||||
# force-cleared so systemd does NOT auto-start it at boot. Pacemaker's
|
||||
# ha-group resource group (configured by scripts/ha/cluster-init.sh)
|
||||
# starts and stops nfs-server as part of the failover sequence after the
|
||||
# XFS mount and iSCSI target are brought up on the new Active node.
|
||||
#
|
||||
# Beszel agent:
|
||||
# Enabled here via enable-agent.nix. The agent KEY (used to pair with
|
||||
# the Beszel hub) is not set yet — add it to hosts/ha-server-{1,2}/host.nix
|
||||
# under services.beszel.agent.environment.KEY once the hub accepts the
|
||||
# new agents, following the pattern in hosts/server/host.nix.
|
||||
{ lib, pkgs, vars, ... }:
|
||||
|
||||
let
|
||||
# Generates /etc/exports lines for all nfsShares data entries. Shared
|
||||
# pattern with modules/build-types/server.nix — both export the same
|
||||
# set of shares, differing only in the storage root they serve from.
|
||||
mkNfsExports = storageRoot:
|
||||
lib.concatMapStrings
|
||||
(share: " ${storageRoot}/${share.subpath} ${vars.lanCidr}${vars.nfsShares.options}\n")
|
||||
(lib.filter builtins.isAttrs (lib.attrValues vars.nfsShares));
|
||||
in
|
||||
{
|
||||
imports = [
|
||||
../ha/pacemaker-stack.nix
|
||||
../ha/iscsi-target.nix
|
||||
../ha/cluster-config.nix
|
||||
../beszel/enable-agent.nix
|
||||
];
|
||||
|
||||
# xfsprogs: mkfs.xfs/xfs_info needed by cluster-init.sh.
|
||||
# openiscsi: iscsiadm needed by acceptance-tests.sh T4 (iSCSI discovery check).
|
||||
environment.systemPackages = [ pkgs.xfsprogs pkgs.openiscsi ];
|
||||
|
||||
services.nfs.server = {
|
||||
enable = true;
|
||||
exports = mkNfsExports vars.haStorageRoot;
|
||||
};
|
||||
|
||||
# Pacemaker controls nfs-server — prevent systemd from starting it at boot
|
||||
# on both nodes (only the Active node should be serving NFS).
|
||||
systemd.services.nfs-server.wantedBy = lib.mkForce [ ];
|
||||
}
|
||||
@@ -21,216 +21,6 @@ let
|
||||
chain ${pxeBaseUrl}/boot.ipxe
|
||||
'';
|
||||
|
||||
debianRelease = "bookworm";
|
||||
debianMirror = "https://deb.debian.org/debian";
|
||||
debianNetbootBase = "${debianMirror}/dists/${debianRelease}/main/installer-amd64/current/images/netboot/debian-installer/amd64";
|
||||
|
||||
rockyRelease = "9";
|
||||
rockyArch = "x86_64";
|
||||
rockyMirror = "https://dl.rockylinux.org/pub/rocky/${rockyRelease}";
|
||||
rockyPxebootBase = "${rockyMirror}/BaseOS/${rockyArch}/os/images/pxeboot";
|
||||
|
||||
debianIpxe = pkgs.writeText "debian.ipxe" ''
|
||||
#!ipxe
|
||||
|
||||
set base ${pxeBaseUrl}
|
||||
|
||||
kernel ''${base}/debian/linux
|
||||
initrd ''${base}/debian/initrd.gz
|
||||
boot
|
||||
'';
|
||||
|
||||
fetchDebianNetboot = pkgs.writeShellScript "fetch-debian-netboot" ''
|
||||
set -eu
|
||||
|
||||
dir="${httpRoot}/debian"
|
||||
mirror="${debianNetbootBase}"
|
||||
|
||||
if [ -f "$dir/linux" ] && [ -f "$dir/initrd.gz" ]; then
|
||||
echo "Debian ${debianRelease} netboot files already present; skipping download."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
echo "Downloading Debian ${debianRelease} netboot kernel and initrd from $mirror ..."
|
||||
${pkgs.curl}/bin/curl -fsSL -o "$dir/linux.tmp" "$mirror/linux"
|
||||
${pkgs.curl}/bin/curl -fsSL -o "$dir/initrd.gz.tmp" "$mirror/initrd.gz"
|
||||
mv "$dir/linux.tmp" "$dir/linux"
|
||||
mv "$dir/initrd.gz.tmp" "$dir/initrd.gz"
|
||||
echo "Debian ${debianRelease} netboot files staged."
|
||||
'';
|
||||
|
||||
# Rocky Linux 9 iPXE script — boots vmlinuz+initrd.img from the staged
|
||||
# /rocky/ directory and hands Anaconda the hosted Kickstart URL.
|
||||
# net.ifnames=0 biosdevname=0 ensures the NIC is eth0 in both the
|
||||
# installer and the installed system (matches the Kickstart NM config).
|
||||
rockyFreeIpaIpxe = pkgs.writeText "rocky-freeipa.ipxe" ''
|
||||
#!ipxe
|
||||
|
||||
set base ${pxeBaseUrl}
|
||||
|
||||
kernel ''${base}/rocky/vmlinuz inst.ks=''${base}/rocky-freeipa.ks inst.repo=${rockyMirror}/BaseOS/${rockyArch}/os/ net.ifnames=0 biosdevname=0 ip=dhcp quiet
|
||||
initrd ''${base}/rocky/initrd.img
|
||||
boot
|
||||
'';
|
||||
|
||||
# Kickstart file for domain-controller.sweet.home.
|
||||
# Installs Rocky Linux 9, sets a static IP, creates wayne with the
|
||||
# admin SSH key, then on first reboot runs ipa-server-install via a
|
||||
# systemd oneshot service. Passwords are generated at %post time,
|
||||
# written to /root/ipa-credentials.txt (chmod 600), and read back by
|
||||
# the first-boot script — never hardcoded here or in the repo.
|
||||
rockyFreeIpaKs = pkgs.writeText "rocky-freeipa.ks" ''
|
||||
#version=RHEL9
|
||||
# Unattended Rocky Linux 9 + FreeIPA install
|
||||
# Target: domain-controller.${vars.homeDomain} ${vars.domainControllerIp}
|
||||
|
||||
url --url=${rockyMirror}/BaseOS/${rockyArch}/os/
|
||||
repo --name=appstream --baseurl=${rockyMirror}/AppStream/${rockyArch}/os/
|
||||
|
||||
lang en_US.UTF-8
|
||||
keyboard us
|
||||
timezone UTC --utc
|
||||
|
||||
# DHCP during install; static IP configured in %post via NM config file
|
||||
network --bootproto=dhcp --device=link --activate
|
||||
network --hostname=domain-controller.sweet.home
|
||||
|
||||
selinux --enforcing
|
||||
firewall --enabled --service=ssh
|
||||
|
||||
rootpw --lock
|
||||
user --name=wayne --groups=wheel --shell=/bin/bash
|
||||
sshkey --username=wayne "${vars.adminSshKey}"
|
||||
|
||||
zerombr
|
||||
clearpart --all --initlabel --drives=sda
|
||||
# Keep net.ifnames=0 biosdevname=0 in the installed GRUB so the NIC
|
||||
# stays eth0 after reboot (matches the NM connection file below).
|
||||
bootloader --location=mbr --boot-drive=sda --append="net.ifnames=0 biosdevname=0"
|
||||
|
||||
part /boot --fstype=xfs --size=1024 --ondisk=sda
|
||||
part swap --fstype=swap --size=2048 --ondisk=sda
|
||||
part / --fstype=xfs --grow --size=1 --ondisk=sda --asprimary
|
||||
|
||||
%packages
|
||||
@^minimal-environment
|
||||
ipa-server
|
||||
ipa-server-dns
|
||||
%end
|
||||
|
||||
reboot
|
||||
|
||||
%post --log=/root/ks-post.log
|
||||
set -euo pipefail
|
||||
|
||||
# -- Static IP: write NM connection file directly (NM not running in chroot) --
|
||||
mkdir -p /etc/NetworkManager/system-connections
|
||||
cat > /etc/NetworkManager/system-connections/eth0.nmconnection << 'NMCONN'
|
||||
[connection]
|
||||
id=eth0
|
||||
type=ethernet
|
||||
interface-name=eth0
|
||||
autoconnect=true
|
||||
|
||||
[ethernet]
|
||||
|
||||
[ipv4]
|
||||
method=manual
|
||||
addresses=${vars.domainControllerIp}/${toString vars.lanPrefixLength}
|
||||
gateway=${vars.lanGateway}
|
||||
dns=${vars.domainControllerIp};
|
||||
dns-search=${vars.homeDomain};
|
||||
|
||||
[ipv6]
|
||||
method=auto
|
||||
NMCONN
|
||||
chmod 600 /etc/NetworkManager/system-connections/eth0.nmconnection
|
||||
|
||||
# -- /etc/hosts: FQDN must resolve to the real IP (not loopback) for IPA --
|
||||
sed -i '/domain-controller/d' /etc/hosts
|
||||
echo '${vars.domainControllerIp} domain-controller.${vars.homeDomain} domain-controller' >> /etc/hosts
|
||||
|
||||
# -- Generate IPA passwords and store securely --
|
||||
DM_PASS=$(openssl rand -base64 24 | tr -dc 'A-Za-z0-9' | head -c 24)
|
||||
ADMIN_PASS=$(openssl rand -base64 24 | tr -dc 'A-Za-z0-9' | head -c 24)
|
||||
printf 'Directory Manager: %s\nIPA Admin: %s\n' "$DM_PASS" "$ADMIN_PASS" \
|
||||
> /root/ipa-credentials.txt
|
||||
chmod 600 /root/ipa-credentials.txt
|
||||
|
||||
# -- First-boot script: reads passwords back, runs ipa-server-install --
|
||||
cat > /usr/local/sbin/freeipa-first-boot.sh << 'FIRSTBOOT'
|
||||
#!/bin/bash
|
||||
set -euo pipefail
|
||||
exec >> /root/freeipa-install.log 2>&1
|
||||
echo "=== FreeIPA first-boot install started at $(date) ==="
|
||||
|
||||
DM_PASS=$(grep '^Directory Manager:' /root/ipa-credentials.txt | awk '{print $NF}')
|
||||
ADMIN_PASS=$(grep '^IPA Admin:' /root/ipa-credentials.txt | awk '{print $NF}')
|
||||
|
||||
ipa-server-install \
|
||||
--realm=SWEET.HOME \
|
||||
--domain=sweet.home \
|
||||
--hostname=domain-controller.sweet.home \
|
||||
--ds-password="$DM_PASS" \
|
||||
--admin-password="$ADMIN_PASS" \
|
||||
--setup-dns \
|
||||
--forwarder=192.168.2.253 \
|
||||
--no-dnssec-validation \
|
||||
--no-ntp \
|
||||
--unattended
|
||||
|
||||
echo "=== FreeIPA install complete at $(date) ==="
|
||||
echo "Credentials: /root/ipa-credentials.txt (save to password manager)"
|
||||
echo "CA backup: /root/cacert.p12 (encrypted with Directory Manager password)"
|
||||
systemctl disable freeipa-first-boot.service
|
||||
FIRSTBOOT
|
||||
chmod 700 /usr/local/sbin/freeipa-first-boot.sh
|
||||
|
||||
# -- Systemd oneshot service: runs freeipa-first-boot.sh on first real boot --
|
||||
cat > /etc/systemd/system/freeipa-first-boot.service << 'UNIT'
|
||||
[Unit]
|
||||
Description=FreeIPA first-boot installation
|
||||
After=network-online.target
|
||||
Wants=network-online.target
|
||||
ConditionPathExists=/root/ipa-credentials.txt
|
||||
|
||||
[Service]
|
||||
Type=oneshot
|
||||
ExecStart=/usr/local/sbin/freeipa-first-boot.sh
|
||||
TimeoutStartSec=1800
|
||||
RemainAfterExit=yes
|
||||
|
||||
[Install]
|
||||
WantedBy=multi-user.target
|
||||
UNIT
|
||||
|
||||
mkdir -p /etc/systemd/system/multi-user.target.wants
|
||||
ln -sf /etc/systemd/system/freeipa-first-boot.service \
|
||||
/etc/systemd/system/multi-user.target.wants/freeipa-first-boot.service
|
||||
|
||||
echo "Kickstart %post complete. FreeIPA installs on first reboot (~20 min)."
|
||||
%end
|
||||
'';
|
||||
|
||||
fetchRockyPxeboot = pkgs.writeShellScript "fetch-rocky-pxeboot" ''
|
||||
set -eu
|
||||
|
||||
dir="${httpRoot}/rocky"
|
||||
base="${rockyPxebootBase}"
|
||||
|
||||
if [ -f "$dir/vmlinuz" ] && [ -f "$dir/initrd.img" ]; then
|
||||
echo "Rocky Linux ${rockyRelease} pxeboot files already present; skipping download."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
echo "Downloading Rocky Linux ${rockyRelease} pxeboot kernel and initrd from $base ..."
|
||||
${pkgs.curl}/bin/curl -fsSL -o "$dir/vmlinuz.tmp" "$base/vmlinuz"
|
||||
${pkgs.curl}/bin/curl -fsSL -o "$dir/initrd.img.tmp" "$base/initrd.img"
|
||||
mv "$dir/vmlinuz.tmp" "$dir/vmlinuz"
|
||||
mv "$dir/initrd.img.tmp" "$dir/initrd.img"
|
||||
echo "Rocky Linux ${rockyRelease} pxeboot files staged."
|
||||
'';
|
||||
|
||||
systemRescueIpxe = pkgs.writeText "systemrescue.ipxe" ''
|
||||
#!ipxe
|
||||
|
||||
@@ -278,27 +68,15 @@ let
|
||||
set base ${pxeBaseUrl}
|
||||
|
||||
menu PXE Boot Menu
|
||||
item auto-installer NixOS Auto-Installer
|
||||
item nixos-minimal NixOS Minimal
|
||||
item debian Debian Minimal
|
||||
item rocky-freeipa FreeIPA Server (Rocky Linux 9)
|
||||
item rescue Rescue Environment
|
||||
item shell iPXE Shell
|
||||
item reboot Reboot
|
||||
item nixos NixOS Installer
|
||||
item rescue Rescue Environment
|
||||
item shell iPXE Shell
|
||||
item reboot Reboot
|
||||
|
||||
choose target && goto ''${target}
|
||||
|
||||
:auto-installer
|
||||
chain ''${base}/auto-installer/netboot.ipxe
|
||||
|
||||
:nixos-minimal
|
||||
chain ''${base}/nixos-minimal/netboot.ipxe
|
||||
|
||||
:debian
|
||||
chain ''${base}/debian.ipxe
|
||||
|
||||
:rocky-freeipa
|
||||
chain ''${base}/rocky-freeipa.ipxe
|
||||
:nixos
|
||||
chain ''${base}/nixos/netboot.ipxe
|
||||
|
||||
:rescue
|
||||
chain ''${base}/systemrescue.ipxe
|
||||
@@ -313,7 +91,6 @@ in
|
||||
{
|
||||
imports = [
|
||||
../pxe-boot/stage-installer-artifacts.nix
|
||||
../pxe-boot/mount-pxe-images.nix
|
||||
];
|
||||
|
||||
environment.systemPackages = with pkgs; [
|
||||
@@ -348,101 +125,36 @@ in
|
||||
openssh.settings.PermitRootLogin = "yes";
|
||||
};
|
||||
|
||||
systemd = {
|
||||
tmpfiles.rules = [
|
||||
"d ${pxeRoot} 0755 root root -"
|
||||
"d ${httpRoot} 0755 root root -"
|
||||
"L+ ${httpRoot}/images - - - - ${vars.nfsShares.pxebootImages.mountpoint}"
|
||||
"d ${httpRoot}/auto-installer 0755 root root -"
|
||||
"d ${httpRoot}/nixos-minimal 0755 root root -"
|
||||
"d ${httpRoot}/systemrescue 0755 root root -"
|
||||
"d ${httpRoot}/debian 0755 root root -"
|
||||
"d ${httpRoot}/ubuntu 0755 root root -"
|
||||
"d ${httpRoot}/rescue 0755 root root -"
|
||||
"d ${httpRoot}/rocky 0755 root root -"
|
||||
"d ${tftpRoot} 0755 root root -"
|
||||
"C+ ${httpRoot}/boot.ipxe 0644 root root - ${bootIpxe}"
|
||||
"C+ ${httpRoot}/menu.ipxe 0644 root root - ${menuIpxe}"
|
||||
"C+ ${httpRoot}/debian.ipxe 0644 root root - ${debianIpxe}"
|
||||
"C+ ${httpRoot}/rocky-freeipa.ipxe 0644 root root - ${rockyFreeIpaIpxe}"
|
||||
"C+ ${httpRoot}/rocky-freeipa.ks 0644 root root - ${rockyFreeIpaKs}"
|
||||
"C+ ${httpRoot}/systemrescue.ipxe 0644 root root - ${systemRescueIpxe}"
|
||||
"C+ ${tftpRoot}/autoexec.ipxe 0644 root root - ${autoexecIpxe}"
|
||||
"C+ ${tftpRoot}/ipxe.efi 0644 root root - ${pkgs.ipxe}/ipxe.efi"
|
||||
"C+ ${tftpRoot}/undionly.kpxe 0644 root root - ${pkgs.ipxe}/undionly.kpxe"
|
||||
systemd.tmpfiles.rules = [
|
||||
"d ${pxeRoot} 0755 root root -"
|
||||
"d ${httpRoot} 0755 root root -"
|
||||
"d ${httpRoot}/images 0755 root root -"
|
||||
"d ${httpRoot}/nixos 0755 root root -"
|
||||
"d ${httpRoot}/systemrescue 0755 root root -"
|
||||
"d ${httpRoot}/ubuntu 0755 root root -"
|
||||
"d ${httpRoot}/rescue 0755 root root -"
|
||||
"d ${tftpRoot} 0755 root root -"
|
||||
"C+ ${httpRoot}/boot.ipxe 0644 root root - ${bootIpxe}"
|
||||
"C+ ${httpRoot}/menu.ipxe 0644 root root - ${menuIpxe}"
|
||||
"C+ ${httpRoot}/systemrescue.ipxe 0644 root root - ${systemRescueIpxe}"
|
||||
"C+ ${tftpRoot}/autoexec.ipxe 0644 root root - ${autoexecIpxe}"
|
||||
"C+ ${tftpRoot}/ipxe.efi 0644 root root - ${pkgs.ipxe}/ipxe.efi"
|
||||
"C+ ${tftpRoot}/undionly.kpxe 0644 root root - ${pkgs.ipxe}/undionly.kpxe"
|
||||
];
|
||||
|
||||
systemd.services.stage-systemrescue = {
|
||||
description = "Stage SystemRescue ISO contents for HTTP PXE boot";
|
||||
after = [
|
||||
"local-fs.target"
|
||||
"systemd-tmpfiles-setup.service"
|
||||
];
|
||||
|
||||
services = {
|
||||
fetch-debian-netboot = {
|
||||
description = "Download Debian ${debianRelease} netboot kernel and initrd for HTTP PXE boot";
|
||||
after = [
|
||||
"local-fs.target"
|
||||
"systemd-tmpfiles-setup.service"
|
||||
"network-online.target"
|
||||
];
|
||||
wants = [ "network-online.target" ];
|
||||
wantedBy = [ "multi-user.target" ];
|
||||
serviceConfig = {
|
||||
Type = "oneshot";
|
||||
ExecStart = fetchDebianNetboot;
|
||||
RemainAfterExit = true;
|
||||
};
|
||||
};
|
||||
|
||||
fetch-rocky-pxeboot = {
|
||||
description = "Download Rocky Linux ${rockyRelease} pxeboot kernel and initrd for HTTP PXE boot";
|
||||
after = [
|
||||
"local-fs.target"
|
||||
"systemd-tmpfiles-setup.service"
|
||||
"network-online.target"
|
||||
];
|
||||
wants = [ "network-online.target" ];
|
||||
wantedBy = [ "multi-user.target" ];
|
||||
serviceConfig = {
|
||||
Type = "oneshot";
|
||||
ExecStart = fetchRockyPxeboot;
|
||||
RemainAfterExit = true;
|
||||
};
|
||||
};
|
||||
|
||||
stage-systemrescue = {
|
||||
description = "Stage SystemRescue ISO contents for HTTP PXE boot";
|
||||
after = [
|
||||
"local-fs.target"
|
||||
"systemd-tmpfiles-setup.service"
|
||||
];
|
||||
wantedBy = [ "multi-user.target" ];
|
||||
serviceConfig = {
|
||||
Type = "oneshot";
|
||||
ExecStart = stageSystemRescue;
|
||||
};
|
||||
};
|
||||
};
|
||||
};
|
||||
|
||||
services.dnsmasq = {
|
||||
enable = true;
|
||||
settings = {
|
||||
# Disable DNS listener — only proxy DHCP is needed here.
|
||||
# Without this dnsmasq tries to bind port 53 which systemd-resolved
|
||||
# already owns, causing startup failure.
|
||||
port = 0;
|
||||
dhcp-range = [ "192.168.2.0,proxy" ];
|
||||
dhcp-match = [
|
||||
"set:ipxe,175"
|
||||
"set:efi64,option:client-arch,7"
|
||||
"set:efi64,option:client-arch,9"
|
||||
];
|
||||
dhcp-userclass = "set:ipxe,iPXE";
|
||||
dhcp-boot = [
|
||||
"tag:ipxe,tag:efi64,http://${vars.pxeServerIp}/boot.ipxe"
|
||||
"tag:ipxe,http://${vars.pxeServerIp}/boot.ipxe"
|
||||
"tag:efi64,ipxe.efi,,${vars.pxeServerIp}"
|
||||
"undionly.kpxe,,${vars.pxeServerIp}"
|
||||
];
|
||||
wantedBy = [ "multi-user.target" ];
|
||||
serviceConfig = {
|
||||
Type = "oneshot";
|
||||
ExecStart = stageSystemRescue;
|
||||
};
|
||||
};
|
||||
|
||||
networking.firewall.allowedTCPPorts = [ vars.ports.pxeBootHttp ];
|
||||
networking.firewall.allowedUDPPorts = [ vars.ports.pxeBootTftp 67 ];
|
||||
networking.firewall.allowedUDPPorts = [ vars.ports.pxeBootTftp ];
|
||||
}
|
||||
|
||||
@@ -1,101 +1,12 @@
|
||||
{ vars, lib, pkgs, ... }:
|
||||
{ vars, lib, ... }:
|
||||
|
||||
let
|
||||
poolName = lib.removePrefix "/" vars.storageRoot;
|
||||
|
||||
# For each NFS share subpath, generate every ancestor path so ZFS datasets
|
||||
# are created parent-first. e.g. "docker/config" → ["docker" "docker/config"]
|
||||
ancestors = path:
|
||||
let parts = lib.splitString "/" path;
|
||||
in lib.imap1 (i: _: lib.concatStringsSep "/" (lib.take i parts)) parts;
|
||||
|
||||
poolDatasets = lib.unique (
|
||||
lib.concatMap (share: ancestors share.subpath)
|
||||
(lib.filter builtins.isAttrs (lib.attrValues vars.nfsShares))
|
||||
);
|
||||
|
||||
# Generates /etc/exports lines for all nfsShares data entries (every
|
||||
# attrset value — excludes the bare `options` string). Both server and
|
||||
# ha-server export the same share set from different storage roots, so
|
||||
# this helper is the single source of truth for the export line format.
|
||||
mkNfsExports = storageRoot:
|
||||
lib.concatMapStrings
|
||||
(share: " ${storageRoot}/${share.subpath} ${vars.lanCidr}${vars.nfsShares.options}\n")
|
||||
(lib.filter builtins.isAttrs (lib.attrValues vars.nfsShares));
|
||||
in
|
||||
{
|
||||
imports = [
|
||||
../beszel/enable-agent.nix
|
||||
../services/zfs/enable-service.nix
|
||||
];
|
||||
|
||||
boot.zfs.extraPools = [ poolName ];
|
||||
|
||||
# On a fresh image deploy the data disk (scsi1) starts blank — no pool
|
||||
# exists yet, so zfs-import-tank.service would spin for 60 s and fail.
|
||||
# This service runs first: if the pool is already present it exits instantly;
|
||||
# otherwise it creates it (with all required datasets) so the standard
|
||||
# import service finds it ready on the very first boot.
|
||||
systemd.services."zfs-init-${poolName}" = {
|
||||
description = "Initialize '${poolName}' ZFS pool on first boot if not present";
|
||||
wantedBy = [ "zfs-import-${poolName}.service" ];
|
||||
before = [ "zfs-import-${poolName}.service" ];
|
||||
after = [ "systemd-udev-settle.service" ];
|
||||
unitConfig.DefaultDependencies = false;
|
||||
serviceConfig = {
|
||||
Type = "oneshot";
|
||||
RemainAfterExit = true;
|
||||
};
|
||||
path = [ pkgs.zfs_unstable ];
|
||||
script = ''
|
||||
# Already imported — nothing to do.
|
||||
if zpool list "${poolName}" >/dev/null 2>&1; then
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Locate the data disk first — used for both the fallback import
|
||||
# attempt and, only if the disk is genuinely blank, pool creation.
|
||||
DATA_DISK=""
|
||||
for candidate in /dev/disk/by-id/scsi-*drive-scsi1; do
|
||||
[[ "$candidate" == *-part* ]] && continue
|
||||
[ -b "$candidate" ] && DATA_DISK="$candidate" && break
|
||||
done
|
||||
|
||||
if [ -z "$DATA_DISK" ]; then
|
||||
echo "zfs-init-${poolName}: no data disk found (expected /dev/disk/by-id/scsi-*drive-scsi1)" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Try importing via the by-id symlink directory first (normal path),
|
||||
# then fall back to scanning the disk directly. The two-step exists
|
||||
# because of a udev race: systemd-udev-settle.service can clear before
|
||||
# /dev/disk/by-id/ entries are fully populated, causing the first
|
||||
# import to fail even when the pool is intact on the disk.
|
||||
if zpool import -d /dev/disk/by-id -N "${poolName}" 2>/dev/null; then
|
||||
exit 0
|
||||
fi
|
||||
if zpool import -d "$DATA_DISK" -N "${poolName}" 2>/dev/null; then
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Both import attempts failed. Before creating a new pool, verify the
|
||||
# disk is genuinely blank — if ZFS label metadata is present the import
|
||||
# failed for some other reason and we must not clobber existing data.
|
||||
if zdb -l "$DATA_DISK" 2>/dev/null | grep -q "name: '${poolName}'"; then
|
||||
echo "zfs-init-${poolName}: $DATA_DISK has ZFS pool '${poolName}' metadata but import failed — refusing to overwrite existing data. Run 'zpool import -d $DATA_DISK ${poolName}' manually to investigate." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Disk is genuinely blank: create the pool. -f is intentionally
|
||||
# omitted so that if we somehow reach this point with an existing pool
|
||||
# on the disk, zpool refuses rather than silently destroying data.
|
||||
echo "zfs-init-${poolName}: creating pool on $DATA_DISK"
|
||||
zpool create "${poolName}" "$DATA_DISK"
|
||||
${lib.concatMapStrings (ds: ''
|
||||
zfs create "${poolName}/${ds}"
|
||||
'') poolDatasets}
|
||||
'';
|
||||
};
|
||||
boot.zfs.extraPools = [ (lib.removePrefix "/" vars.storageRoot) ];
|
||||
|
||||
systemd.services.nfs-server = {
|
||||
after = [ "zfs-mount.service" ];
|
||||
@@ -104,12 +15,14 @@ in
|
||||
|
||||
services.nfs.server = {
|
||||
enable = true;
|
||||
exports = mkNfsExports vars.storageRoot;
|
||||
exports = ''
|
||||
${vars.storageRoot}/${vars.nfsShares.dockerConfig.subpath} ${vars.lanCidr}(rw,sync,no_subtree_check,no_root_squash)
|
||||
${vars.storageRoot}/${vars.nfsShares.dockerVolumes.subpath} ${vars.lanCidr}(rw,sync,no_subtree_check,no_root_squash)
|
||||
${vars.storageRoot}/${vars.nfsShares.dockerDatabases.subpath} ${vars.lanCidr}(rw,sync,no_subtree_check,no_root_squash)
|
||||
${vars.storageRoot}/${vars.nfsShares.nextcloudData.subpath} ${vars.lanCidr}(rw,sync,no_subtree_check,no_root_squash)
|
||||
${vars.storageRoot}/${vars.nfsShares.raspiVolumes.subpath} ${vars.lanCidr}(rw,sync,no_subtree_check,no_root_squash)
|
||||
'';
|
||||
};
|
||||
|
||||
# mountd (20048) is needed for showmount/NFSv3 mount protocol — without it
|
||||
# clients can reach portmapper (111) and get the mountd port back, then
|
||||
# time out trying to connect to it. All three ports need TCP and UDP.
|
||||
networking.firewall.allowedTCPPorts = [ vars.ports.nfsRpcbind vars.ports.nfsd vars.ports.nfsMountd ];
|
||||
networking.firewall.allowedUDPPorts = [ vars.ports.nfsRpcbind vars.ports.nfsd vars.ports.nfsMountd ];
|
||||
networking.firewall.allowedTCPPorts = [ vars.ports.nfsRpcbind vars.ports.nfsd ];
|
||||
}
|
||||
|
||||
@@ -0,0 +1,21 @@
|
||||
{ ... }:
|
||||
|
||||
{
|
||||
imports = [
|
||||
../tailscale/exit-node.nix
|
||||
];
|
||||
|
||||
# "server", not "both": this build type only ever advertises itself as an
|
||||
# exit node (see ../tailscale/exit-node.nix) -- it doesn't advertise LAN
|
||||
# subnet routes, so it doesn't need the "client"-side loose reverse-path
|
||||
# filtering that "both" would also turn on. Deliberately left unbundled
|
||||
# from LAN-subnet-route advertisement so this build type stays valid on
|
||||
# every platform, including linode (a remote VPS with no network path to
|
||||
# the home LAN at all).
|
||||
services.tailscale.useRoutingFeatures = "server";
|
||||
|
||||
# Forwarded exit-node traffic arrives on tailscale0 already
|
||||
# tailscale-authenticated -- the firewall's normal per-port allow-list
|
||||
# would otherwise drop it. Standard NixOS/Tailscale exit-node guidance.
|
||||
networking.firewall.trustedInterfaces = [ "tailscale0" ];
|
||||
}
|
||||
@@ -1,44 +0,0 @@
|
||||
{ vars, ... }:
|
||||
|
||||
{
|
||||
imports = [
|
||||
../tailscale/subnet-router.nix
|
||||
../tailscale/ts-dns-forwarder.nix
|
||||
../beszel/enable-agent.nix
|
||||
];
|
||||
|
||||
# "server", not "both": this build type advertises LAN subnet routes but
|
||||
# doesn't use another tailscale exit node itself, so it doesn't need the
|
||||
# "client"-side loose reverse-path filtering that "both" would also enable.
|
||||
# Deliberately kept explicit here (not just relying on subnet-router.nix's
|
||||
# own setting) so the intent is clear at the build-type level.
|
||||
services.tailscale.useRoutingFeatures = "server";
|
||||
|
||||
# Advertise the LAN subnet so Tailscale peers can route back to LAN machines.
|
||||
# Must also be approved in the Tailscale admin console (Machines → Edit route settings).
|
||||
services.tailscale.extraUpFlags = [ "--advertise-routes=${vars.lanCidr}" ];
|
||||
|
||||
networking.firewall = {
|
||||
# Forwarded subnet-router traffic arrives on tailscale0 already
|
||||
# tailscale-authenticated -- the firewall's normal per-port allow-list
|
||||
# would otherwise drop it. Standard NixOS/Tailscale subnet-router guidance.
|
||||
trustedInterfaces = [ "tailscale0" ];
|
||||
|
||||
# SNAT LAN traffic going into Tailscale so the remote peer sees it as
|
||||
# coming from this router's Tailscale IP rather than a raw LAN IP.
|
||||
# Without this, Tailscale drops forwarded packets whose source is not a
|
||||
# recognised Tailscale address.
|
||||
#
|
||||
# We target POSTROUTING directly (always-existing built-in chain) rather
|
||||
# than nixos-nat-post: extraCommands runs after the old nixos-nat-post is
|
||||
# deleted but before the new one is created, so -A nixos-nat-post silently
|
||||
# fails. The -C check makes the rule idempotent across firewall reloads.
|
||||
extraCommands = ''
|
||||
iptables -t nat -C POSTROUTING -s ${vars.lanCidr} -o tailscale0 -j MASQUERADE 2>/dev/null || \
|
||||
iptables -t nat -A POSTROUTING -s ${vars.lanCidr} -o tailscale0 -j MASQUERADE
|
||||
'';
|
||||
extraStopCommands = ''
|
||||
iptables -t nat -D POSTROUTING -s ${vars.lanCidr} -o tailscale0 -j MASQUERADE 2>/dev/null || true
|
||||
'';
|
||||
};
|
||||
}
|
||||
@@ -1,8 +0,0 @@
|
||||
{ ... }:
|
||||
|
||||
{
|
||||
imports = [
|
||||
../tor/enable-relay.nix
|
||||
../beszel/enable-agent.nix
|
||||
];
|
||||
}
|
||||
@@ -1,30 +0,0 @@
|
||||
{ pkgs, ... }: {
|
||||
# Defines the SSH host key as a clan vars generator so that:
|
||||
# - `clan vars generate <target>` creates and encrypts the key pair
|
||||
# - The private key lives at vars/per-machine/<target>/openssh/ssh_host_ed25519_key/secret
|
||||
# (sops binary-encrypted, admin-key-only; decrypted by the build script)
|
||||
# - The public key lives at vars/per-machine/<target>/openssh/ssh_host_ed25519_key.pub/value
|
||||
# (plaintext; used by sync-host-keys.sh to derive the sops age fingerprint)
|
||||
#
|
||||
# neededFor = "activation" means clan's deployment tool would upload this
|
||||
# before running nixos-rebuild/nixos-install (for VM/baremetal via
|
||||
# nixos-anywhere). For lxc-* hosts, the build script bakes it into the
|
||||
# tarball directly via NIXOS_HOST_KEYS_DIR -- the neededFor value here
|
||||
# simply ensures it is NOT mapped to sops.secrets (which would try to
|
||||
# decrypt it at runtime as a regular service secret, which is wrong: the
|
||||
# SSH host key reaches the container via the tarball, not sops).
|
||||
clan.core.vars.generators.openssh = {
|
||||
files."ssh_host_ed25519_key" = {
|
||||
secret = true;
|
||||
neededFor = "activation";
|
||||
};
|
||||
files."ssh_host_ed25519_key.pub" = {
|
||||
secret = false;
|
||||
neededFor = "activation";
|
||||
};
|
||||
runtimeInputs = [ pkgs.openssh ];
|
||||
script = ''
|
||||
ssh-keygen -t ed25519 -N "" -C "" -f "$out/ssh_host_ed25519_key"
|
||||
'';
|
||||
};
|
||||
}
|
||||
@@ -1,7 +1,50 @@
|
||||
_:
|
||||
{ config, pkgs, lib, vars, ... }:
|
||||
|
||||
let
|
||||
# Flake attribute names are now <platform>-<buildtype> (e.g. proxmox-docker)
|
||||
# and no longer match networking.hostName, since a host's hostname stays
|
||||
# fixed while the platform backing it can change. Each nixosConfiguration
|
||||
# stamps its own active target name into /etc/flake-target at build time.
|
||||
mySwitchCmd = ''
|
||||
sudo nixos-rebuild switch \
|
||||
--no-write-lock-file \
|
||||
--refresh \
|
||||
--flake git+https://${vars.lanDomain}/beatzaplenty/nixos.git#$(cat /etc/flake-target)
|
||||
'';
|
||||
myTestCmd = ''
|
||||
sudo nixos-rebuild test \
|
||||
--no-write-lock-file \
|
||||
--refresh \
|
||||
--flake git+https://${vars.lanDomain}/beatzaplenty/nixos.git#$(cat /etc/flake-target)
|
||||
'';
|
||||
|
||||
# lxc-* hosts pre-seed their SSH host key at build time (see
|
||||
# modules/platforms/lxc.nix) so sops-nix's .sops.yaml recipient matches on
|
||||
# first boot -- without it, secrets permanently fail to decrypt (see that
|
||||
# file's comment for the confirmed failure). That requires --impure plus
|
||||
# NIXOS_HOST_KEYS_DIR pointing at the repo's host-keys/ dir, same pattern
|
||||
# docs/auto-installer.md uses for the installer ISO. A function, not a
|
||||
# shellAlias, since the target name has to interpolate into the middle of
|
||||
# the flake attribute path, not just append after it. Must be run from the
|
||||
# repo root, same as every other host-keys/ command in this repo.
|
||||
buildImageFn = ''
|
||||
buildImage() {
|
||||
if [ -z "$1" ]; then
|
||||
echo "usage: buildImage <flake-target> (e.g. lxc-docker)" >&2
|
||||
return 1
|
||||
fi
|
||||
NIXOS_HOST_KEYS_DIR="$(pwd)/host-keys" nix build --impure \
|
||||
".#nixosConfigurations.$1.config.system.build.tarball"
|
||||
}
|
||||
'';
|
||||
in
|
||||
{
|
||||
# Switch-nix, Test-nix, and buildImage are defined system-wide in
|
||||
# modules/common/configuration.nix so all users (including IPA accounts)
|
||||
# get them. Add any Home-Manager-only per-user shell config here.
|
||||
programs.bash = {
|
||||
enable = true;
|
||||
shellAliases = {
|
||||
"Switch-nix" = mySwitchCmd;
|
||||
"Test-nix" = myTestCmd;
|
||||
};
|
||||
initExtra = buildImageFn;
|
||||
};
|
||||
}
|
||||
|
||||
@@ -1,45 +1,17 @@
|
||||
{ config, lib, pkgs, vars, ... }:
|
||||
|
||||
let
|
||||
switchCmd = ''
|
||||
sudo nixos-rebuild switch \
|
||||
--no-write-lock-file \
|
||||
--refresh \
|
||||
--flake git+https://${vars.lanDomain}/beatzaplenty/nixos.git#$(cat /etc/flake-target)
|
||||
'';
|
||||
testCmd = ''
|
||||
sudo nixos-rebuild test \
|
||||
--no-write-lock-file \
|
||||
--refresh \
|
||||
--flake git+https://${vars.lanDomain}/beatzaplenty/nixos.git#$(cat /etc/flake-target)
|
||||
'';
|
||||
buildImageFn = ''
|
||||
buildImage() {
|
||||
if [ -z "$1" ]; then
|
||||
echo "usage: buildImage <flake-target> (e.g. lxc-docker)" >&2
|
||||
return 1
|
||||
fi
|
||||
NIXOS_HOST_KEYS_DIR="$(pwd)/host-keys" nix build --impure \
|
||||
".#nixosConfigurations.$1.config.system.build.tarball"
|
||||
}
|
||||
'';
|
||||
in
|
||||
{
|
||||
imports = [
|
||||
./set-locale.nix
|
||||
../ipa/client.nix
|
||||
];
|
||||
imports =
|
||||
[
|
||||
# Include the results of the hardware scan.
|
||||
# ./hardware-configuration.nix
|
||||
./set-locale.nix
|
||||
];
|
||||
# Use the GRUB 2 boot loader.
|
||||
# boot.loader.grub.enable = true;
|
||||
#boot.loader.grub.device = "/dev/sda"; # or "nodev" for efi only
|
||||
|
||||
# System-wide shell config so all users (including IPA accounts) get the
|
||||
# same management aliases as the local nixos user's Home Manager provides.
|
||||
programs.bash = {
|
||||
shellAliases = {
|
||||
"Switch-nix" = switchCmd;
|
||||
"Test-nix" = testCmd;
|
||||
};
|
||||
interactiveShellInit = buildImageFn;
|
||||
};
|
||||
networking.networkmanager.enable = true;
|
||||
networking.networkmanager.enable = true; # Easiest to use and most distros use this by default.
|
||||
|
||||
# Recommended over the true default (bypasses ZFS's own import safeguards)
|
||||
# per the option's own docs; matches hosts/docker/host.nix and
|
||||
@@ -59,7 +31,6 @@ in
|
||||
btop
|
||||
git
|
||||
gcr
|
||||
jq
|
||||
];
|
||||
|
||||
# Secrets shared by every host, decrypted at activation via each host's
|
||||
@@ -89,32 +60,23 @@ in
|
||||
!include ${config.sops.templates."nix-github-token.conf".path}
|
||||
'';
|
||||
|
||||
users = {
|
||||
# With mutableUsers = false, update-users-groups.pl enforces hashedPasswordFile
|
||||
# on every activation regardless of whether the account already exists in
|
||||
# /etc/shadow. The default (true) only applies hashedPasswordFile to newly-
|
||||
# created accounts — which means a freshly-built proxmox disk image (where
|
||||
# activation runs without a usable sops key, so both accounts land in shadow
|
||||
# with ‘!’) will never have its passwords fixed by subsequent boots.
|
||||
mutableUsers = false;
|
||||
#Set root password
|
||||
users.users.root = {
|
||||
hashedPasswordFile = config.sops.secrets."root-hashedPassword".path;
|
||||
};
|
||||
|
||||
users.root = {
|
||||
hashedPasswordFile = config.sops.secrets."root-hashedPassword".path;
|
||||
};
|
||||
|
||||
users.${vars.primaryUser} = {
|
||||
isNormalUser = true;
|
||||
extraGroups = [ "wheel" ]; # Enable ‘sudo’ for the user.
|
||||
packages = with pkgs; [
|
||||
tree
|
||||
];
|
||||
hashedPasswordFile = config.sops.secrets."nixos-hashedPassword".path;
|
||||
openssh.authorizedKeys.keys = [
|
||||
vars.adminSshKey
|
||||
"ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAICMJhrfFayLBG+gWtO6oAvgambw5nWWgztiTFEaaaVRH debian@surface"
|
||||
"ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIGygkCljN6uKpdJbHTOQtn8ZnH+wKXDLAwrDFbLrE/65 nixos@nixos"
|
||||
];
|
||||
};
|
||||
# Define a user account. Don't forget to set a password with ‘passwd’.
|
||||
users.users.${vars.primaryUser} = {
|
||||
isNormalUser = true;
|
||||
extraGroups = [ "wheel" ]; # Enable ‘sudo’ for the user.
|
||||
packages = with pkgs; [
|
||||
tree
|
||||
];
|
||||
hashedPasswordFile = config.sops.secrets."nixos-hashedPassword".path;
|
||||
openssh.authorizedKeys.keys = [
|
||||
vars.adminSshKey
|
||||
"ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAICMJhrfFayLBG+gWtO6oAvgambw5nWWgztiTFEaaaVRH debian@surface"
|
||||
];
|
||||
};
|
||||
|
||||
|
||||
|
||||
@@ -1,35 +0,0 @@
|
||||
# Shared activation-script logic to preserve the SSH host key across
|
||||
# nixos-rebuild on platforms that embed the key via environment.etc (lxc and
|
||||
# proxmox). When NIXOS_HOST_KEYS_DIR is not set the key is absent from
|
||||
# environment.etc, and NixOS's etc activation removes any /etc file not in
|
||||
# the new generation — which would destroy the live key and break sops-nix
|
||||
# decryption permanently. These scripts save the key to /run before etc
|
||||
# removes it, then restore it afterward.
|
||||
#
|
||||
# Explicit deps enforce the correct ordering: without them the topological
|
||||
# sort places preserveSshHostKey after etc (confirmed live on lxc-tor-relay:
|
||||
# position 7 vs etc's position 5), so the key is gone before it can be saved.
|
||||
_: {
|
||||
system.activationScripts = {
|
||||
preserveSshHostKey = ''
|
||||
if [ -f /etc/ssh/ssh_host_ed25519_key ]; then
|
||||
cp /etc/ssh/ssh_host_ed25519_key /run/sshd-host-key-preserve.tmp
|
||||
cp /etc/ssh/ssh_host_ed25519_key.pub /run/sshd-host-key-preserve.pub.tmp
|
||||
fi
|
||||
'';
|
||||
|
||||
restoreSshHostKey = {
|
||||
deps = [ "etc" ];
|
||||
text = ''
|
||||
if [ ! -f /etc/ssh/ssh_host_ed25519_key ] && [ -f /run/sshd-host-key-preserve.tmp ]; then
|
||||
install -m 0600 /run/sshd-host-key-preserve.tmp /etc/ssh/ssh_host_ed25519_key
|
||||
install -m 0644 /run/sshd-host-key-preserve.pub.tmp /etc/ssh/ssh_host_ed25519_key.pub
|
||||
fi
|
||||
rm -f /run/sshd-host-key-preserve.tmp /run/sshd-host-key-preserve.pub.tmp
|
||||
'';
|
||||
};
|
||||
|
||||
etc = { deps = [ "preserveSshHostKey" ]; };
|
||||
setupSecrets = { deps = [ "restoreSshHostKey" ]; };
|
||||
};
|
||||
}
|
||||
@@ -1,87 +0,0 @@
|
||||
{ vars, ... }:
|
||||
|
||||
{
|
||||
# ZFS RAID0 (striped, no redundancy) root pool for the bare-metal gui
|
||||
# host — two disks, each contributing its own top-level vdev. disko's
|
||||
# zpool `mode` defaults to "" (plain stripe) when left unset, which is
|
||||
# what gives RAID0 semantics here rather than mirror/raidz.
|
||||
#
|
||||
# Device paths are placeholders until the real hardware profile lands —
|
||||
# fill in vars.guiRootDisk1/guiRootDisk2 (stable /dev/disk/by-id/...
|
||||
# paths, not /dev/sdX) before running disko against real hardware. Swap
|
||||
# is deliberately left out for now — sizing that sensibly needs the
|
||||
# box's actual RAM size, which comes with the hardware profile too.
|
||||
#
|
||||
# Not yet imported anywhere: this awaits the new bare-metal platform
|
||||
# module (alongside modules/boot/efi.nix for systemd-boot, matching
|
||||
# modules/platforms/proxmox.nix's pattern) once the hardware config is
|
||||
# in hand.
|
||||
disko.devices = {
|
||||
disk = {
|
||||
disk1 = {
|
||||
type = "disk";
|
||||
device = vars.guiRootDisk1;
|
||||
|
||||
content = {
|
||||
type = "gpt";
|
||||
|
||||
partitions = {
|
||||
esp = {
|
||||
priority = 1;
|
||||
name = "ESP";
|
||||
size = "512M";
|
||||
type = "EF00";
|
||||
|
||||
content = {
|
||||
type = "filesystem";
|
||||
format = "vfat";
|
||||
mountpoint = "/boot";
|
||||
mountOptions = [ "umask=0077" ];
|
||||
};
|
||||
};
|
||||
|
||||
zfs = {
|
||||
size = "100%";
|
||||
|
||||
content = {
|
||||
type = "zfs";
|
||||
pool = "rpool";
|
||||
};
|
||||
};
|
||||
};
|
||||
};
|
||||
};
|
||||
|
||||
disk2 = {
|
||||
type = "disk";
|
||||
device = vars.guiRootDisk2;
|
||||
|
||||
content = {
|
||||
type = "gpt";
|
||||
|
||||
partitions = {
|
||||
zfs = {
|
||||
size = "100%";
|
||||
|
||||
content = {
|
||||
type = "zfs";
|
||||
pool = "rpool";
|
||||
};
|
||||
};
|
||||
};
|
||||
};
|
||||
};
|
||||
};
|
||||
|
||||
zpool.rpool = {
|
||||
type = "zpool";
|
||||
|
||||
rootFsOptions = {
|
||||
compression = "zstd";
|
||||
"com.sun:auto-snapshot" = "false";
|
||||
};
|
||||
mountpoint = "/";
|
||||
options.ashift = "12";
|
||||
};
|
||||
};
|
||||
}
|
||||
@@ -1,17 +1,21 @@
|
||||
{ lib, pkgs, vars, ... }:
|
||||
{ pkgs, ... }:
|
||||
|
||||
{
|
||||
# virtualisation.docker.enable = true;
|
||||
virtualisation.docker = {
|
||||
enable = true;
|
||||
package = pkgs.docker;
|
||||
# listenOptions = [
|
||||
# "unix:///var/run/docker.sock"
|
||||
# "tcp://0.0.0.0:2375"
|
||||
#];
|
||||
|
||||
# daemon.settings = {
|
||||
# metrics-addr = "0.0.0.0:9323";
|
||||
# experimental = true;
|
||||
# };
|
||||
};
|
||||
# Pin the docker group GID to match the IPA "docker-access" group so that
|
||||
# IPA group membership alone grants access to the Docker socket. Any user
|
||||
# whose supplementary groups (resolved by SSSD from IPA) include GID
|
||||
# vars.dockerAccessGid will pass the socket group-permission check without
|
||||
# any per-host users.groups.docker.members entry.
|
||||
users.groups.docker.gid = lib.mkForce vars.dockerAccessGid;
|
||||
users.users.${vars.primaryUser}.extraGroups = [ "docker" ];
|
||||
|
||||
environment.systemPackages = with pkgs; [
|
||||
docker-compose
|
||||
docker-buildx
|
||||
|
||||
@@ -1,84 +1,65 @@
|
||||
{ config, lib, pkgs, vars, ... }:
|
||||
|
||||
let
|
||||
# `x-systemd.automount` never works inside a Linux container (LXC
|
||||
# included, regardless of privilege) -- confirmed live on lxc-docker:
|
||||
# systemd logs "Starting of <unit>.automount unsupported" for every
|
||||
# share and never mounts them. Mount eagerly there instead, with
|
||||
# `nofail` so a boot with the NFS server unreachable doesn't hang
|
||||
# (the VM platforms rely on automount itself to get that same
|
||||
# non-blocking behavior, so they don't need `nofail` too).
|
||||
automountOpts = if config.boot.isContainer then [ "nofail" ] else [ "x-systemd.automount" ];
|
||||
|
||||
# A bare hostname here never resolves reliably: systemd-resolved only
|
||||
# ever tries LLMNR for single-label names (never DNS, regardless of any
|
||||
# configured search domain), and a *global* search domain (the first fix
|
||||
# attempted here) backfires worse -- confirmed live on lxc-docker, adding
|
||||
# `networking.search` made systemd-resolved prioritize its domain-matched
|
||||
# but server-less global scope over eth0's correctly-configured one for
|
||||
# every "*.sweet.home" query, silently sending them to public fallback
|
||||
# DNS instead. `resolvectl query --interface=eth0 server.sweet.home`
|
||||
# resolved fine throughout, proving the LAN DNS server was never the
|
||||
# problem -- only the ambient, unqualified device string was. Using the
|
||||
# FQDN directly sidesteps all of that, matching the pattern
|
||||
# ../raspi/mount-data.nix already uses for the same reason.
|
||||
nfsServer = "${vars.nfsServerHost}.${vars.homeDomain}";
|
||||
in
|
||||
{
|
||||
fileSystems = {
|
||||
${vars.nfsShares.dockerConfig.mountpoint} = {
|
||||
device = "${nfsServer}:${vars.storageRoot}/${vars.nfsShares.dockerConfig.subpath}";
|
||||
device = "${vars.nfsServerHost}:${vars.storageRoot}/${vars.nfsShares.dockerConfig.subpath}";
|
||||
fsType = "nfs";
|
||||
|
||||
options = [
|
||||
"nfsvers=4.2"
|
||||
"_netdev"
|
||||
"x-systemd.automount"
|
||||
"noatime"
|
||||
] ++ automountOpts;
|
||||
];
|
||||
};
|
||||
|
||||
${vars.nfsShares.dockerDatabases.mountpoint} = {
|
||||
device = "${nfsServer}:${vars.storageRoot}/${vars.nfsShares.dockerDatabases.subpath}";
|
||||
device = "${vars.nfsServerHost}:${vars.storageRoot}/${vars.nfsShares.dockerDatabases.subpath}";
|
||||
fsType = "nfs";
|
||||
|
||||
options = [
|
||||
"nfsvers=4.2"
|
||||
"_netdev"
|
||||
"x-systemd.automount"
|
||||
"noatime"
|
||||
] ++ automountOpts;
|
||||
];
|
||||
};
|
||||
|
||||
${vars.nfsShares.dockerVolumes.mountpoint} = {
|
||||
device = "${nfsServer}:${vars.storageRoot}/${vars.nfsShares.dockerVolumes.subpath}";
|
||||
device = "${vars.nfsServerHost}:${vars.storageRoot}/${vars.nfsShares.dockerVolumes.subpath}";
|
||||
fsType = "nfs";
|
||||
|
||||
options = [
|
||||
"nfsvers=4.2"
|
||||
"_netdev"
|
||||
"x-systemd.automount"
|
||||
"noatime"
|
||||
] ++ automountOpts;
|
||||
];
|
||||
};
|
||||
|
||||
${vars.nfsShares.nextcloudData.mountpoint} = {
|
||||
device = "${nfsServer}:${vars.storageRoot}/${vars.nfsShares.nextcloudData.subpath}";
|
||||
device = "${vars.nfsServerHost}:${vars.storageRoot}/${vars.nfsShares.nextcloudData.subpath}";
|
||||
fsType = "nfs";
|
||||
|
||||
options = [
|
||||
"nfsvers=4.2"
|
||||
"_netdev"
|
||||
"x-systemd.automount"
|
||||
"noatime"
|
||||
] ++ automountOpts;
|
||||
];
|
||||
};
|
||||
|
||||
${vars.nfsShares.raspiVolumes.mountpoint} = {
|
||||
device = "${nfsServer}:${vars.storageRoot}/${vars.nfsShares.raspiVolumes.subpath}";
|
||||
device = "${vars.nfsServerHost}:${vars.storageRoot}/${vars.nfsShares.raspiVolumes.subpath}";
|
||||
fsType = "nfs";
|
||||
|
||||
options = [
|
||||
"nfsvers=4.2"
|
||||
"_netdev"
|
||||
"x-systemd.automount"
|
||||
"noatime"
|
||||
] ++ automountOpts;
|
||||
];
|
||||
};
|
||||
};
|
||||
}
|
||||
|
||||
@@ -1,129 +0,0 @@
|
||||
# Cluster-wide HA config shared by both ha-server nodes.
|
||||
#
|
||||
# Covers everything that is identical on both nodes and references cluster
|
||||
# topology (node IPs, hostnames, DRBD resource). Per-node identity
|
||||
# (hostname, static IP, stateVersion) lives in hosts/ha-server-{1,2}/host.nix.
|
||||
#
|
||||
# Corosync authkey:
|
||||
# /etc/corosync/authkey (mode 0400) is managed by sops-nix below.
|
||||
# Bootstrap: run scripts/ha/cluster-init.sh on node1 to generate the key,
|
||||
# then encrypt it with: sops -e --input-type binary /etc/corosync/authkey > secrets/ha-corosync-authkey
|
||||
# Both host keys must be registered via sync-host-keys.sh first so both nodes can decrypt it.
|
||||
#
|
||||
# DRBD fencing:
|
||||
# resource-only with crm-fence-peer.sh: DRBD calls the Pacemaker-aware
|
||||
# crm-fence-peer.sh handler before promoting. The handler checks the CIB
|
||||
# to confirm the peer's DRBD resource is stopped and returns 7 (successfully
|
||||
# fenced), allowing safe promotion without requiring power-fencing (STONITH).
|
||||
# The unfence handler crm-unfence-peer.sh clears the outdate flag when the
|
||||
# peer reconnects. This is the correct setting for Pacemaker+DRBD clusters
|
||||
# with STONITH disabled; crm-fence-peer.sh replaces the need for a separate
|
||||
# STONITH device during the testing phase. Switch to resource-and-stonith
|
||||
# once the fence_pve_ssh STONITH resource is active (see
|
||||
# scripts/ha/cluster-enable-stonith.sh).
|
||||
{ lib, vars, ... }:
|
||||
{
|
||||
# Root SSH access — same key set as nixos user so all admin keys can reach root.
|
||||
users.users.root.openssh.authorizedKeys.keys = [
|
||||
vars.adminSshKey
|
||||
"ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAICMJhrfFayLBG+gWtO6oAvgambw5nWWgztiTFEaaaVRH debian@surface"
|
||||
"ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIGygkCljN6uKpdJbHTOQtn8ZnH+wKXDLAwrDFbLrE/65 nixos@nixos"
|
||||
];
|
||||
|
||||
# Passwordless sudo for wheel — operator SSHes as nixos and uses sudo for
|
||||
# cluster management commands (drbdadm, crm*, pcs, etc.)
|
||||
security.sudo.wheelNeedsPassword = lib.mkForce false;
|
||||
|
||||
# DRBD lock-file directory (drbd-utils checks for it; missing → harmless but noisy warnings).
|
||||
systemd.tmpfiles.rules = [ "d /var/lib/drbd 0750 root root -" ];
|
||||
|
||||
services.drbd = {
|
||||
enable = true;
|
||||
config = ''
|
||||
global {
|
||||
usage-count yes;
|
||||
}
|
||||
|
||||
common {
|
||||
net {
|
||||
protocol C;
|
||||
ping-int 1;
|
||||
verify-alg sha256;
|
||||
after-sb-0pri discard-zero-changes;
|
||||
after-sb-1pri discard-secondary;
|
||||
}
|
||||
disk {
|
||||
fencing resource-only;
|
||||
}
|
||||
handlers {
|
||||
fence-peer "/run/current-system/sw/lib/drbd/crm-fence-peer.sh";
|
||||
unfence-peer "/run/current-system/sw/lib/drbd/crm-unfence-peer.sh";
|
||||
}
|
||||
}
|
||||
|
||||
resource ha-data {
|
||||
volume 0 {
|
||||
device /dev/drbd0;
|
||||
disk /dev/sdb;
|
||||
meta-disk internal;
|
||||
}
|
||||
|
||||
on ${vars.haServer1Host} {
|
||||
address ${vars.haServer1StorageIp}:${toString vars.ports.haServerDrbd};
|
||||
}
|
||||
|
||||
on ${vars.haServer2Host} {
|
||||
address ${vars.haServer2StorageIp}:${toString vars.ports.haServerDrbd};
|
||||
}
|
||||
}
|
||||
'';
|
||||
};
|
||||
|
||||
# /etc/corosync/authkey — sops binary secret, identical on both nodes.
|
||||
# Decryptable by both ha-server host keys (added by sync-host-keys.sh).
|
||||
sops.secrets.corosync_authkey = {
|
||||
sopsFile = ../../secrets/ha-corosync-authkey;
|
||||
format = "binary";
|
||||
path = "/etc/corosync/authkey";
|
||||
mode = "0400";
|
||||
restartUnits = [ "corosync.service" ];
|
||||
};
|
||||
|
||||
# NixOS common config enables NetworkManager by default; HA cluster nodes
|
||||
# need stable static IPs with predictable interface names — NM is not suitable.
|
||||
networking.networkmanager.enable = lib.mkForce false;
|
||||
|
||||
# services.corosync.enable is set by modules/ha/pacemaker-stack.nix.
|
||||
services.corosync = {
|
||||
clusterName = "ha-cluster";
|
||||
nodelist = [
|
||||
{ nodeid = 1; name = vars.haServer1Host; ring_addrs = [ vars.haServer1StorageIp ]; }
|
||||
{ nodeid = 2; name = vars.haServer2Host; ring_addrs = [ vars.haServer2StorageIp ]; }
|
||||
];
|
||||
};
|
||||
|
||||
networking.firewall = {
|
||||
allowedTCPPorts = [
|
||||
vars.ports.haServerIscsi
|
||||
vars.ports.haServerPacemakerRemoted
|
||||
vars.ports.haServerPcsd
|
||||
vars.ports.haServerDrbd
|
||||
vars.ports.nfsRpcbind
|
||||
vars.ports.nfsd
|
||||
vars.ports.nfsMountd
|
||||
];
|
||||
allowedUDPPorts = [
|
||||
vars.ports.haServerCorosync1
|
||||
vars.ports.haServerCorosync2
|
||||
vars.ports.haServerCorosyncCrypto
|
||||
vars.ports.nfsRpcbind
|
||||
vars.ports.nfsd
|
||||
vars.ports.nfsMountd
|
||||
];
|
||||
extraCommands = ''
|
||||
iptables -A INPUT -s ${vars.haServer1Ip}/32 -j ACCEPT
|
||||
iptables -A INPUT -s ${vars.haServer2Ip}/32 -j ACCEPT
|
||||
iptables -A INPUT -s ${vars.haStorageCidr} -j ACCEPT
|
||||
'';
|
||||
};
|
||||
}
|
||||
@@ -1,99 +0,0 @@
|
||||
# LIO iSCSI target service (targetctl) for NixOS HA clusters.
|
||||
#
|
||||
# Provides the targetctl.service that saves/restores LIO configuration from
|
||||
# /etc/target/saveconfig.json. Pacemaker manages this service via its
|
||||
# systemd resource agent (class="systemd" type="targetctl").
|
||||
#
|
||||
# Why ExecStop is not simply "targetctl save":
|
||||
# targetctl save writes the LIO config to JSON but does NOT remove the LIO
|
||||
# target from the kernel's configfs. As a result, any fileio backing store
|
||||
# that LIO has open (e.g. iscsi-lun.img on an XFS-over-DRBD filesystem)
|
||||
# stays referenced in the kernel. The subsequent XFS umount from the
|
||||
# Filesystem OCF resource then returns EBUSY and either hangs for the full
|
||||
# op-stop timeout or fails outright, blocking the entire failover.
|
||||
#
|
||||
# The ExecStop script here additionally tears down the kernel LIO state
|
||||
# via rtslib_fb after saving, so the backing-store file descriptor is
|
||||
# released and umount succeeds immediately.
|
||||
#
|
||||
# Empty-config guard:
|
||||
# The save step is skipped when no iSCSI targets are currently active.
|
||||
# This prevents the secondary node (where LIO was never started) from
|
||||
# overwriting a valid saveconfig.json with an empty one when Pacemaker
|
||||
# stops the iscsi-target resource as part of a failover or cleanup.
|
||||
{ pkgs, ... }:
|
||||
|
||||
let
|
||||
python3 = pkgs.python3.withPackages (ps: [ ps.rtslib-fb ]);
|
||||
targetctl = "${python3}/bin/targetctl";
|
||||
|
||||
targetctlStop = pkgs.writeScript "targetctl-stop" ''
|
||||
#!${python3}/bin/python3
|
||||
import subprocess, sys
|
||||
import rtslib_fb
|
||||
|
||||
root = rtslib_fb.RTSRoot()
|
||||
targets = list(root.targets)
|
||||
if targets:
|
||||
subprocess.run(
|
||||
["${targetctl}", "save", "/etc/target/saveconfig.json"],
|
||||
capture_output=True,
|
||||
)
|
||||
print(f"saved {len(targets)} iSCSI target(s)")
|
||||
else:
|
||||
print("no active LIO targets — saveconfig.json unchanged")
|
||||
|
||||
for target in targets:
|
||||
try:
|
||||
for tpg in list(target.tpgs):
|
||||
tpg.enable = False
|
||||
target.delete()
|
||||
except Exception as e:
|
||||
print(f"warn (target): {e}", file=sys.stderr)
|
||||
for so in list(root.storage_objects):
|
||||
try:
|
||||
so.delete()
|
||||
except Exception as e:
|
||||
print(f"warn (backstore): {e}", file=sys.stderr)
|
||||
print("LIO kernel target cleared")
|
||||
'';
|
||||
in
|
||||
{
|
||||
boot.kernelModules = [
|
||||
"target_core_mod"
|
||||
"iscsi_target_mod"
|
||||
"target_core_file"
|
||||
"target_core_pscsi"
|
||||
"target_core_user"
|
||||
"configfs"
|
||||
];
|
||||
|
||||
systemd = {
|
||||
mounts = [{
|
||||
where = "/sys/kernel/config";
|
||||
what = "configfs";
|
||||
type = "configfs";
|
||||
wantedBy = [ "multi-user.target" ];
|
||||
before = [ "targetctl.service" ];
|
||||
}];
|
||||
services.targetctl = {
|
||||
description = "LIO iSCSI target config save/restore";
|
||||
wantedBy = [ "multi-user.target" ];
|
||||
after = [ "sys-kernel-config.mount" "network.target" ];
|
||||
requires = [ "sys-kernel-config.mount" ];
|
||||
serviceConfig = {
|
||||
Type = "oneshot";
|
||||
RemainAfterExit = true;
|
||||
ExecStart = "${targetctl} restore /etc/target/saveconfig.json";
|
||||
ExecStop = "${targetctlStop}";
|
||||
};
|
||||
unitConfig.ConditionFileNotEmpty = "/etc/target/saveconfig.json";
|
||||
};
|
||||
tmpfiles.rules = [
|
||||
"d /etc/target 0750 root root -"
|
||||
"f /etc/target/saveconfig.json 0640 root root -"
|
||||
];
|
||||
};
|
||||
|
||||
environment.systemPackages = [ pkgs.targetcli-fb ];
|
||||
}
|
||||
@@ -1,94 +0,0 @@
|
||||
# Pacemaker + Corosync HA stack for NixOS with known-good workarounds.
|
||||
#
|
||||
# Issues fixed here (confirmed through live testing on NixOS 25.11):
|
||||
#
|
||||
# 1. StateDirectory ownership reset: systemd's StateDirectory=pacemaker
|
||||
# creates /var/lib/pacemaker owned root:root. pacemaker-based (the CIB
|
||||
# daemon) runs as the hacluster user and calls pcmk__daemon_can_write,
|
||||
# which requires the CIB directory to be owned by hacluster or be
|
||||
# group-writable by haclient. Workaround: remove StateDirectory and let
|
||||
# ExecStartPre create every required subdirectory with correct ownership.
|
||||
#
|
||||
# 2. HA_SBIN_DIR wrong path: ocf-shellfuncs sets HA_SBIN_DIR to the Nix
|
||||
# store path of the resource-agents derivation's /sbin, which doesn't
|
||||
# exist. The DRBD OCF agent uses ${HA_SBIN_DIR}/crm_master, so it exits
|
||||
# 127 without this override. Fix: export HA_SBIN_DIR=/run/current-system/sw/bin.
|
||||
#
|
||||
# 3. Broad PATH for OCF agents: the resource executor (pacemaker-execd) runs
|
||||
# OCF agent scripts as children. NixOS provides no implicit PATH for
|
||||
# system services; without an explicit PATH the agents can't find ip, ss,
|
||||
# mount, umount, drbdadm, etc.
|
||||
#
|
||||
# 4. FUSER=true: the Filesystem OCF agent calls check_binary $FUSER (default:
|
||||
# fuser from psmisc), which is not installed. Setting FUSER=true makes
|
||||
# check_binary succeed (true is always in PATH) and the subsequent
|
||||
# "$FUSER -km $mountpoint" becomes a no-op. Pair with force_unmount=false
|
||||
# on each Filesystem resource unless you want lazy unmount behaviour.
|
||||
{ lib, pkgs, ... }:
|
||||
|
||||
let
|
||||
ocfBinPath = lib.concatStringsSep ":" [
|
||||
"${pkgs.iproute2}/bin"
|
||||
"${pkgs.iproute2}/sbin"
|
||||
"${pkgs.iputils}/bin"
|
||||
"${pkgs.util-linux}/bin"
|
||||
"${pkgs.util-linux}/sbin"
|
||||
"${pkgs.gawk}/bin"
|
||||
"${pkgs.gnugrep}/bin"
|
||||
"${pkgs.gnused}/bin"
|
||||
"${pkgs.coreutils}/bin"
|
||||
"${pkgs.bash}/bin"
|
||||
"${pkgs.procps}/bin"
|
||||
"${pkgs.xfsprogs}/bin"
|
||||
"${pkgs.drbd}/bin"
|
||||
"${pkgs.python3}/bin"
|
||||
"/run/current-system/sw/bin"
|
||||
"/run/current-system/sw/sbin"
|
||||
"/usr/local/sbin"
|
||||
"/usr/local/bin"
|
||||
"/usr/sbin"
|
||||
"/usr/bin"
|
||||
"/sbin"
|
||||
"/bin"
|
||||
];
|
||||
|
||||
# Single pre-start script: schemas symlink + directory ownership.
|
||||
# Runs before pacemakerd so pacemaker-based finds hacluster-owned dirs.
|
||||
preStartCmd = "${pkgs.bash}/bin/bash -c '"
|
||||
+ "ln -sfn ${pkgs.pacemaker}/share/pacemaker /var/lib/pacemaker/schemas; "
|
||||
+ "for d in /var/lib/pacemaker /var/lib/pacemaker/cib /var/lib/pacemaker/cores "
|
||||
+ "/var/lib/pacemaker/pengine /var/lib/pacemaker/blackbox "
|
||||
+ "/var/lib/pacemaker/hostcache; do "
|
||||
+ "mkdir -p \"\\$d\" && chown hacluster:pacemaker \"\\$d\" && chmod 2770 \"\\$d\"; "
|
||||
+ "done'";
|
||||
|
||||
ocfEnv = {
|
||||
PATH = lib.mkForce ocfBinPath;
|
||||
OCF_ROOT = "${pkgs.ocf-resource-agents}/usr/lib/ocf";
|
||||
HA_SBIN_DIR = "/run/current-system/sw/bin";
|
||||
FUSER = "true";
|
||||
};
|
||||
in
|
||||
{
|
||||
users.groups.haclient = { };
|
||||
|
||||
services.corosync.enable = true;
|
||||
services.pacemaker.enable = true;
|
||||
|
||||
systemd.services = {
|
||||
pacemaker = {
|
||||
serviceConfig = {
|
||||
StateDirectory = lib.mkForce "";
|
||||
ExecStartPre = lib.mkBefore [ preStartCmd ];
|
||||
};
|
||||
environment = ocfEnv;
|
||||
};
|
||||
pacemaker-execd.environment = ocfEnv;
|
||||
};
|
||||
|
||||
environment.systemPackages = with pkgs; [
|
||||
corosync
|
||||
pacemaker
|
||||
ocf-resource-agents
|
||||
];
|
||||
}
|
||||
@@ -1,23 +0,0 @@
|
||||
# Adapted from the output of `nixos-generate-config`, run from a live GUI
|
||||
# ISO boot on the actual gui-host hardware (AMD CPU). fileSystems and
|
||||
# swapDevices are deliberately omitted -- the live ISO had no formatted
|
||||
# disks to detect, and disko (modules/disko/baremetal.nix) generates both
|
||||
# from the declarative zpool layout anyway.
|
||||
{ config, lib, pkgs, modulesPath, ... }:
|
||||
|
||||
{
|
||||
imports =
|
||||
[
|
||||
(modulesPath + "/installer/scan/not-detected.nix")
|
||||
];
|
||||
|
||||
boot = {
|
||||
initrd.availableKernelModules = [ "xhci_pci" "ahci" "usbhid" "usb_storage" "sd_mod" ];
|
||||
initrd.kernelModules = [ ];
|
||||
kernelModules = [ "kvm-amd" ];
|
||||
extraModulePackages = [ ];
|
||||
};
|
||||
|
||||
nixpkgs.hostPlatform = lib.mkDefault "x86_64-linux";
|
||||
hardware.cpu.amd.updateMicrocode = lib.mkDefault config.hardware.enableRedistributableFirmware;
|
||||
}
|
||||
+134
-15
@@ -45,21 +45,140 @@
|
||||
disko
|
||||
];
|
||||
|
||||
# Auto-install script, kept as a real, version-controlled shell file at
|
||||
# scripts/installer/auto-install.sh rather than an inline Nix string.
|
||||
# It sources scripts/env.sh itself (for LAN_DOMAIN, same as every other
|
||||
# script in this repo) rather than relying on Nix-level templating, so
|
||||
# it behaves identically whether it's run straight from a git checkout
|
||||
# or from here -- baking scripts/env.sh in alongside it at a matching
|
||||
# relative path (installer/auto-install.sh -> ../env.sh) is what makes
|
||||
# that resolve correctly in both places.
|
||||
etc = {
|
||||
"nixos-installer/env.sh".source = ../../scripts/env.sh;
|
||||
# Write auto-install script to /root
|
||||
etc."auto-install.sh" = {
|
||||
text = ''
|
||||
#!/run/current-system/sw/bin/bash
|
||||
set -eux
|
||||
|
||||
"nixos-installer/installer/auto-install.sh" = {
|
||||
source = ../../scripts/installer/auto-install.sh;
|
||||
mode = "0755";
|
||||
};
|
||||
set -euo pipefail
|
||||
|
||||
export FLAKE_BASE_URL="git+https://${vars.lanDomain}/beatzaplenty/nixos.git"
|
||||
|
||||
echo "Fetching available NixOS hosts from flake..."
|
||||
# Two categories deliberately excluded from the menu:
|
||||
# lxc-* — these build a config.system.build.tarball meant for
|
||||
# `pct restore` on Proxmox directly, not an install.
|
||||
# Running nixos-install against one here would
|
||||
# bind-mount / onto /mnt and then refuse to touch the
|
||||
# filesystem it's currently running on — see
|
||||
# docs/auto-installer.md.
|
||||
# installer — this *is* the installer image's own flake target,
|
||||
# not a deployable host; "installing" it means
|
||||
# nixos-install-ing a copy of the installer into
|
||||
# itself.
|
||||
mapfile -t options < <(
|
||||
nix eval --json --no-use-registries --no-accept-flake-config --extra-experimental-features "flakes nix-command" \
|
||||
"''${FLAKE_BASE_URL}#nixosConfigurations" \
|
||||
--apply builtins.attrNames \
|
||||
| jq -r '.[]
|
||||
| select(startswith("lxc-") | not)
|
||||
| select(. != "installer")'
|
||||
)
|
||||
|
||||
if [[ ''${#options[@]} -eq 0 ]]; then
|
||||
echo "ERROR: No NixOS hosts found in ''${FLAKE_BASE_URL}#nixosConfigurations" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "Note: lxc-* targets aren't installed this way — build them with"
|
||||
echo " nix build .#nixosConfigurations.<name>.config.system.build.tarball"
|
||||
echo "and 'pct restore' the result on Proxmox directly. See docs/auto-installer.md."
|
||||
|
||||
echo "Choose the flake profile to install:"
|
||||
select choice in "''${options[@]}"; do
|
||||
if [[ -n "$choice" ]]; then
|
||||
echo "You selected: $choice"
|
||||
break
|
||||
else
|
||||
echo "Invalid selection. Try again."
|
||||
fi
|
||||
done
|
||||
|
||||
echo "Starting install with flake: ''${FLAKE_BASE_URL}#''${choice}"
|
||||
|
||||
# Optional: confirm before proceeding
|
||||
read -rp "Proceed with installation? (y/N): " confirm
|
||||
if [[ ! "$confirm" =~ ^[Yy]$ ]]; then
|
||||
echo "Aborted."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# A nix-cache host is *the* substituter/remote-builder for every other
|
||||
# host once installed (its own config explicitly excludes itself from
|
||||
# using either — see buildType != "nix-cache" in the nixos flake.nix).
|
||||
# Installing one shouldn't depend on a nix-cache substituter either,
|
||||
# for the same reason — plus in practice "nix-cache" only resolves over
|
||||
# Tailscale, which a fresh installer environment was never connected to
|
||||
# anyway, so it's dead weight even for non-nix-cache installs until
|
||||
# that's sorted out. Override it away here specifically for nix-cache
|
||||
# targets to keep install-time behaviour consistent with run-time.
|
||||
nix_extra_opts=()
|
||||
if [[ "''${choice}" == *-nix-cache ]]; then
|
||||
echo "Installing a nix-cache host — skipping the nix-cache substituter."
|
||||
nix_extra_opts+=(--option substituters "https://cache.nixos.org/")
|
||||
fi
|
||||
|
||||
# Every host reachable through this menu has a Disko config (lxc-*
|
||||
# is filtered out above, and is the only category that doesn't —
|
||||
# see docs/auto-installer.md), so this can run unconditionally: no
|
||||
# need to probe the flake first and branch on whether Disko applies.
|
||||
disko --mode destroy,format,mount \
|
||||
--flake "''${FLAKE_BASE_URL}#''${choice}" "''${nix_extra_opts[@]}" --yes-wipe-all-disks
|
||||
|
||||
# sops-nix derives this host's decryption key from its own SSH host key
|
||||
# at *activation* time, which runs before systemd would otherwise
|
||||
# generate one on first boot. Without pre-seeding it here, secrets
|
||||
# (including the login password) fail to decrypt on first boot.
|
||||
# Generate the key with scripts/prepare-host-key.sh first.
|
||||
#
|
||||
# Two places a key can come from, checked in order:
|
||||
# /etc/host-keys — baked into this image at build time (see
|
||||
# modules/installer/host-keys.nix; only present
|
||||
# if built with NIXOS_HOST_KEYS_DIR set)
|
||||
# /root/host-keys — scp'd in manually after boot (older fallback,
|
||||
# still supported for images built without keys)
|
||||
mkdir -p /root/host-keys
|
||||
if [[ -f "/etc/host-keys/''${choice}_ssh_host_ed25519_key" ]]; then
|
||||
echo "Found baked-in SSH host key for ''${choice}, installing to target..."
|
||||
install -D -m 0600 "/etc/host-keys/''${choice}_ssh_host_ed25519_key" /mnt/etc/ssh/ssh_host_ed25519_key
|
||||
install -D -m 0644 "/etc/host-keys/''${choice}_ssh_host_ed25519_key.pub" /mnt/etc/ssh/ssh_host_ed25519_key.pub
|
||||
elif [[ -f "/root/host-keys/''${choice}_ssh_host_ed25519_key" ]]; then
|
||||
echo "Found pre-seeded SSH host key for ''${choice}, installing to target..."
|
||||
install -D -m 0600 "/root/host-keys/''${choice}_ssh_host_ed25519_key" /mnt/etc/ssh/ssh_host_ed25519_key
|
||||
install -D -m 0644 "/root/host-keys/''${choice}_ssh_host_ed25519_key.pub" /mnt/etc/ssh/ssh_host_ed25519_key.pub
|
||||
else
|
||||
echo "WARNING: no SSH host key found for ''${choice} (checked /etc/host-keys and /root/host-keys)"
|
||||
echo "sops-nix secrets (including the login password) will NOT decrypt on first boot."
|
||||
echo "Run scripts/prepare-host-key.sh for host ''${choice} on your admin workstation first,"
|
||||
echo "then either rebuild this image with NIXOS_HOST_KEYS_DIR set, or scp the result to"
|
||||
echo "/root/host-keys/ on this machine."
|
||||
read -rp "Continue without a pre-seeded key anyway? (y/N): " skip_key
|
||||
if [[ ! "$skip_key" =~ ^[Yy]$ ]]; then
|
||||
echo "Aborted."
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
|
||||
mkdir -p /mnt/install-tmp
|
||||
export TMPDIR=/mnt/install-tmp
|
||||
|
||||
nixos-install \
|
||||
--flake "''${FLAKE_BASE_URL}#''${choice}" \
|
||||
"''${nix_extra_opts[@]}" \
|
||||
--no-root-password
|
||||
|
||||
|
||||
rm -rf /mnt/install-tmp
|
||||
# Redundant copy of the host's private key — the real one is now at
|
||||
# /etc/ssh/ssh_host_ed25519_key. Nothing NixOS-managed ever cleans this
|
||||
# up on its own since it was written imperatively, not declaratively.
|
||||
rm -rf /root/host-keys
|
||||
sleep 10
|
||||
reboot
|
||||
'';
|
||||
|
||||
mode = "0755";
|
||||
};
|
||||
};
|
||||
|
||||
@@ -73,7 +192,7 @@
|
||||
# file-copying/chown.
|
||||
programs.bash.loginShellInit = ''
|
||||
if [ -n "$PS1" ] && [ ! -e "$HOME/.auto_install_ran" ]; then
|
||||
sudo /etc/nixos-installer/installer/auto-install.sh
|
||||
sudo /etc/auto-install.sh
|
||||
touch "$HOME/.auto_install_ran"
|
||||
fi
|
||||
'';
|
||||
|
||||
@@ -1,194 +0,0 @@
|
||||
# Fully declarative FreeIPA domain membership.
|
||||
#
|
||||
# Imported by modules/common/configuration.nix — no per-host wiring needed.
|
||||
# Enables itself automatically on any host that has a sops-encrypted keytab
|
||||
# at secrets/<hostname>.keytab; is a no-op for all other hosts.
|
||||
#
|
||||
# To enroll a new host:
|
||||
# 0. scripts/secrets/sync-host-keys.sh <flake-target>
|
||||
# 1. scripts/ipa/create-nixos-ipa-host-account.sh [--ip <addr>] <hostname>
|
||||
# (adds .sops.yaml rule, runs ipa host-add, encrypts keytab in one step)
|
||||
# 2. git add secrets/<hostname>.keytab .sops.yaml && git commit
|
||||
# 3. Deploy — no further steps required.
|
||||
#
|
||||
# Manual fallback (if the script isn't usable):
|
||||
# a. On the FreeIPA server: ipa host-add <fqdn> [--ip-address=<ip>] --force
|
||||
# b. On the FreeIPA server: ipa-getkeytab -s <ipa-server> -p host/<fqdn> -k /tmp/<host>.keytab
|
||||
# c. From the repo root (path must match for sops creation rule to apply):
|
||||
# cp /tmp/<host>.keytab secrets/<host>.keytab
|
||||
# sops -e --input-type binary -i secrets/<host>.keytab
|
||||
# d. Commit secrets/<host>.keytab and the updated .sops.yaml, then deploy.
|
||||
#
|
||||
# vars dependencies: homeDomain, ipaServer, domainControllerIp, ipaUser
|
||||
|
||||
{ config, lib, pkgs, vars, ... }:
|
||||
|
||||
let
|
||||
keytabPath = ../../secrets + "/${config.networking.hostName}.keytab";
|
||||
enabled = builtins.pathExists keytabPath;
|
||||
|
||||
realm = lib.strings.toUpper vars.homeDomain;
|
||||
fqdn = "${config.networking.hostName}.${vars.homeDomain}";
|
||||
# "sweet.home" -> "dc=sweet,dc=home"
|
||||
basedn = lib.strings.concatMapStringsSep "," (c: "dc=${c}") (lib.strings.splitString "." vars.homeDomain);
|
||||
# security.ipa.certificate expects a derivation (package), not a raw path.
|
||||
caCertPkg = pkgs.writeText "ipa-ca.crt" (builtins.readFile ../../certs/ipa-ca.crt);
|
||||
in
|
||||
lib.mkIf enabled {
|
||||
networking.domain = lib.mkDefault vars.homeDomain;
|
||||
networking.nameservers = lib.mkDefault [ vars.domainControllerIp ];
|
||||
|
||||
security = {
|
||||
ipa = {
|
||||
enable = true;
|
||||
domain = vars.homeDomain;
|
||||
inherit realm;
|
||||
server = vars.ipaServer;
|
||||
certificate = caCertPkg;
|
||||
inherit basedn;
|
||||
ipaHostname = fqdn;
|
||||
offlinePasswords = true;
|
||||
cacheCredentials = true;
|
||||
};
|
||||
|
||||
# Create the home directory on first login if it doesn't exist yet.
|
||||
# IPA users have no pre-created home on the host; without this sshd
|
||||
# opens a session to a non-existent directory and resets the connection.
|
||||
# lightdm also needs this so the GUI login path can create the home dir
|
||||
# if it was not pre-seeded by the tmpfiles rule above (e.g. on first boot
|
||||
# before SSSD has resolved the user).
|
||||
pam.services = {
|
||||
sshd.makeHomeDir = true;
|
||||
lightdm.makeHomeDir = true;
|
||||
};
|
||||
|
||||
# HM with useUserPackages = true (flake.nix) sets users.users.${ipaUser}.packages,
|
||||
# which forces the stub into /etc/passwd. pam_sss.so with the "localusers" flag
|
||||
# (added by NixOS when SSSD is enabled) then skips SSSD for any user it finds in
|
||||
# local /etc/passwd — including this stub — falling through to pam_unix, which has
|
||||
# no password for the stub → sudo auth always fails.
|
||||
#
|
||||
# Fix: NOPASSWD for the IPA user. The IPA user already authenticated to reach a
|
||||
# shell (SSH public key from IPA or Kerberos), so re-prompting via a broken PAM
|
||||
# path is security theater on a single-admin homelab.
|
||||
sudo.extraRules = [{
|
||||
users = [ vars.ipaUser ];
|
||||
commands = [{ command = "ALL"; options = [ "NOPASSWD" ]; }];
|
||||
}];
|
||||
};
|
||||
|
||||
systemd = {
|
||||
# Fetch SSH public keys from IPA so users can log in with the key stored
|
||||
# in their IPA profile rather than needing ~/.ssh/authorized_keys on every
|
||||
# host. sss_ssh_authorizedkeys queries SSSD (which queries IPA LDAP).
|
||||
#
|
||||
# /nix/store is 1775 (group-writable by nixbld). OpenSSH 10.0+ rejects
|
||||
# AuthorizedKeysCommand binaries whose path contains any group-writable
|
||||
# component, silently skipping the command. Copy to /usr/local/bin (all
|
||||
# components root-owned, 755) so the path passes sshd's safety check.
|
||||
tmpfiles.rules = [
|
||||
"d /usr/local 0755 root root - -"
|
||||
"d /usr/local/bin 0755 root root - -"
|
||||
"C+ /usr/local/bin/sss_ssh_authorizedkeys 0555 root root - ${pkgs.sssd}/bin/sss_ssh_authorizedkeys"
|
||||
# Pre-create the IPA user's home dir so Home Manager activation succeeds
|
||||
# even before their first login. On a fresh system SSSD may not have
|
||||
# resolved the user yet — tmpfiles warns and skips in that case (non-fatal),
|
||||
# and pam_mkhomedir covers the first-login path as a fallback.
|
||||
"d /home/${vars.ipaUser} 0700 ${vars.ipaUser} ${vars.ipaUser} - -"
|
||||
];
|
||||
|
||||
# security.ipa enables Kerberos (security.krb5) which causes systemd to
|
||||
# start auth-rpcgss-module.service and rpc-gssd.service for Kerberos NFS
|
||||
# authentication. LXC containers can't load the auth_rpcgss kernel module
|
||||
# and don't have /var/lib/nfs/rpc_pipefs, so both services fail.
|
||||
#
|
||||
# The NixOS IPA module already adds a drop-in for auth-rpcgss-module.service
|
||||
# with ConditionPathExists=/etc/krb5.keytab. We use lib.mkForce to win the
|
||||
# text conflict and add ConditionVirtualization=!container alongside it so
|
||||
# the service is skipped (not failed) in containers that do have a keytab.
|
||||
# Same fix for rpc-gssd.service which also fails in containers.
|
||||
units = lib.mkIf config.boot.isContainer {
|
||||
"auth-rpcgss-module.service" = {
|
||||
overrideStrategy = "asDropinIfExists";
|
||||
text = lib.mkForce ''
|
||||
[Unit]
|
||||
ConditionPathExists=
|
||||
ConditionPathExists=/etc/krb5.keytab
|
||||
ConditionVirtualization=!container
|
||||
'';
|
||||
};
|
||||
# rpc-gssd also has ConditionPathExists from the NixOS IPA module (and an
|
||||
# X-Restart-Triggers store path from systemd.nix). Use mkForce to win;
|
||||
# omit X-Restart-Triggers since this service is skipped in containers anyway.
|
||||
"rpc-gssd.service" = {
|
||||
overrideStrategy = "asDropinIfExists";
|
||||
text = lib.mkForce ''
|
||||
[Unit]
|
||||
ConditionPathExists=
|
||||
ConditionPathExists=/etc/krb5.keytab
|
||||
ConditionVirtualization=!container
|
||||
'';
|
||||
};
|
||||
};
|
||||
|
||||
# home-manager-<user>.service fails on first enrollment because /home/wayne
|
||||
# doesn't exist until the user's first login (pam_mkhomedir creates it then).
|
||||
# ConditionPathExists makes systemd skip the service (exit 0, condition not
|
||||
# met) instead of failing. After first login the dir exists and subsequent
|
||||
# rebuilds activate HM normally.
|
||||
services."home-manager-${vars.ipaUser}".unitConfig.ConditionPathExists =
|
||||
"/home/${vars.ipaUser}";
|
||||
};
|
||||
|
||||
services.openssh.extraConfig = ''
|
||||
AuthorizedKeysCommand /usr/local/bin/sss_ssh_authorizedkeys %u
|
||||
AuthorizedKeysCommandUser nobody
|
||||
'';
|
||||
|
||||
# Host keytab: pre-provisioned on the IPA server, sops-encrypted binary.
|
||||
# Placed at /etc/krb5.keytab before SSSD starts so the host authenticates
|
||||
# to IPA without running ipa-client-install.
|
||||
sops.secrets."ipa-host-keytab" = {
|
||||
sopsFile = keytabPath;
|
||||
format = "binary";
|
||||
path = "/etc/krb5.keytab";
|
||||
owner = "root";
|
||||
group = "root";
|
||||
mode = "0600";
|
||||
restartUnits = [ "sssd.service" ];
|
||||
};
|
||||
|
||||
# NixOS requires isNormalUser/isSystemUser + group on any entry in
|
||||
# users.users. HM with useUserPackages = true (set in flake.nix) adds a stub
|
||||
# entry for each HM user so it can install packages to
|
||||
# /etc/profiles/per-user/<name>/. This definition satisfies those assertions.
|
||||
# With security.ipa setting "passwd: sss files" in nsswitch, SSSD's IPA entry
|
||||
# takes priority for NSS lookups — this local stub is only a fallback when
|
||||
# SSSD is unreachable (at which point auth fails anyway).
|
||||
users.users.${vars.ipaUser} = {
|
||||
isNormalUser = true;
|
||||
group = "users";
|
||||
extraGroups = [ "wheel" ];
|
||||
createHome = false;
|
||||
};
|
||||
|
||||
# Home Manager config for the IPA primary user, applied on every enrolled
|
||||
# host. Manages what IPA doesn't: dotfiles, user-scoped packages, session
|
||||
# variables. Switch-nix/Test-nix/buildImage are system-wide (configuration.nix)
|
||||
# so they don't need to be repeated here.
|
||||
#
|
||||
# homeDirectory uses mkForce because HM's NixOS integration module sets it to
|
||||
# "/var/empty" for users not found in config.users.users at eval time (SSSD
|
||||
# users aren't visible there).
|
||||
home-manager.users.${vars.ipaUser} = { pkgs, ... }: {
|
||||
home = {
|
||||
username = vars.ipaUser;
|
||||
homeDirectory = lib.mkForce "/home/${vars.ipaUser}";
|
||||
stateVersion = "26.05";
|
||||
packages = with pkgs; [ tmux sshfs ];
|
||||
sessionVariables.EDITOR = "nano";
|
||||
};
|
||||
programs.home-manager.enable = true;
|
||||
programs.bash.enable = true;
|
||||
};
|
||||
}
|
||||
@@ -1,43 +0,0 @@
|
||||
{ config, lib, vars, ... }:
|
||||
|
||||
{
|
||||
# Prestages a NetworkManager connection profile for vars.wifiSsid so the
|
||||
# host associates on first boot with no manual nmtui/nmcli step. Guarded
|
||||
# on a non-empty SSID so leaving the placeholder blank in variables.nix
|
||||
# is a no-op rather than an empty, broken profile — fill it in once the
|
||||
# network is known.
|
||||
#
|
||||
# The password itself lives in secrets/gui.yaml, not variables.nix --
|
||||
# NetworkManager's ensureProfiles renders `psk = "$WIFI_PASSWORD"`
|
||||
# literally into the store (see nixpkgs' own ensureProfiles example,
|
||||
# which does the same for exactly this reason) and its systemd service
|
||||
# envsubst-expands it from environmentFiles at activation time, so the
|
||||
# real value only ever touches /run (root-only, UMask 0177), never the
|
||||
# Nix store.
|
||||
sops.secrets."wifi-password" = lib.mkIf (vars.wifiSsid != "") {
|
||||
sopsFile = ../../secrets/gui.yaml;
|
||||
};
|
||||
|
||||
sops.templates."wifi-password.env" = lib.mkIf (vars.wifiSsid != "") {
|
||||
content = "WIFI_PASSWORD=${config.sops.placeholder."wifi-password"}";
|
||||
};
|
||||
|
||||
networking.networkmanager.ensureProfiles = lib.mkIf (vars.wifiSsid != "") {
|
||||
environmentFiles = [ config.sops.templates."wifi-password.env".path ];
|
||||
|
||||
profiles.${vars.wifiSsid} = {
|
||||
connection = {
|
||||
id = vars.wifiSsid;
|
||||
type = "wifi";
|
||||
};
|
||||
wifi = {
|
||||
mode = "infrastructure";
|
||||
ssid = vars.wifiSsid;
|
||||
};
|
||||
wifi-security = {
|
||||
key-mgmt = "wpa-psk";
|
||||
psk = "$WIFI_PASSWORD";
|
||||
};
|
||||
};
|
||||
};
|
||||
}
|
||||
@@ -3,7 +3,7 @@
|
||||
{
|
||||
nix.settings = {
|
||||
substituters = [
|
||||
"http://${vars.nixCacheHost}.${vars.homeDomain}"
|
||||
"http://${vars.nixCacheHost}"
|
||||
"https://cache.nixos.org/"
|
||||
];
|
||||
trusted-public-keys = [
|
||||
|
||||
@@ -1,19 +1,15 @@
|
||||
{ pkgs, vars, ... }:
|
||||
|
||||
{
|
||||
# Authenticate as nixremote using the client host's own default root SSH
|
||||
# identity (/root/.ssh/id_ed25519) rather than a separately-named key --
|
||||
# matches vars.remoteBuilderAuthorizedKeys, which already authorizes
|
||||
# each host's own default key (one entry per host, not a shared
|
||||
# dedicated keypair). If this host doesn't have one yet:
|
||||
# sudo -u root ssh-keygen -t ed25519 -N '' -f /root/.ssh/id_ed25519
|
||||
# # then add its .pub to vars.remoteBuilderAuthorizedKeys and rebuild nix-cache
|
||||
# sudo ssh -i /root/.ssh/id_ed25519 nixremote@nix-cache.sweet.home nix-store --version
|
||||
# Install the remote builder key on each client host (do not commit private keys):
|
||||
# sudo install -d -m 0700 /root/.ssh
|
||||
# sudo install -m 0600 ./nixremote /root/.ssh/nixremote
|
||||
# sudo ssh -i /root/.ssh/nixremote nixremote@nix-cache nix-store --version
|
||||
# Trust nix-cache's SSH host key declaratively so the nix-daemon (root)
|
||||
# can connect the first time without a manual ssh-keyscan/known_hosts
|
||||
# step on every new client.
|
||||
programs.ssh.knownHosts."${vars.nixCacheHost}.${vars.homeDomain}" = {
|
||||
hostNames = [ "${vars.nixCacheHost}.${vars.homeDomain}" ];
|
||||
programs.ssh.knownHosts.${vars.nixCacheHost} = {
|
||||
hostNames = [ vars.nixCacheHost ];
|
||||
publicKey = vars.nixCacheHostKey;
|
||||
};
|
||||
|
||||
@@ -22,9 +18,9 @@
|
||||
|
||||
buildMachines = [
|
||||
{
|
||||
hostName = "${vars.nixCacheHost}.${vars.homeDomain}";
|
||||
hostName = vars.nixCacheHost;
|
||||
sshUser = vars.remoteBuilderUser;
|
||||
sshKey = "/root/.ssh/id_ed25519";
|
||||
sshKey = "/root/.ssh/${vars.remoteBuilderUser}";
|
||||
inherit (pkgs.stdenv.hostPlatform) system;
|
||||
maxJobs = 4;
|
||||
speedFactor = 2;
|
||||
|
||||
@@ -20,7 +20,7 @@
|
||||
nginx = {
|
||||
enable = true;
|
||||
recommendedProxySettings = true;
|
||||
virtualHosts."${vars.nixCacheHost}.${vars.homeDomain}" = {
|
||||
virtualHosts.${vars.nixCacheHost} = {
|
||||
locations."/" = {
|
||||
proxyPass = "http://${config.services.nix-serve.bindAddress}:${toString config.services.nix-serve.port}";
|
||||
};
|
||||
|
||||
@@ -1,38 +0,0 @@
|
||||
{ ... }:
|
||||
|
||||
{
|
||||
imports = [
|
||||
../hardware-configuration/baremetal.nix
|
||||
../boot/efi.nix
|
||||
../disko/baremetal.nix
|
||||
../services/zfs/enable-service.nix
|
||||
];
|
||||
|
||||
# Needed for real wifi/bluetooth/GPU firmware blobs and CPU microcode
|
||||
# updates (hardware-configuration/baremetal.nix's amd.updateMicrocode
|
||||
# keys off this) -- irrelevant on the linode/proxmox/lxc platforms,
|
||||
# which are all VMs with no real hardware to load firmware for.
|
||||
hardware.enableRedistributableFirmware = true;
|
||||
|
||||
# AMD GPU: the amdgpu kernel driver autoloads from the PCI ID with no
|
||||
# extra boot.kernelModules entry needed; this is the userspace half --
|
||||
# the dedicated Xorg driver (not just the generic modesetting fallback)
|
||||
# plus Mesa OpenGL/Vulkan (amdgpu/RADV), same firmware blobs as above.
|
||||
# 32-bit support is for compatibility with 32-bit apps/games.
|
||||
services.xserver.videoDrivers = [ "amdgpu" ];
|
||||
|
||||
hardware.graphics = {
|
||||
enable = true;
|
||||
enable32Bit = true;
|
||||
};
|
||||
|
||||
# The systemd-based initrd (default here since this host has a ZFS root --
|
||||
# see modules/disko/baremetal.nix) locks the root account by default, so
|
||||
# sulogin refuses to hand over a shell if something in the initrd (e.g.
|
||||
# the ZFS pool import) fails and it drops to emergency mode -- confirmed
|
||||
# live: it just loops re-entering the target instead of prompting. This
|
||||
# only affects the pre-switch-root initrd shell, not the installed
|
||||
# system's own login, and is worth the tradeoff on a box already reachable
|
||||
# at the physical console.
|
||||
boot.initrd.systemd.emergencyAccess = true;
|
||||
}
|
||||
@@ -1,4 +1,4 @@
|
||||
{ config, lib, modulesPath, flakeTarget, ... }:
|
||||
{ lib, modulesPath, flakeTarget, ... }:
|
||||
|
||||
let
|
||||
# Bakes this exact flake target's pre-generated SSH host key straight
|
||||
@@ -14,7 +14,7 @@ let
|
||||
# Without this, config.system.build.tarball's built-in system just
|
||||
# generates a fresh host key at first boot like any other host would --
|
||||
# but sops-nix derives its decryption key from *this* file, and
|
||||
# .sops.yaml only trusts whatever key scripts/secrets/sync-host-keys.sh already
|
||||
# .sops.yaml only trusts whatever key scripts/sync-host-keys.sh already
|
||||
# registered for this exact target name. A freshly-generated key can
|
||||
# never match that, so every secret (including this host's own login)
|
||||
# permanently fails to decrypt. Confirmed live: sops-install-secrets
|
||||
@@ -26,7 +26,7 @@ let
|
||||
hostKeysDir = /. + hostKeysDirStr;
|
||||
|
||||
# flakeTarget ("${platform}-${buildType}") comes in via specialArgs from
|
||||
# flake.nix's mkTarget -- exactly the name scripts/secrets/sync-host-keys.sh
|
||||
# flake.nix's mkTarget -- exactly the name scripts/sync-host-keys.sh
|
||||
# registers keys under. Deliberately not read back from
|
||||
# config.environment.etc."flake-target" (which is set to the same value)
|
||||
# -- this module also *contributes* to environment.etc below, and a
|
||||
@@ -52,34 +52,14 @@ in
|
||||
# LXC container does).
|
||||
imports = [
|
||||
(modulesPath + "/virtualisation/proxmox-lxc.nix")
|
||||
../common/preserve-ssh-host-key.nix
|
||||
];
|
||||
|
||||
proxmoxLXC = {
|
||||
# host.nix declares each host's real hostname (networking.hostName);
|
||||
# keep that instead of letting Proxmox's ambient container config win.
|
||||
manageHostName = true;
|
||||
# Unprivileged by default -- matches how these containers are actually
|
||||
# created (scripts/proxmox/create-proxmox-resource.sh reads this value
|
||||
# back to decide `pct create`'s --unprivileged flag, so the two stay
|
||||
# in sync).
|
||||
#
|
||||
# Any lxc-* host with an NFS fileSystem must be privileged: the kernel's
|
||||
# NFS client doesn't set FS_USERNS_MOUNT, so mounting NFS from inside
|
||||
# *any* non-init user namespace -- which is exactly what an unprivileged
|
||||
# container's UID-mapped root runs in -- is rejected at the VFS layer
|
||||
# with EPERM, no matter what Proxmox's own `mount=nfs;nfs4` container
|
||||
# feature allows at the AppArmor layer (confirmed live: TCP to the NFS
|
||||
# server succeeds, the server's export table matches the container's IP,
|
||||
# and `mount.nfs: Operation not permitted` still fires immediately with
|
||||
# no corresponding denial anywhere in the server's logs -- a kernel-level
|
||||
# rejection, not a network or export-permission one). Deriving this from
|
||||
# fileSystems rather than a per-host override keeps it self-consistent:
|
||||
# any new lxc-* host that declares an NFS mount automatically gets the
|
||||
# privilege level it needs without a separate manual flag.
|
||||
privileged = builtins.any
|
||||
(fs: fs.fsType == "nfs" || fs.fsType == "nfs4")
|
||||
(builtins.attrValues config.fileSystems);
|
||||
# Unprivileged matches how these containers are actually created.
|
||||
privileged = false;
|
||||
};
|
||||
|
||||
boot.loader = {
|
||||
@@ -112,12 +92,9 @@ in
|
||||
# sops-nix's "for users" secrets (password hashes -- installed by the
|
||||
# activation script itself, not a systemd service, since they need to
|
||||
# exist *before* user creation) nor the user-creation step that
|
||||
# consumes them ever run on a real lxc-* boot. In this config sops-nix
|
||||
# does NOT generate its own boot-time service (confirmed live: no
|
||||
# sops-nix.service in systemctl list-unit-files on a deployed
|
||||
# lxc-tor-relay container); /run/secrets is a tmpfs cleared on every
|
||||
# reboot, so secrets must be reinstalled on each non-first boot by
|
||||
# nixos-lxc-sops-reinstall (below).
|
||||
# consumes them ever run on a real lxc-* boot. Regular secrets
|
||||
# (nix-serve's key, beszel's token, etc.) work anyway because sops-nix
|
||||
# provides its own systemd service for those.
|
||||
#
|
||||
# A systemd service, not boot.postBootCommands: tried that first (it's
|
||||
# a genuine, generally-invoked hook -- nixos/modules/system/boot/stage-2-init.sh,
|
||||
@@ -168,41 +145,4 @@ in
|
||||
touch /var/lib/nixos-lxc-first-boot-activated
|
||||
'';
|
||||
};
|
||||
|
||||
# Reinstalls sops secrets on every non-first boot. /run/secrets is a
|
||||
# tmpfs that is cleared on each reboot; without this service, secrets
|
||||
# are permanently absent after the first boot and every service that
|
||||
# reads from /run/secrets fails on start.
|
||||
#
|
||||
# wantedBy/before network.target: switch-to-configuration test requires
|
||||
# D-Bus to restart systemd targets after running activation scripts. D-Bus
|
||||
# is available once basic.target completes (the default After=basic.target
|
||||
# that DefaultDependencies would otherwise add). Placing the service before
|
||||
# network.target ensures secrets are ready before any network-dependent
|
||||
# service (including beszel-agent and nix-serve) starts, while running late
|
||||
# enough that D-Bus is already up.
|
||||
#
|
||||
# ConditionPathExists=... skips this service on the genuine first boot
|
||||
# (the marker doesn't exist yet); nixos-lxc-first-boot-activate handles
|
||||
# that case. On every subsequent boot the condition passes and secrets
|
||||
# are reinstalled before user services start.
|
||||
#
|
||||
# SuccessExitStatus=11: switch-to-configuration exits 11 when it cannot
|
||||
# acquire the activation lock (another switch is already in progress).
|
||||
# During a nixos-rebuild switch the activation already installs secrets, so
|
||||
# treating the lock-held case as success is correct.
|
||||
systemd.services.nixos-lxc-sops-reinstall = {
|
||||
description = "Reinstall sops secrets on each non-first boot (LXC, /run is tmpfs)";
|
||||
wantedBy = [ "network.target" ];
|
||||
before = [ "network.target" ];
|
||||
unitConfig.ConditionPathExists = "/var/lib/nixos-lxc-first-boot-activated";
|
||||
serviceConfig = {
|
||||
Type = "oneshot";
|
||||
RemainAfterExit = true;
|
||||
SuccessExitStatus = "11";
|
||||
};
|
||||
script = ''
|
||||
/run/current-system/bin/switch-to-configuration test
|
||||
'';
|
||||
};
|
||||
}
|
||||
|
||||
@@ -1,50 +1,9 @@
|
||||
{ lib, flakeTarget, ... }:
|
||||
{ ... }:
|
||||
|
||||
let
|
||||
# Bakes this exact flake target's pre-generated SSH host key straight
|
||||
# into /etc/ssh/ -- mirrors lxc.nix's builtins.getEnv pattern (impure
|
||||
# and empty under normal `nix build`/`nix eval`, so this is a no-op
|
||||
# unless explicitly opted into with NIXOS_HOST_KEYS_DIR=... --impure).
|
||||
#
|
||||
# Unlike --pre-format-files (which places files on the QEMU builder VM's
|
||||
# rootfs, not the target disk), embedding via environment.etc here means
|
||||
# nixos-install's own activation step installs the key onto the target
|
||||
# disk. sshd-keygen then finds it already present and skips generation,
|
||||
# so the disk image boots with the clan-registered key and sops can
|
||||
# decrypt on first boot.
|
||||
#
|
||||
# Without this, nixos-install's sshd-keygen activation generates a fresh
|
||||
# key (unregistered in .sops.yaml), sops decryption fails permanently,
|
||||
# and password hashes are never applied -- confirmed live: passwords
|
||||
# stayed '!' even with mutableUsers = false because hashedPasswordFile
|
||||
# pointed to a path that sops never wrote.
|
||||
hostKeysDirStr = builtins.getEnv "NIXOS_HOST_KEYS_DIR";
|
||||
hasHostKeysDir = hostKeysDirStr != "" && builtins.pathExists hostKeysDirStr;
|
||||
hostKeysDir = /. + hostKeysDirStr;
|
||||
|
||||
privKeyFile = hostKeysDir + "/${flakeTarget}_ssh_host_ed25519_key";
|
||||
pubKeyFile = hostKeysDir + "/${flakeTarget}_ssh_host_ed25519_key.pub";
|
||||
hasKeyForThisTarget =
|
||||
hasHostKeysDir
|
||||
&& builtins.pathExists privKeyFile
|
||||
&& builtins.pathExists pubKeyFile;
|
||||
in
|
||||
{
|
||||
imports = [
|
||||
../hardware-configuration/vm/proxmox.nix
|
||||
../boot/efi.nix
|
||||
../disko/proxmox.nix
|
||||
../common/preserve-ssh-host-key.nix
|
||||
];
|
||||
|
||||
environment.etc = lib.mkIf hasKeyForThisTarget {
|
||||
"ssh/ssh_host_ed25519_key" = {
|
||||
source = privKeyFile;
|
||||
mode = "0600";
|
||||
};
|
||||
"ssh/ssh_host_ed25519_key.pub" = {
|
||||
source = pubKeyFile;
|
||||
mode = "0644";
|
||||
};
|
||||
};
|
||||
}
|
||||
|
||||
@@ -1,42 +0,0 @@
|
||||
{ config, lib, vars, ... }:
|
||||
|
||||
let
|
||||
# Use the same FQDN approach as docker/mount-data.nix — a bare hostname is
|
||||
# unreliable: systemd-resolved only tries LLMNR for single-label names, and
|
||||
# a global search domain causes it to skip the interface-scoped LAN DNS.
|
||||
nfsServer = "${vars.nfsServerHost}.${vars.homeDomain}";
|
||||
in
|
||||
{
|
||||
fileSystems.${vars.nfsShares.pxebootImages.mountpoint} = {
|
||||
device = "${nfsServer}:${vars.storageRoot}/${vars.nfsShares.pxebootImages.subpath}";
|
||||
fsType = "nfs";
|
||||
options = [
|
||||
"_netdev"
|
||||
"noatime"
|
||||
] ++ (if config.boot.isContainer
|
||||
# NFSv4 requires rpc_pipefs (sunrpc filesystem), which Proxmox LXC
|
||||
# containers block unless `features: mount=nfs` is set. Use NFSv3+nolock
|
||||
# instead: no rpc_pipefs dependency at the protocol level, and rpcbind
|
||||
# on the server handles port resolution without needing client-side
|
||||
# sunrpc infrastructure. nofail keeps boot clean if server is unreachable.
|
||||
then [ "nfsvers=3" "proto=tcp" "nolock" "nofail" ]
|
||||
else [ "nfsvers=4.2" "x-systemd.automount" ]);
|
||||
};
|
||||
|
||||
# NixOS pulls var-lib-nfs-rpc_pipefs.mount (the sunrpc filesystem) into
|
||||
# nfs-client.target for any nfs fileSystems entry. In LXC containers the
|
||||
# sunrpc mount is blocked by Proxmox's AppArmor profile, causing it to fail
|
||||
# and the activation to report an error even though our mount uses nofail.
|
||||
# Add ConditionVirtualization=!container via drop-in so systemd skips the
|
||||
# unit entirely in containers (skip = inactive, not failed), which keeps
|
||||
# nfs-client.target green and activation clean.
|
||||
systemd.units = lib.mkIf config.boot.isContainer {
|
||||
"var-lib-nfs-rpc_pipefs.mount" = {
|
||||
overrideStrategy = "asDropin";
|
||||
text = ''
|
||||
[Unit]
|
||||
ConditionVirtualization=!container
|
||||
'';
|
||||
};
|
||||
};
|
||||
}
|
||||
@@ -1,32 +1,23 @@
|
||||
{ netbootSystem, netbootMinimalSystem, ... }:
|
||||
{ netbootSystem, ... }:
|
||||
|
||||
let
|
||||
# config.system.build.kernel and .netbootRamdisk are directories, not the
|
||||
# files themselves — nixpkgs' own system.build.kexecTree does the same
|
||||
# ${...}/<file> dereference for the same reason.
|
||||
mkStageRules = { dirName, system }:
|
||||
let
|
||||
inherit (system.config.system.boot.loader) kernelFile;
|
||||
dir = "/srv/pxe/http/${dirName}";
|
||||
in
|
||||
[
|
||||
# Declared here too (not just in build-types/pxe-boot.nix) so this
|
||||
# module's C+ rules don't depend on cross-module list-merge ordering —
|
||||
# tmpfiles' C type needs the target directory to already exist.
|
||||
"d ${dir} 0755 root root -"
|
||||
"C+ ${dir}/${kernelFile} 0644 root root - ${system.config.system.build.kernel}/${kernelFile}"
|
||||
"C+ ${dir}/initrd 0644 root root - ${system.config.system.build.netbootRamdisk}/initrd"
|
||||
"C+ ${dir}/netboot.ipxe 0644 root root - ${system.config.system.build.netbootIpxeScript}/netboot.ipxe"
|
||||
];
|
||||
inherit (netbootSystem.config.system.boot.loader) kernelFile;
|
||||
in
|
||||
{
|
||||
# Builds this flake's own installer netboot image (the same one
|
||||
# `nix build .#pxe` produces) plus the vanilla NixOS minimal netboot image
|
||||
# (`nix build .#pxe-minimal`), and stages both where menu.ipxe's
|
||||
# :auto-installer / :nixos-minimal entries expect them, so the pxe-boot
|
||||
# host is self-contained — no manual operator step to populate
|
||||
# /srv/pxe/http after deploy.
|
||||
systemd.tmpfiles.rules =
|
||||
mkStageRules { dirName = "auto-installer"; system = netbootSystem; }
|
||||
++ mkStageRules { dirName = "nixos-minimal"; system = netbootMinimalSystem; };
|
||||
# `nix build .#pxe` produces) and stages it where menu.ipxe's :nixos
|
||||
# entry expects it, so the pxe-boot host is self-contained — no manual
|
||||
# operator step to populate /srv/pxe/http/nixos after deploy.
|
||||
systemd.tmpfiles.rules = [
|
||||
# Declared here too (not just in build-types/pxe-boot.nix) so this
|
||||
# module's C+ rules don't depend on cross-module list-merge ordering —
|
||||
# tmpfiles' C type needs the target directory to already exist.
|
||||
"d /srv/pxe/http/nixos 0755 root root -"
|
||||
"C+ /srv/pxe/http/nixos/${kernelFile} 0644 root root - ${netbootSystem.config.system.build.kernel}/${kernelFile}"
|
||||
"C+ /srv/pxe/http/nixos/initrd 0644 root root - ${netbootSystem.config.system.build.netbootRamdisk}/initrd"
|
||||
"C+ /srv/pxe/http/nixos/netboot.ipxe 0644 root root - ${netbootSystem.config.system.build.netbootIpxeScript}/netboot.ipxe"
|
||||
];
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
{ config, lib, vars, ... }:
|
||||
{ vars, ... }:
|
||||
|
||||
{
|
||||
fileSystems.${vars.raspiMountpoint} = {
|
||||
@@ -9,15 +9,6 @@
|
||||
"_netdev"
|
||||
"noatime"
|
||||
|
||||
# Explicitly use NFSv4.2 if supported
|
||||
"nfsvers=4.2"
|
||||
] ++ lib.optionals (!config.boot.isContainer) [
|
||||
# `x-systemd.automount` never works inside a Linux container (LXC
|
||||
# included) -- confirmed live on lxc-docker: systemd logs "Starting
|
||||
# of <unit>.automount unsupported" and never mounts it. `nofail`
|
||||
# above already keeps boot non-blocking there, so plain eager
|
||||
# mounting is fine.
|
||||
|
||||
# Don't mount until first access
|
||||
"x-systemd.automount"
|
||||
|
||||
@@ -26,6 +17,9 @@
|
||||
|
||||
# Give the Pi/Tailscale a little time to appear
|
||||
"x-systemd.device-timeout=10s"
|
||||
|
||||
# Explicitly use NFSv4.2 if supported
|
||||
"nfsvers=4.2"
|
||||
];
|
||||
};
|
||||
|
||||
|
||||
@@ -0,0 +1,19 @@
|
||||
_:
|
||||
|
||||
{
|
||||
services.tailscale = {
|
||||
enable = true;
|
||||
|
||||
# extraSetFlags (tailscale set, via the always-on tailscaled-set
|
||||
# service), not extraUpFlags -- extraUpFlags is only ever applied by
|
||||
# tailscaled-autoconnect, which itself only runs when
|
||||
# services.tailscale.authKeyFile is set (nothing in this repo sets one,
|
||||
# so tailscale up is a manual, one-time operator step on every host that
|
||||
# uses this service). extraSetFlags has no such gate, so
|
||||
# --advertise-exit-node self-reapplies on every boot once the operator
|
||||
# has authenticated the node once.
|
||||
extraSetFlags = [
|
||||
"--advertise-exit-node"
|
||||
];
|
||||
};
|
||||
}
|
||||
@@ -1,35 +0,0 @@
|
||||
{ pkgs, ... }:
|
||||
|
||||
{
|
||||
imports = [ ./enable-service.nix ];
|
||||
|
||||
services.tailscale = {
|
||||
# Enables the sysctl forwarding settings subnet routers need;
|
||||
# without this, --advertise-routes has no effect.
|
||||
useRoutingFeatures = "server";
|
||||
|
||||
# Lets peers reach this node directly over the tailscale UDP port
|
||||
# instead of relaying through DERP.
|
||||
openFirewall = true;
|
||||
};
|
||||
|
||||
# Tailscale recommends these ethtool flags on the uplink interface to get
|
||||
# full UDP GRO throughput on subnet routers (https://tailscale.com/s/ethtool-config-udp-gro).
|
||||
# The interface is derived from the default route so it works regardless of
|
||||
# what the NIC is named on a given host.
|
||||
systemd.services.tailscale-udp-gro = {
|
||||
description = "Enable UDP GRO forwarding on uplink for Tailscale subnet router";
|
||||
after = [ "network-online.target" ];
|
||||
wants = [ "network-online.target" ];
|
||||
wantedBy = [ "multi-user.target" ];
|
||||
path = [ pkgs.ethtool pkgs.iproute2 ];
|
||||
serviceConfig = {
|
||||
Type = "oneshot";
|
||||
RemainAfterExit = true;
|
||||
ExecStart = pkgs.writeShellScript "tailscale-udp-gro" ''
|
||||
NETDEV=$(ip -o route get 8.8.8.8 | cut -f 5 -d " ")
|
||||
ethtool -K "$NETDEV" rx-udp-gro-forwarding on rx-gro-list off
|
||||
'';
|
||||
};
|
||||
};
|
||||
}
|
||||
@@ -1,60 +0,0 @@
|
||||
{ vars, ... }:
|
||||
|
||||
{
|
||||
# Run dnsmasq on the LAN interface as a forwarding-only resolver for
|
||||
# *.ts.net (Tailscale MagicDNS names). FreeIPA's bind-dyndb-ldap
|
||||
# cannot reach 100.100.100.100 (Tailscale's internal resolver) directly
|
||||
# because the DC is not a Tailscale node. This host IS a Tailscale node
|
||||
# and can reach 100.100.100.100 via its tailscale0 interface, so it
|
||||
# acts as an intermediary: FreeIPA has a conditional forward zone for
|
||||
# ts.net pointing here (vars.tailscaleRouterIp), and this dnsmasq
|
||||
# instance forwards those queries onward to Tailscale's resolver.
|
||||
#
|
||||
# Configure FreeIPA once after deploying this host:
|
||||
# kinit admin
|
||||
# ipa dnsforwardzone-add ${vars.tailnetDomain} \
|
||||
# --forwarder=${vars.tailscaleRouterIp} \
|
||||
# --forward-policy=only
|
||||
# Note: IPA refuses to shadow ts.net (a real public TLD); use the
|
||||
# tailnet-specific subdomain (vars.tailnetDomain) instead.
|
||||
services.dnsmasq = {
|
||||
enable = true;
|
||||
# NixOS's dnsmasq module defaults resolveLocalQueries to true, which adds
|
||||
# 127.0.0.1 to networking.nameservers and makes dnsmasq bind to
|
||||
# listen-address=127.0.0.1. This instance is not the host's local
|
||||
# resolver — it only serves IPA's conditional forwarder for tailnet names.
|
||||
# The host uses domainControllerIp directly (networking.nameservers in
|
||||
# host.nix). Without this, all host DNS goes through dnsmasq, which has
|
||||
# no upstream for general queries (no-resolv=true), breaking resolution.
|
||||
resolveLocalQueries = false;
|
||||
settings = {
|
||||
# Listen only on the LAN interface — not tailscale0 or loopback.
|
||||
# bind-interfaces prevents dnsmasq from binding to 0.0.0.0 and
|
||||
# then filtering by interface later; combined with `interface` this
|
||||
# ensures it genuinely listens only on eth0.
|
||||
bind-interfaces = true;
|
||||
interface = [ vars.lxcLanInterface ];
|
||||
|
||||
# Forward-only: no local /etc/hosts or /etc/resolv.conf reading,
|
||||
# no negative caching of NXDOMAIN for names this instance doesn't
|
||||
# serve. All ts.net queries come from FreeIPA's conditional forwarder
|
||||
# and must be answered by Tailscale's resolver.
|
||||
no-hosts = true;
|
||||
no-resolv = true;
|
||||
|
||||
# Tailscale's internal "Quad100" resolver — reachable from any
|
||||
# Tailscale node via the tailscale0 interface. Scoped to the
|
||||
# specific tailnet subdomain (vars.tailnetDomain) rather than
|
||||
# all of ts.net: FreeIPA refuses to shadow ts.net (a real public
|
||||
# TLD with DNSimple nameservers) so the conditional forward zone
|
||||
# in FreeIPA must use the tailnet-specific subdomain instead:
|
||||
# ipa dnsforwardzone-add ${vars.tailnetDomain} \
|
||||
# --forwarder=${vars.tailscaleRouterIp} \
|
||||
# --forward-policy=only
|
||||
server = [ "/${vars.tailnetDomain}/100.100.100.100" ];
|
||||
};
|
||||
};
|
||||
|
||||
networking.firewall.allowedUDPPorts = [ 53 ];
|
||||
networking.firewall.allowedTCPPorts = [ 53 ];
|
||||
}
|
||||
@@ -1,35 +0,0 @@
|
||||
{ pkgs, vars, ... }:
|
||||
|
||||
{
|
||||
services.tor = {
|
||||
enable = true;
|
||||
|
||||
# Opens settings.ORPort (and DirPort, unset here) in the firewall —
|
||||
# see the nixpkgs tor module's own networking.firewall.mkIf block.
|
||||
openFirewall = true;
|
||||
|
||||
relay = {
|
||||
enable = true;
|
||||
# Plain middle/guard relay, not "exit" — relays onion traffic between
|
||||
# other Tor nodes without ever making requests to the public internet
|
||||
# on a user's behalf, avoiding the abuse complaints and legal exposure
|
||||
# an exit node invites.
|
||||
role = "relay";
|
||||
};
|
||||
|
||||
settings.ORPort = vars.ports.torRelayOrPort;
|
||||
|
||||
# Unix control socket at /run/tor/control (GroupWritable, group "tor")
|
||||
# -- what nyx below actually monitors the relay through. Nyx's own
|
||||
# default control-socket path (/var/run/tor/control) resolves to the
|
||||
# same place, so no extra nyx config is needed.
|
||||
controlSocket.enable = true;
|
||||
};
|
||||
|
||||
# Lets the primary user's shell session read/write the control socket
|
||||
# above without being root -- otherwise nyx fails to authenticate against
|
||||
# it at all.
|
||||
users.users.${vars.primaryUser}.extraGroups = [ "tor" ];
|
||||
|
||||
environment.systemPackages = [ pkgs.nyx ];
|
||||
}
|
||||
Binary file not shown.
@@ -0,0 +1,134 @@
|
||||
# Spec: Remove Sensitive Information from NixOS Flake
|
||||
|
||||
## Goal
|
||||
|
||||
Every secret currently readable in plaintext anywhere in this repo (working tree *and* git history) gets removed, replaced with `sops-nix`-managed encrypted references, and rotated. When this is done, the repo should be safe to make public without exposing anything about the systems it configures.
|
||||
|
||||
Treat this as three sequential milestones. Do not start git history rewriting (Milestone 3) until Milestones 1 and 2 are fully verified and the flake still builds. This should be its own branch (`refactor/secrets`) until fully verified, then merged.
|
||||
|
||||
---
|
||||
|
||||
## Milestone 1 — Audit
|
||||
|
||||
Before touching anything, produce a complete inventory. Do not guess at scope — grep the whole tree and the whole history.
|
||||
|
||||
1. Run a secret scanner across the working tree and full history. Use both, since they catch different things:
|
||||
- `gitleaks detect --source . -v --log-opts="--all"` (scans history too)
|
||||
- `trufflehog git file://. --since-commit=$(git rev-list --max-parents=0 HEAD) --only-verified=false`
|
||||
If neither is installed, add them via a temporary `nix-shell -p gitleaks trufflehog` — don't install anything globally on the host.
|
||||
|
||||
2. Manually grep for the categories below, since scanners miss config-specific patterns:
|
||||
- `hashedPassword`, `password`, `initialPassword`, `initialHashedPassword` in any `users.users.*` block
|
||||
- `age.secrets`, `sops.secrets` (if any partial secrets work already exists — check for it)
|
||||
- PSK / `preSharedKey`, `privateKeyFile` inline values (vs. file references) for WireGuard
|
||||
- `authKey`, `apiToken`, `api_key`, `token =`, `secret =` in service modules (Tailscale, Cloudflare, backup tools, etc.)
|
||||
- SSH private key material: search for `BEGIN OPENSSH PRIVATE KEY` / `BEGIN RSA PRIVATE KEY` literals
|
||||
- TLS cert/key pairs committed under e.g. `secrets/`, `certs/`, `pki/`
|
||||
- Real name, personal email, home address, or anything in comments/hostnames that maps a machine to your physical identity or network layout (e.g. hostnames like `wayne-desktop`, static LAN IPs, ISP-identifying info)
|
||||
- `.env` files, `secrets.nix`, `secrets.yaml`, or any file that looks like it was meant to be gitignored but wasn't
|
||||
|
||||
3. Produce `secrets-inventory.md` (temporary, delete before finishing) listing: file path, line, secret type, and which host/service it belongs to. This becomes the checklist for Milestone 2 — every row must be either migrated to sops or deleted, with nothing left unaccounted for.
|
||||
|
||||
---
|
||||
|
||||
## Milestone 2 — Migrate to sops-nix
|
||||
|
||||
### 2.1 Set up sops-nix
|
||||
|
||||
1. Add the flake input:
|
||||
```nix
|
||||
sops-nix.url = "github:Mic92/sops-nix";
|
||||
sops-nix.inputs.nixpkgs.follows = "nixpkgs";
|
||||
```
|
||||
2. Import `sops-nix.nixosModules.sops` into each host's module list (or into a shared `common.nix` if all hosts use it).
|
||||
3. Generate an age keypair **per host** (not one shared key for everything — a compromised host shouldn't decrypt every other host's secrets):
|
||||
```
|
||||
nix-shell -p age --run "age-keygen -o /var/lib/sops-nix/key.txt"
|
||||
```
|
||||
Print the public key (`age-keygen -y`) for each host — you'll need it for `.sops.yaml`.
|
||||
4. Also generate one age key for yourself (your admin workstation) so you can edit secrets without needing to SSH into a host: store it at `~/.config/sops/age/keys.txt`, back it up somewhere outside this repo (password manager, offline). **If this key is lost, every secret encrypted with it is unrecoverable — losing the age key is equivalent to losing the secrets.**
|
||||
5. Create `.sops.yaml` at the repo root defining creation rules: which age public keys can decrypt which secrets files, keyed by path regex, so e.g. `secrets/hostA.yaml` is decryptable by your admin key + hostA's key, `secrets/hostB.yaml` by your admin key + hostB's key.
|
||||
|
||||
### 2.2 Migrate each secret category from the inventory
|
||||
|
||||
For each row in `secrets-inventory.md`:
|
||||
|
||||
- **Password hashes**: generate hash with `mkpasswd -m sha-512` (or `bcrypt` if your setup wants that), store under `sops.secrets."<name>/hashedPassword"`, reference via `users.users.<name>.hashedPasswordFile = config.sops.secrets."<name>/hashedPassword".path;`. Do not put the *plaintext* password anywhere, only the hash, and only the hash goes into the encrypted sops file.
|
||||
- **API tokens / auth keys**: move the raw value into the per-host sops YAML, reference in the module via `config.sops.secrets."<service>/token".path` — most NixOS service modules that take a token also accept a `*File` variant (e.g. `environmentFile`, `tokenFile`); use that instead of passing the value directly.
|
||||
- **Private keys / certs**: move the PEM/key content wholesale into a sops secret, output as a file with appropriate `sops.secrets.<name>.path`, `owner`, `mode`, `restartUnits` so the depending service (sshd, wireguard, nginx) reloads when the secret changes.
|
||||
- **Personal/identifying info**: this doesn't belong in sops (it's not "secret," it's just information you don't want public). Replace real names/emails with placeholders or move to a small untracked `local.nix` that's `.gitignore`'d and imported conditionally, with a documented template (`local.nix.example`) committed instead.
|
||||
|
||||
### 2.3 Verify before moving on
|
||||
|
||||
- `nixos-rebuild dry-build --flake .#<host>` succeeds for every host.
|
||||
- `sudo nixos-rebuild switch --flake .#<host>` on at least one real machine (or a VM) confirms secrets decrypt and services start.
|
||||
- Confirm decrypted secrets land under `/run/secrets/` (not the Nix store — anything placed in `/nix/store` is world-readable by design, so sops-nix's runtime-only placement is the whole point; double check no module accidentally pulls a secret path into a store-built config file).
|
||||
- Re-run the grep/scanner sweep from Milestone 1 against the *working tree only* (not history yet) — it should now come back clean.
|
||||
|
||||
---
|
||||
|
||||
## Milestone 3 — Scrub git history
|
||||
|
||||
Do this only after Milestone 2 is merged to your main branch and confirmed working, since it rewrites every commit SHA from the point of the earliest offending commit onward.
|
||||
|
||||
**This is destructive and irreversible on your local clone. Back up first:**
|
||||
```
|
||||
cp -r /path/to/nixos-repo /path/to/nixos-repo-backup-$(date +%F)
|
||||
```
|
||||
|
||||
1. Install `git-filter-repo` (not the older `git filter-branch` / BFG — filter-repo is the currently maintained, faster, safer tool):
|
||||
```
|
||||
nix-shell -p git-filter-repo
|
||||
```
|
||||
2. Use the `secrets-inventory.md` list to build a list of literal strings/paths to strip. Two approaches, use both:
|
||||
- Path-based: if whole files were secret (e.g. `secrets.nix`, a `.env`, a private key file), remove them entirely from history:
|
||||
```
|
||||
git filter-repo --path secrets.nix --path .env --invert-paths
|
||||
```
|
||||
- Value-based: for secrets embedded inline in files you're keeping (not deleting the whole file), use `--replace-text` with a file listing each literal secret string to replace with `***REMOVED***`:
|
||||
```
|
||||
git filter-repo --replace-text expressions.txt
|
||||
```
|
||||
3. After filtering, verify: run the Milestone 1 scanners again against full history (`--log-opts="--all"`). They must come back clean.
|
||||
4. Force-push the rewritten history:
|
||||
```
|
||||
git push origin --force --all
|
||||
git push origin --force --tags
|
||||
```
|
||||
5. **Every other clone of this repo (other machines, WSL instances, CI) must be deleted and re-cloned fresh** — a `git pull` against rewritten history will not work cleanly and risks resurrecting the old commits. Don't try to reconcile old clones; throw them away and re-clone.
|
||||
6. If this repo has ever been pushed to a public host (GitHub, etc.) or a fork/mirror exists, treat every secret that was ever in history as **permanently compromised regardless of the rewrite** — caches, forks, and Wayback-style archives can retain old commits indefinitely. History scrubbing prevents *future* exposure via `git clone`; it does not undo past exposure.
|
||||
|
||||
---
|
||||
|
||||
## Milestone 4 — Rotate everything
|
||||
|
||||
Because the secrets were exposed in history (even briefly, even in a private repo), the migration is not complete until every credential in the inventory has been **rotated**, not just re-encrypted. Re-encrypting an already-leaked value protects it going forward but doesn't undo the leak.
|
||||
|
||||
For each row in the original inventory:
|
||||
- Password hashes → change the actual account password, regenerate the hash, update the sops file.
|
||||
- API tokens/auth keys → revoke the old token in the issuing service's dashboard (Cloudflare, Tailscale, backup provider, etc.) and generate a new one.
|
||||
- SSH/WireGuard private keys → generate new keypairs, update the corresponding public key wherever it's trusted (authorized_keys, peer configs, etc.), retire the old ones.
|
||||
- TLS certs → reissue if the private key was exposed.
|
||||
|
||||
Keep `secrets-inventory.md` open during this step and check off each row as rotated. Delete the file only once every row is checked off — it should not be committed.
|
||||
|
||||
---
|
||||
|
||||
## Ongoing prevention
|
||||
|
||||
Add a pre-commit hook (or a `nix flake check` step) running `gitleaks protect --staged` so a secret can't be committed again by accident. Document in the repo README (briefly) that new secrets go through `sops <file>` to edit, never as plaintext in a tracked file.
|
||||
|
||||
---
|
||||
|
||||
## Definition of done
|
||||
|
||||
- [ ] Milestone 1 inventory complete and reviewed
|
||||
- [ ] All hosts have per-host age keys; admin key backed up outside the repo
|
||||
- [ ] Every inventoried secret migrated to sops-nix, referenced via `*File`/`sops.secrets.*.path`, nothing plaintext in the working tree
|
||||
- [ ] `nixos-rebuild dry-build` and at least one real `switch` verified per host
|
||||
- [ ] Working-tree scanner sweep clean
|
||||
- [ ] History rewritten with `git-filter-repo`, force-pushed, full-history scanner sweep clean
|
||||
- [ ] All other clones deleted and re-cloned from the rewritten history
|
||||
- [ ] Every credential in the original inventory rotated (not just re-encrypted)
|
||||
- [ ] Pre-commit secret scanning hook added
|
||||
- [ ] `secrets-inventory.md` deleted from the working directory (never committed)
|
||||
@@ -129,6 +129,5 @@ echo "flake.lock still points at the old input revisions until refreshed. Either
|
||||
echo " nix flake update nixpkgs home-manager # just these two inputs"
|
||||
echo " nix flake update # everything — see docs/flake-lock-automation.md"
|
||||
echo
|
||||
echo "Then run 'bash scripts/codex-maintenance.sh --full-check --dry-run' before"
|
||||
echo "committing — a channel bump can shift option defaults across every host,"
|
||||
echo "and only --dry-run actually builds anything to catch that."
|
||||
echo "Then run 'bash scripts/codex-maintenance.sh dry-run' before committing —"
|
||||
echo "a channel bump can shift option defaults across every host."
|
||||
|
||||
+59
-267
@@ -1,73 +1,22 @@
|
||||
#!/usr/bin/env bash
|
||||
# Validation entry point for CI and local/agent review.
|
||||
#
|
||||
# Default mode (what CI runs on every push/PR): fmt-check, statix, and eval
|
||||
# are scoped to files that actually changed against a base ref, plus
|
||||
# whichever hosts/packages those changes can affect. This exists because
|
||||
# the unscoped sweep below is slow enough to time out CI runners -- see
|
||||
# --full-check.
|
||||
#
|
||||
# --full-check: the historical full sweep (every host, every package,
|
||||
# fmt --check ./statix check . over the whole tree). Slow -- minutes, not
|
||||
# seconds. CI never passes this; run it locally before a release or after
|
||||
# touching modules/common/*, flake.nix, or variables.nix if you want extra
|
||||
# confidence beyond what the changed-files scope already covers for those
|
||||
# paths (see below).
|
||||
#
|
||||
# --dry-run: adds `nix build --dry-run --no-link` for whatever scope is
|
||||
# active (changed-files scope by default, full scope under --full-check).
|
||||
#
|
||||
# Per-host/per-package eval and dry-run build calls run concurrently (see
|
||||
# scripts/lib/nix-parallel.sh) since they're independent of each other.
|
||||
# Concurrency defaults to core count capped by available memory (~1GB/job)
|
||||
# rather than plain core count, since each concurrent `nix eval` evaluates a
|
||||
# whole NixOS system closure and can OOM a small/memory-constrained CI
|
||||
# runner otherwise; override via NIX_PARALLEL_JOBS if a runner has more (or
|
||||
# less) room than that estimate assumes.
|
||||
set -euo pipefail
|
||||
|
||||
script_dir="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
# shellcheck source=lib/nix-bootstrap.sh
|
||||
source "${script_dir}/lib/nix-bootstrap.sh"
|
||||
# shellcheck source=lib/nix-eval.sh
|
||||
source "${script_dir}/lib/nix-eval.sh"
|
||||
# shellcheck source=lib/nix-parallel.sh
|
||||
source "${script_dir}/lib/nix-parallel.sh"
|
||||
export NIX_CONFIG="${NIX_CONFIG:-}
|
||||
experimental-features = nix-command flakes
|
||||
accept-flake-config = false
|
||||
warn-dirty = false
|
||||
"
|
||||
|
||||
repo_root="$(cd "${script_dir}/.." && pwd)"
|
||||
cd "$repo_root"
|
||||
MODE="${1:-validate}"
|
||||
|
||||
full_check=false
|
||||
dry_run=false
|
||||
|
||||
usage() {
|
||||
cat <<'EOF'
|
||||
Usage: scripts/codex-maintenance.sh [--full-check] [--dry-run]
|
||||
|
||||
--full-check Run the full sweep: fmt-check and statix over the whole
|
||||
repo, eval every host and package. Slow. Never run by CI.
|
||||
--dry-run Additionally run `nix build --dry-run --no-link` for
|
||||
whatever scope is active.
|
||||
|
||||
With neither flag (the CI default), fmt-check/statix/eval are scoped to
|
||||
files changed against a base ref (env MAINT_BASE_SHA, else the PR base,
|
||||
else HEAD^), plus the hosts/packages those changes can affect.
|
||||
EOF
|
||||
ensure_nix_profile() {
|
||||
if [ -f /nix/var/nix/profiles/default/etc/profile.d/nix-daemon.sh ]; then
|
||||
. /nix/var/nix/profiles/default/etc/profile.d/nix-daemon.sh
|
||||
elif [ -f "$HOME/.nix-profile/etc/profile.d/nix.sh" ]; then
|
||||
. "$HOME/.nix-profile/etc/profile.d/nix.sh"
|
||||
fi
|
||||
}
|
||||
|
||||
for arg in "$@"; do
|
||||
case "$arg" in
|
||||
--full-check) full_check=true ;;
|
||||
--dry-run) dry_run=true ;;
|
||||
-h|--help) usage; exit 0 ;;
|
||||
*)
|
||||
echo "Unknown argument: $arg" >&2
|
||||
usage >&2
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
done
|
||||
|
||||
ensure_nix_profile
|
||||
|
||||
if ! command -v nix >/dev/null 2>&1; then
|
||||
@@ -75,6 +24,13 @@ if ! command -v nix >/dev/null 2>&1; then
|
||||
exit 127
|
||||
fi
|
||||
|
||||
hosts_json="$(nix eval --json --no-use-registries --no-accept-flake-config .#nixosConfigurations --apply builtins.attrNames)"
|
||||
hosts="$(echo "$hosts_json" | jq -r '.[]')"
|
||||
|
||||
echo "Hosts:"
|
||||
echo "$hosts"
|
||||
|
||||
echo
|
||||
echo "Checking for obvious committed secrets..."
|
||||
if grep -RInE 'github_pat_|ghp_|access-tokens|hashedPassword[[:space:]]*=' \
|
||||
--exclude-dir=.git \
|
||||
@@ -86,235 +42,71 @@ else
|
||||
echo "No obvious token patterns found."
|
||||
fi
|
||||
|
||||
mapfile -t all_hosts < <(list_flake_targets .)
|
||||
mapfile -t all_packages < <(nix eval --json "${NIX_EVAL_FLAGS[@]}" .#packages.x86_64-linux --apply builtins.attrNames | jq -r '.[]')
|
||||
|
||||
# host_targets_for_dir <hosts-subdir-name>
|
||||
# Prints the nixosConfigurations target names whose hostPath is
|
||||
# ./hosts/<dir>/host.nix, derived straight from flake.nix's generatedTargets
|
||||
# (one mkTarget { ... } call per line) rather than a hand-maintained table,
|
||||
# so it can't drift the way a copied mapping would.
|
||||
host_targets_for_dir() {
|
||||
local dir="$1"
|
||||
grep -oE '^[[:space:]]*[A-Za-z0-9_-]+ = mkTarget \{[^}]*hostPath = \./hosts/'"${dir}"'/host\.nix;[^}]*\};' flake.nix \
|
||||
| sed -E 's/^[[:space:]]*([A-Za-z0-9_-]+) = mkTarget.*/\1/' \
|
||||
|| true
|
||||
}
|
||||
|
||||
declare -a changed_files=()
|
||||
scope_desc="full repo"
|
||||
|
||||
if ! $full_check; then
|
||||
resolve_base_ref() {
|
||||
if [[ -n "${MAINT_BASE_SHA:-}" ]] && git cat-file -e "${MAINT_BASE_SHA}^{commit}" 2>/dev/null; then
|
||||
echo "$MAINT_BASE_SHA"
|
||||
return
|
||||
fi
|
||||
if git rev-parse --verify -q HEAD^ >/dev/null 2>&1; then
|
||||
echo "HEAD^"
|
||||
return
|
||||
fi
|
||||
git hash-object -t tree /dev/null
|
||||
}
|
||||
|
||||
base_ref="$(resolve_base_ref)"
|
||||
echo
|
||||
echo "Changed-files scope: diffing against ${base_ref}"
|
||||
mapfile -t changed_files < <(git diff --name-only --diff-filter=ACMR "$base_ref" -- . | sort -u)
|
||||
|
||||
if [[ ${#changed_files[@]} -eq 0 ]]; then
|
||||
echo "No changed files detected."
|
||||
else
|
||||
printf ' %s\n' "${changed_files[@]}"
|
||||
fi
|
||||
scope_desc="changed files only (base: ${base_ref})"
|
||||
fi
|
||||
|
||||
# Whole-tree fmt/lint always run under --full-check; otherwise scoped below.
|
||||
declare -a changed_nix_files=()
|
||||
for f in "${changed_files[@]:-}"; do
|
||||
[[ "$f" == *.nix && -f "$f" ]] && changed_nix_files+=("$f")
|
||||
done
|
||||
|
||||
echo
|
||||
echo "Checking Nix formatting with nixpkgs-fmt..."
|
||||
if $full_check; then
|
||||
nix run "${NIX_EVAL_FLAGS[@]}" github:NixOS/nixpkgs/nixos-25.11#nixpkgs-fmt -- --check .
|
||||
elif [[ ${#changed_nix_files[@]} -gt 0 ]]; then
|
||||
nix run "${NIX_EVAL_FLAGS[@]}" github:NixOS/nixpkgs/nixos-25.11#nixpkgs-fmt -- --check "${changed_nix_files[@]}"
|
||||
else
|
||||
echo "No changed .nix files; skipping."
|
||||
fi
|
||||
nix run --no-use-registries --no-accept-flake-config github:NixOS/nixpkgs/nixos-25.11#nixpkgs-fmt -- --check .
|
||||
|
||||
echo
|
||||
echo "Running statix lint..."
|
||||
if $full_check; then
|
||||
nix run "${NIX_EVAL_FLAGS[@]}" github:NixOS/nixpkgs/nixos-25.11#statix -- check .
|
||||
elif [[ ${#changed_nix_files[@]} -gt 0 ]]; then
|
||||
for f in "${changed_nix_files[@]}"; do
|
||||
nix run "${NIX_EVAL_FLAGS[@]}" github:NixOS/nixpkgs/nixos-25.11#statix -- check "$f"
|
||||
done
|
||||
else
|
||||
echo "No changed .nix files; skipping."
|
||||
fi
|
||||
|
||||
# Figure out which hosts/packages this run needs to eval (and, under
|
||||
# --dry-run, build). full_check always means "everything"; otherwise a
|
||||
# change to flake.nix/flake.lock/variables.nix/modules/common/* (repo-wide
|
||||
# inputs) or to any other modules/*.nix outside platforms//build-types
|
||||
# (whose blast radius isn't safely inferable from the path alone -- see
|
||||
# CLAUDE.md's "Grep modules/build-types/*.nix for each build type's imports
|
||||
# list") also falls back to everything, on the same reasoning CLAUDE.md
|
||||
# already gives interactive sessions for when to run the full sweep.
|
||||
# Anything more targeted -- a host.nix, a platform module, a build-type
|
||||
# module -- narrows to just the hosts it can affect.
|
||||
declare -A affected_hosts=()
|
||||
eval_packages=false
|
||||
|
||||
if $full_check; then
|
||||
for h in "${all_hosts[@]}"; do affected_hosts[$h]=1; done
|
||||
eval_packages=true
|
||||
else
|
||||
full_fallback=false
|
||||
for f in "${changed_files[@]:-}"; do
|
||||
case "$f" in
|
||||
flake.nix|flake.lock|variables.nix|modules/common/*)
|
||||
full_fallback=true
|
||||
;;
|
||||
esac
|
||||
done
|
||||
|
||||
if ! $full_fallback; then
|
||||
for f in "${changed_files[@]:-}"; do
|
||||
case "$f" in
|
||||
hosts/*/*)
|
||||
hostdir="${f#hosts/}"
|
||||
hostdir="${hostdir%%/*}"
|
||||
while IFS= read -r t; do
|
||||
[[ -n "$t" ]] && affected_hosts[$t]=1
|
||||
done < <(host_targets_for_dir "$hostdir")
|
||||
;;
|
||||
modules/platforms/*.nix)
|
||||
platform="$(basename "$f" .nix)"
|
||||
for h in "${all_hosts[@]}"; do
|
||||
[[ "$h" == "${platform}-"* ]] && affected_hosts[$h]=1
|
||||
done
|
||||
;;
|
||||
modules/build-types/*.nix)
|
||||
buildtype="$(basename "$f" .nix)"
|
||||
for h in "${all_hosts[@]}"; do
|
||||
[[ "$h" == *"-${buildtype}" ]] && affected_hosts[$h]=1
|
||||
done
|
||||
;;
|
||||
modules/installer/*)
|
||||
# iso.nix (imported by both the "installer" nixosConfigurations
|
||||
# target and netbootSystem, which backs packages.pxe) pulls in
|
||||
# common.nix, so a common.nix change reaches all three.
|
||||
affected_hosts[installer]=1
|
||||
eval_packages=true
|
||||
;;
|
||||
modules/pxe-boot/*)
|
||||
# stage-installer-artifacts.nix is imported by
|
||||
# modules/build-types/pxe-boot.nix only -- same blast radius as a
|
||||
# build-types/*.nix change, not a packages one.
|
||||
for h in "${all_hosts[@]}"; do
|
||||
[[ "$h" == *"-pxe-boot" ]] && affected_hosts[$h]=1
|
||||
done
|
||||
;;
|
||||
modules/*)
|
||||
full_fallback=true
|
||||
;;
|
||||
esac
|
||||
done
|
||||
fi
|
||||
|
||||
if $full_fallback; then
|
||||
echo
|
||||
echo "Changed files affect shared config; falling back to evaluating every host/package."
|
||||
for h in "${all_hosts[@]}"; do affected_hosts[$h]=1; done
|
||||
eval_packages=true
|
||||
fi
|
||||
fi
|
||||
|
||||
mapfile -t hosts < <(for h in "${!affected_hosts[@]}"; do echo "$h"; done | sort)
|
||||
nix run --no-use-registries --no-accept-flake-config github:NixOS/nixpkgs/nixos-25.11#statix -- check .
|
||||
|
||||
echo
|
||||
echo "Checking nix-cache host key for drift..."
|
||||
if bash "${script_dir}/secrets/sync-nix-cache-host-key.sh" --check; then
|
||||
:
|
||||
else
|
||||
drift_status=$?
|
||||
if [[ "$drift_status" -eq 2 ]]; then
|
||||
echo "nix-cache unreachable from here -- skipping host-key drift check."
|
||||
else
|
||||
echo "WARNING: nix-cache's host key has drifted from variables.nix (see above)." >&2
|
||||
echo " Run 'bash scripts/secrets/sync-nix-cache-host-key.sh' to fix." >&2
|
||||
fi
|
||||
fi
|
||||
echo "Evaluating host toplevel derivations..."
|
||||
for host in $hosts; do
|
||||
echo "==> $host"
|
||||
nix eval --raw --no-use-registries --no-accept-flake-config ".#nixosConfigurations.${host}.config.system.build.toplevel.drvPath"
|
||||
|
||||
echo
|
||||
if [[ ${#hosts[@]} -eq 0 ]]; then
|
||||
echo "No hosts affected by changed files; skipping host eval."
|
||||
else
|
||||
echo "Evaluating host toplevel derivations (${scope_desc}, up to ${NIX_PARALLEL_JOBS} at a time)..."
|
||||
# lxc-* hosts deploy via a directly pct-restore-able tarball instead of
|
||||
# nixos-install (see docs/auto-installer.md); proxmox-* hosts can
|
||||
# alternatively be built as a standalone disk image (see
|
||||
# docs/proxmox-images.md). Both are otherwise-unvalidated buildable
|
||||
# surface, easy to silently break without this.
|
||||
declare -a host_eval_jobs=()
|
||||
for host in "${hosts[@]}"; do
|
||||
host_eval_jobs+=("${host}${NIX_PARALLEL_SEP}.#nixosConfigurations.${host}.config.system.build.toplevel.drvPath")
|
||||
case "$host" in
|
||||
lxc-*)
|
||||
host_eval_jobs+=("${host} (tarball)${NIX_PARALLEL_SEP}.#nixosConfigurations.${host}.config.system.build.tarball.drvPath")
|
||||
;;
|
||||
proxmox-*)
|
||||
host_eval_jobs+=("${host} (diskoImagesScript)${NIX_PARALLEL_SEP}.#nixosConfigurations.${host}.config.system.build.diskoImagesScript.drvPath")
|
||||
;;
|
||||
esac
|
||||
done
|
||||
run_nix_parallel host_eval_jobs eval --raw "${NIX_EVAL_FLAGS[@]}"
|
||||
fi
|
||||
case "$host" in
|
||||
lxc-*)
|
||||
echo "==> $host (tarball)"
|
||||
nix eval --raw --no-use-registries --no-accept-flake-config ".#nixosConfigurations.${host}.config.system.build.tarball.drvPath"
|
||||
;;
|
||||
proxmox-*)
|
||||
echo "==> $host (diskoImagesScript)"
|
||||
nix eval --raw --no-use-registries --no-accept-flake-config ".#nixosConfigurations.${host}.config.system.build.diskoImagesScript.drvPath"
|
||||
;;
|
||||
esac
|
||||
done
|
||||
|
||||
echo
|
||||
if ! $eval_packages; then
|
||||
echo "No packages affected by changed files; skipping package eval."
|
||||
else
|
||||
echo "Evaluating buildable packages (up to ${NIX_PARALLEL_JOBS} at a time)..."
|
||||
declare -a package_eval_jobs=()
|
||||
for pkg in "${all_packages[@]}"; do
|
||||
package_eval_jobs+=("packages.x86_64-linux.${pkg}${NIX_PARALLEL_SEP}.#packages.x86_64-linux.${pkg}")
|
||||
done
|
||||
run_nix_parallel package_eval_jobs eval --raw "${NIX_EVAL_FLAGS[@]}"
|
||||
fi
|
||||
echo "Evaluating buildable packages..."
|
||||
packages_json="$(nix eval --json --no-use-registries --no-accept-flake-config .#packages.x86_64-linux --apply builtins.attrNames)"
|
||||
packages="$(echo "$packages_json" | jq -r '.[]')"
|
||||
for pkg in $packages; do
|
||||
echo "==> packages.x86_64-linux.${pkg}"
|
||||
nix eval --raw --no-use-registries --no-accept-flake-config ".#packages.x86_64-linux.${pkg}"
|
||||
done
|
||||
|
||||
if $dry_run; then
|
||||
if [[ "$MODE" == "dry-run" ]]; then
|
||||
echo
|
||||
echo "Running dry-run builds for the active scope (up to ${NIX_PARALLEL_JOBS} at a time). This will not create result symlinks."
|
||||
declare -a host_build_jobs=()
|
||||
for host in "${hosts[@]:-}"; do
|
||||
host_build_jobs+=("Dry-run build: ${host}${NIX_PARALLEL_SEP}.#nixosConfigurations.${host}.config.system.build.toplevel")
|
||||
echo "Running dry-run builds for all hosts. This will not create result symlinks."
|
||||
for host in $hosts; do
|
||||
echo "==> Dry-run build: $host"
|
||||
nix build --dry-run --no-link --no-use-registries --no-accept-flake-config ".#nixosConfigurations.${host}.config.system.build.toplevel"
|
||||
|
||||
case "$host" in
|
||||
lxc-*)
|
||||
host_build_jobs+=("Dry-run build: ${host} (tarball)${NIX_PARALLEL_SEP}.#nixosConfigurations.${host}.config.system.build.tarball")
|
||||
echo "==> Dry-run build: $host (tarball)"
|
||||
nix build --dry-run --no-link --no-use-registries --no-accept-flake-config ".#nixosConfigurations.${host}.config.system.build.tarball"
|
||||
;;
|
||||
proxmox-*)
|
||||
host_build_jobs+=("Dry-run build: ${host} (diskoImagesScript)${NIX_PARALLEL_SEP}.#nixosConfigurations.${host}.config.system.build.diskoImagesScript")
|
||||
echo "==> Dry-run build: $host (diskoImagesScript)"
|
||||
nix build --dry-run --no-link --no-use-registries --no-accept-flake-config ".#nixosConfigurations.${host}.config.system.build.diskoImagesScript"
|
||||
;;
|
||||
esac
|
||||
done
|
||||
run_nix_parallel host_build_jobs build --dry-run --no-link "${NIX_EVAL_FLAGS[@]}"
|
||||
|
||||
if $eval_packages; then
|
||||
echo
|
||||
echo "Running dry-run builds for packages."
|
||||
declare -a package_build_jobs=()
|
||||
for pkg in "${all_packages[@]}"; do
|
||||
package_build_jobs+=("Dry-run build: packages.x86_64-linux.${pkg}${NIX_PARALLEL_SEP}.#packages.x86_64-linux.${pkg}")
|
||||
done
|
||||
run_nix_parallel package_build_jobs build --dry-run --no-link "${NIX_EVAL_FLAGS[@]}"
|
||||
fi
|
||||
echo
|
||||
echo "Running dry-run builds for all packages."
|
||||
for pkg in $packages; do
|
||||
echo "==> Dry-run build: packages.x86_64-linux.${pkg}"
|
||||
nix build --dry-run --no-link --no-use-registries --no-accept-flake-config ".#packages.x86_64-linux.${pkg}"
|
||||
done
|
||||
fi
|
||||
|
||||
echo
|
||||
|
||||
+22
-19
@@ -1,11 +1,19 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
script_dir="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
# shellcheck source=lib/nix-bootstrap.sh
|
||||
source "${script_dir}/lib/nix-bootstrap.sh"
|
||||
# shellcheck source=lib/nix-eval.sh
|
||||
source "${script_dir}/lib/nix-eval.sh"
|
||||
export NIX_CONFIG="${NIX_CONFIG:-}
|
||||
experimental-features = nix-command flakes
|
||||
accept-flake-config = false
|
||||
warn-dirty = false
|
||||
"
|
||||
|
||||
ensure_nix_profile() {
|
||||
if [ -f /nix/var/nix/profiles/default/etc/profile.d/nix-daemon.sh ]; then
|
||||
. /nix/var/nix/profiles/default/etc/profile.d/nix-daemon.sh
|
||||
elif [ -f "$HOME/.nix-profile/etc/profile.d/nix.sh" ]; then
|
||||
. "$HOME/.nix-profile/etc/profile.d/nix.sh"
|
||||
fi
|
||||
}
|
||||
|
||||
install_nix_if_missing() {
|
||||
if command -v nix >/dev/null 2>&1; then
|
||||
@@ -41,17 +49,6 @@ warn-dirty = false
|
||||
build-users-group = nixbld
|
||||
EOF
|
||||
|
||||
# The official installer's single-user root path still shells out to
|
||||
# `sudo` to create /nix even though it already knows it's running as
|
||||
# root -- confirmed live against a sudo-less minimal Debian/Proxmox
|
||||
# node, where it fails with "sudo: not found" and prints this exact
|
||||
# mkdir/chown as the manual fix. Pre-create it so that branch of the
|
||||
# installer is skipped entirely.
|
||||
if [ ! -d /nix ]; then
|
||||
mkdir -m 0755 /nix
|
||||
chown root /nix
|
||||
fi
|
||||
|
||||
sh <(curl -L https://nixos.org/nix/install) --no-daemon
|
||||
else
|
||||
sh <(curl -L https://nixos.org/nix/install) --no-daemon
|
||||
@@ -68,7 +65,6 @@ cat > "$HOME/.config/nix/nix.conf" <<'EOF'
|
||||
experimental-features = nix-command flakes
|
||||
accept-flake-config = false
|
||||
warn-dirty = false
|
||||
build-users-group =
|
||||
EOF
|
||||
|
||||
echo "Nix version:"
|
||||
@@ -83,6 +79,13 @@ if ! command -v jq >/dev/null 2>&1; then
|
||||
fi
|
||||
|
||||
echo "Available NixOS hosts:"
|
||||
list_flake_targets .
|
||||
hosts="$(nix eval --json --no-use-registries --no-accept-flake-config .#nixosConfigurations --apply builtins.attrNames | jq -r '.[]')"
|
||||
echo "$hosts"
|
||||
|
||||
echo "Codex setup complete. Run bash scripts/codex-maintenance.sh to validate changes."
|
||||
echo "Evaluating all host toplevel derivations..."
|
||||
for host in $hosts; do
|
||||
echo "==> Evaluating $host"
|
||||
nix eval --raw --no-use-registries --no-accept-flake-config ".#nixosConfigurations.${host}.config.system.build.toplevel.drvPath"
|
||||
done
|
||||
|
||||
echo "Codex setup complete."
|
||||
|
||||
Executable
+556
@@ -0,0 +1,556 @@
|
||||
#!/usr/bin/env bash
|
||||
# Creates new Proxmox VMs/LXC containers from this flake, and reconfigures
|
||||
# existing ones -- the manual workflows in docs/proxmox-images.md (VM) and
|
||||
# docs/auto-installer.md's "LXC hosts" section (container), automated.
|
||||
#
|
||||
# Usage:
|
||||
# scripts/create-proxmox-resource.sh --type lxc|vm --host <name> [options]
|
||||
# scripts/create-proxmox-resource.sh --type lxc|vm --list
|
||||
# scripts/create-proxmox-resource.sh --modify --vmid <n> [--cores N] [--memory MB] [--grow-disk GB]
|
||||
#
|
||||
# SAFETY:
|
||||
# - The default (create) mode only ever creates a NEW resource -- it
|
||||
# refuses to run if the target VMID already exists on the node, or if
|
||||
# a VM/CT identified as --host already exists under any other VMID
|
||||
# (checked live against the node; --allow-duplicate-host overrides).
|
||||
# - --modify only ever touches a resource you name explicitly via
|
||||
# --vmid, shows exactly what will change first, and (outside
|
||||
# --dry-run) always requires typing that VMID back to confirm before
|
||||
# anything is sent to the node. There is no bulk/implicit modify.
|
||||
# - Neither mode can start/stop/delete a resource. Not implemented on
|
||||
# purpose -- ask before adding it.
|
||||
#
|
||||
# See --help for the full option list.
|
||||
set -euo pipefail
|
||||
|
||||
repo_root="$(cd "$(dirname "$0")/.." && pwd)"
|
||||
# shellcheck source=env.sh
|
||||
source "${repo_root}/scripts/env.sh"
|
||||
|
||||
sync_keys="${repo_root}/scripts/sync-host-keys.sh"
|
||||
|
||||
usage() {
|
||||
cat <<EOF
|
||||
Usage: $0 --type lxc|vm --host <name> [options] (create)
|
||||
$0 --type lxc|vm --list (list --host values)
|
||||
$0 --modify --vmid <n> [options] (reconfigure)
|
||||
|
||||
Create mode (default):
|
||||
--type lxc|vm lxc = container, built as a CT template tarball.
|
||||
vm = VM, built as a Disko .raw disk image (UEFI/OVMF).
|
||||
--host <name> Which host identity to deploy -- matches
|
||||
config.networking.hostName (server, docker,
|
||||
nix-cache, nixos, pxe-boot, nix-minimal). Use
|
||||
--list to see what's available for --type.
|
||||
--name <name> Proxmox display name/hostname (default: --host's
|
||||
value, e.g. nix-cache -- for lxc this becomes the
|
||||
guest's real networking.hostName too, since
|
||||
proxmoxLXC.manageHostName pulls it from Proxmox's
|
||||
own container config, so it must match host.nix
|
||||
regardless of build type)
|
||||
--vmid <n> Numeric VMID (default: next free, via
|
||||
\`pvesh get /cluster/nextid\` on the node).
|
||||
Refuses to run if this ID already exists.
|
||||
--disk-size <GB> lxc only: rootfs size for \`pct create\`
|
||||
(default: \$PROXMOX_DEFAULT_LXC_DISK_GB, ${PROXMOX_DEFAULT_LXC_DISK_GB}).
|
||||
--image <path> Use this local image/tarball instead of
|
||||
checking the node / building one from the flake.
|
||||
--force-rebuild Skip the "does the node already have this
|
||||
image" check -- always build fresh and
|
||||
overwrite what's there.
|
||||
--allow-duplicate-host Required if a VM/CT identified as --host
|
||||
already exists on the node (checked live via
|
||||
qm/pct, not any file in this repo) --
|
||||
otherwise refused, since it'd share that
|
||||
host's hostName/hostId.
|
||||
|
||||
Modify mode (reconfigure an EXISTING resource -- requires --modify):
|
||||
--modify Switch to modify mode.
|
||||
--vmid <n> Required: which existing resource to change.
|
||||
Type/VM-vs-CT is auto-detected on the node.
|
||||
--grow-disk <GB> Grow the primary disk by this many GB
|
||||
(qm/pct resize; Proxmox only supports
|
||||
growing, never shrinking, an existing disk).
|
||||
At least one of --cores / --memory / --grow-disk is required. Always
|
||||
prints the current -> new values and requires typing the VMID back to
|
||||
confirm, even outside --dry-run.
|
||||
|
||||
Shared:
|
||||
--cores <n> create: default \$PROXMOX_DEFAULT_CORES (${PROXMOX_DEFAULT_CORES}).
|
||||
modify: omit to leave unchanged.
|
||||
--memory <MB> create: default \$PROXMOX_DEFAULT_MEMORY_MB (${PROXMOX_DEFAULT_MEMORY_MB}).
|
||||
modify: omit to leave unchanged.
|
||||
--swap <MB> lxc only, create time: \`--memory\` doesn't
|
||||
touch swap -- it silently stays at Proxmox's
|
||||
own 512M default otherwise. (default: matches
|
||||
whatever --memory resolves to)
|
||||
--storage <pool> (default: \$PROXMOX_STORAGE, ${PROXMOX_STORAGE})
|
||||
--iso-storage <pool> (default: \$PROXMOX_ISO_STORAGE, ${PROXMOX_ISO_STORAGE})
|
||||
--bridge <bridge> (default: \$PROXMOX_BRIDGE, ${PROXMOX_BRIDGE})
|
||||
--node <host> Proxmox node to SSH into (default:
|
||||
\$PROXMOX_HOST, ${PROXMOX_HOST})
|
||||
--dry-run Print the full plan; touch nothing
|
||||
local or remote, no prompts.
|
||||
-h, --help
|
||||
|
||||
Config for --storage/--bridge/--node/etc. lives in scripts/env.sh -- edit
|
||||
that instead of passing the same flag every time.
|
||||
EOF
|
||||
}
|
||||
|
||||
dry_run=0
|
||||
modify=0
|
||||
type=""
|
||||
host=""
|
||||
name=""
|
||||
vmid=""
|
||||
cores=""
|
||||
memory=""
|
||||
swap=""
|
||||
disk_size=""
|
||||
grow_disk=""
|
||||
image=""
|
||||
storage="$PROXMOX_STORAGE"
|
||||
iso_storage="$PROXMOX_ISO_STORAGE"
|
||||
bridge="$PROXMOX_BRIDGE"
|
||||
node="$PROXMOX_HOST"
|
||||
do_list=0
|
||||
allow_duplicate_host=0
|
||||
force_rebuild=0
|
||||
|
||||
while [[ $# -gt 0 ]]; do
|
||||
case "$1" in
|
||||
--type) type="$2"; shift 2 ;;
|
||||
--host) host="$2"; shift 2 ;;
|
||||
--name) name="$2"; shift 2 ;;
|
||||
--vmid) vmid="$2"; shift 2 ;;
|
||||
--cores) cores="$2"; shift 2 ;;
|
||||
--memory) memory="$2"; shift 2 ;;
|
||||
--swap) swap="$2"; shift 2 ;;
|
||||
--disk-size) disk_size="$2"; shift 2 ;;
|
||||
--grow-disk) grow_disk="$2"; shift 2 ;;
|
||||
--image) image="$2"; shift 2 ;;
|
||||
--storage) storage="$2"; shift 2 ;;
|
||||
--iso-storage) iso_storage="$2"; shift 2 ;;
|
||||
--bridge) bridge="$2"; shift 2 ;;
|
||||
--node) node="$2"; shift 2 ;;
|
||||
--list) do_list=1; shift ;;
|
||||
--allow-duplicate-host) allow_duplicate_host=1; shift ;;
|
||||
--force-rebuild) force_rebuild=1; shift ;;
|
||||
--modify) modify=1; shift ;;
|
||||
--dry-run) dry_run=1; shift ;;
|
||||
-h | --help) usage; exit 0 ;;
|
||||
*) echo "Unknown option: $1" >&2; usage >&2; exit 1 ;;
|
||||
esac
|
||||
done
|
||||
|
||||
ssh_target="${PROXMOX_SSH_USER}@${node}"
|
||||
|
||||
remote() {
|
||||
if [[ "$dry_run" -eq 1 ]]; then
|
||||
echo "[dry-run] ssh ${ssh_target} -- $*"
|
||||
else
|
||||
ssh "$ssh_target" "$@"
|
||||
fi
|
||||
}
|
||||
|
||||
# ============================================================ modify mode
|
||||
cmd_modify() {
|
||||
if [[ -z "$vmid" ]]; then
|
||||
echo "ERROR: --modify requires --vmid." >&2
|
||||
exit 1
|
||||
fi
|
||||
if [[ -z "$cores" && -z "$memory" && -z "$grow_disk" ]]; then
|
||||
echo "ERROR: --modify needs at least one of --cores / --memory / --grow-disk." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "Looking up VMID ${vmid} on ${node}..."
|
||||
local kind current_cores current_memory disk_key
|
||||
if ssh "$ssh_target" "qm status ${vmid}" >/dev/null 2>&1; then
|
||||
kind="vm"
|
||||
disk_key="scsi0"
|
||||
elif ssh "$ssh_target" "pct status ${vmid}" >/dev/null 2>&1; then
|
||||
kind="lxc"
|
||||
disk_key="rootfs"
|
||||
else
|
||||
echo "ERROR: VMID ${vmid} doesn't exist on ${node} -- nothing to modify." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
local config_cmd="qm config ${vmid}"
|
||||
[[ "$kind" == "lxc" ]] && config_cmd="pct config ${vmid}"
|
||||
local current_config
|
||||
current_config="$(ssh "$ssh_target" "$config_cmd")"
|
||||
current_cores="$(echo "$current_config" | grep -oP '^cores:\s*\K\S+' || echo '?')"
|
||||
current_memory="$(echo "$current_config" | grep -oP '^memory:\s*\K\S+' || echo '?')"
|
||||
|
||||
echo
|
||||
echo "VMID ${vmid} is a ${kind} on ${node}. Planned changes:"
|
||||
[[ -n "$cores" ]] && echo " cores: ${current_cores} -> ${cores}"
|
||||
[[ -n "$memory" ]] && echo " memory: ${current_memory} MB -> ${memory} MB"
|
||||
[[ -n "$grow_disk" ]] && echo " ${disk_key}: grow by +${grow_disk}G (Proxmox can only grow, not shrink, an existing disk)"
|
||||
|
||||
if [[ "$dry_run" -eq 1 ]]; then
|
||||
echo
|
||||
echo "[dry-run] Nothing was changed."
|
||||
return
|
||||
fi
|
||||
|
||||
echo
|
||||
read -rp "Type the VMID (${vmid}) to confirm these changes: " confirm
|
||||
if [[ "$confirm" != "$vmid" ]]; then
|
||||
echo "Cancelled -- input didn't match ${vmid}."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
local set_cmd="qm set"
|
||||
local resize_cmd="qm resize"
|
||||
[[ "$kind" == "lxc" ]] && set_cmd="pct set" && resize_cmd="pct resize"
|
||||
|
||||
if [[ -n "$cores" || -n "$memory" ]]; then
|
||||
local args=""
|
||||
[[ -n "$cores" ]] && args="${args} --cores ${cores}"
|
||||
[[ -n "$memory" ]] && args="${args} --memory ${memory}"
|
||||
remote "${set_cmd} ${vmid}${args}"
|
||||
fi
|
||||
if [[ -n "$grow_disk" ]]; then
|
||||
remote "${resize_cmd} ${vmid} ${disk_key} +${grow_disk}G"
|
||||
fi
|
||||
|
||||
echo
|
||||
echo "Done. VMID ${vmid} updated."
|
||||
}
|
||||
|
||||
if [[ "$modify" -eq 1 ]]; then
|
||||
cmd_modify
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# ============================================================= create mode
|
||||
if [[ "$type" != "lxc" && "$type" != "vm" ]]; then
|
||||
echo "ERROR: --type must be 'lxc' or 'vm'." >&2
|
||||
usage >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
platform_prefix="lxc"
|
||||
[[ "$type" == "vm" ]] && platform_prefix="proxmox"
|
||||
[[ -z "$cores" ]] && cores="$PROXMOX_DEFAULT_CORES"
|
||||
[[ -z "$memory" ]] && memory="$PROXMOX_DEFAULT_MEMORY_MB"
|
||||
|
||||
# --- discover / resolve the flake target from --host --------------------
|
||||
list_hosts() {
|
||||
local target hostname
|
||||
for target in $(nix eval --json --no-use-registries --no-accept-flake-config \
|
||||
"${repo_root}#nixosConfigurations" --apply builtins.attrNames 2>/dev/null \
|
||||
| jq -r --arg p "${platform_prefix}-" '.[] | select(startswith($p))'); do
|
||||
hostname="$(nix eval --raw --no-use-registries --no-accept-flake-config \
|
||||
"${repo_root}#nixosConfigurations.${target}.config.networking.hostName" 2>/dev/null)"
|
||||
printf ' %-12s -> %s\n' "$hostname" "$target"
|
||||
done
|
||||
}
|
||||
|
||||
if [[ "$do_list" -eq 1 ]]; then
|
||||
echo "Available --host values for --type ${type}:"
|
||||
list_hosts
|
||||
exit 0
|
||||
fi
|
||||
|
||||
if [[ -z "$host" ]]; then
|
||||
echo "ERROR: --host is required (or use --list to see options)." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
flake_target=""
|
||||
for target in $(nix eval --json --no-use-registries --no-accept-flake-config \
|
||||
"${repo_root}#nixosConfigurations" --apply builtins.attrNames \
|
||||
| jq -r --arg p "${platform_prefix}-" '.[] | select(startswith($p))'); do
|
||||
hn="$(nix eval --raw --no-use-registries --no-accept-flake-config \
|
||||
"${repo_root}#nixosConfigurations.${target}.config.networking.hostName")"
|
||||
if [[ "$hn" == "$host" ]]; then
|
||||
flake_target="$target"
|
||||
break
|
||||
fi
|
||||
done
|
||||
|
||||
if [[ -z "$flake_target" ]]; then
|
||||
echo "ERROR: no ${platform_prefix}-* target has hostName '${host}'." >&2
|
||||
echo "Available:" >&2
|
||||
list_hosts >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# The container/VM's real identity is --host (e.g. "nix-cache"), validated
|
||||
# above against config.networking.hostName -- not the flake target name
|
||||
# (e.g. "lxc-nix-cache"), which is build-type-specific and only exists to
|
||||
# pick which platform variant to build. Defaulting --name to the flake
|
||||
# target would make lxc's --hostname (which proxmoxLXC.manageHostName
|
||||
# feeds straight into the guest's real hostname) disagree with host.nix.
|
||||
[[ -z "$name" ]] && name="$host"
|
||||
|
||||
# --- refuse to duplicate a host that's already live on the node ---------
|
||||
# Queries the node itself (qm/pct's own name/hostname config), not any
|
||||
# static list in this repo -- a file can't track whether a resource still
|
||||
# actually exists, and this used to be checked against variables.nix's
|
||||
# deployedTargets, which drifted stale (it kept naming a VM as "the real
|
||||
# deployment" well after that VM had been destroyed, blocking its own
|
||||
# redeploy) until that list was dropped in favour of this live check. This
|
||||
# only catches guests identified with the default --name (== --host, what
|
||||
# this script itself always uses unless --name is overridden) -- a guest
|
||||
# manually renamed on the node afterwards wouldn't match, but nothing here
|
||||
# creates guests that way.
|
||||
if [[ "$allow_duplicate_host" -eq 1 ]]; then
|
||||
echo
|
||||
echo "--allow-duplicate-host: skipping the check for an existing '${host}' on ${node}."
|
||||
elif [[ "$dry_run" -eq 1 ]]; then
|
||||
echo
|
||||
echo "[dry-run] would check ${node} for an existing VM/CT identified as '${host}'"
|
||||
else
|
||||
echo
|
||||
echo "==> Checking ${node} for an existing VM/CT identified as '${host}'..."
|
||||
ssh_check_status=0
|
||||
existing="$(ssh "$ssh_target" bash -s -- "$host" <<'REMOTE_SCRIPT'
|
||||
target="$1"
|
||||
for id in $(qm list 2>/dev/null | awk 'NR>1{print $1}'); do
|
||||
n="$(qm config "$id" 2>/dev/null | grep -oP '^name:\s*\K\S+' || true)"
|
||||
[[ "$n" == "$target" ]] && echo "vm ${id} ${n}"
|
||||
done
|
||||
for id in $(pct list 2>/dev/null | awk 'NR>1{print $1}'); do
|
||||
n="$(pct config "$id" 2>/dev/null | grep -oP '^hostname:\s*\K\S+' || true)"
|
||||
[[ "$n" == "$target" ]] && echo "lxc ${id} ${n}"
|
||||
done
|
||||
REMOTE_SCRIPT
|
||||
)" || ssh_check_status=$?
|
||||
if [[ "$ssh_check_status" -ne 0 ]]; then
|
||||
echo "ERROR: couldn't reach ${node} (ssh exited ${ssh_check_status}) to check for an" >&2
|
||||
echo "existing '${host}' resource -- refusing to guess. Fix connectivity and retry," >&2
|
||||
echo "or pass --allow-duplicate-host if you're sure none exists (this skips the" >&2
|
||||
echo "check entirely)." >&2
|
||||
exit 1
|
||||
fi
|
||||
if [[ -n "$existing" ]]; then
|
||||
echo "ERROR: '${host}' already exists on ${node}:" >&2
|
||||
echo "$existing" | while read -r kind id n; do
|
||||
echo " - ${kind} VMID ${id} (${n})" >&2
|
||||
done
|
||||
echo "Refusing to create a second resource sharing this identity. Pass" >&2
|
||||
echo "--allow-duplicate-host to create one anyway (it gets its own distinct" >&2
|
||||
echo "sops key and VMID -- the existing resource(s) above are left untouched)," >&2
|
||||
echo "or use --modify to reconfigure the existing one instead." >&2
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
|
||||
echo "Target: ${flake_target} (host=${host}, type=${type}) -> Proxmox resource '${name}'"
|
||||
|
||||
# Decide on nix-cache once, here -- this is the earliest point that needs
|
||||
# it (sync-host-keys.sh below needs nix-shell packages regardless of
|
||||
# whether an image ends up getting built later), and the decision is
|
||||
# exported so that subprocess -- and this script's own later build step,
|
||||
# if it gets there -- both reuse it instead of probing again.
|
||||
nix_extra_opts
|
||||
|
||||
# --- make sure this target has a registered host key --------------------
|
||||
echo
|
||||
echo "==> Ensuring host key exists and is registered..."
|
||||
sync_args=("$flake_target")
|
||||
[[ "$dry_run" -eq 1 ]] && sync_args+=(--dry-run)
|
||||
bash "$sync_keys" "${sync_args[@]}"
|
||||
|
||||
# --- VMID: pick one, and refuse to touch anything that already exists ---
|
||||
echo
|
||||
if [[ -z "$vmid" ]]; then
|
||||
if [[ "$dry_run" -eq 1 ]]; then
|
||||
vmid="<next-free-vmid>"
|
||||
echo "[dry-run] would ask ${node} for the next free VMID (pvesh get /cluster/nextid)"
|
||||
else
|
||||
vmid="$(ssh "$ssh_target" "pvesh get /cluster/nextid" | tr -d '[:space:]')"
|
||||
echo "Auto-assigned VMID: ${vmid}"
|
||||
fi
|
||||
else
|
||||
echo "Requested VMID: ${vmid}"
|
||||
fi
|
||||
|
||||
if [[ "$dry_run" -eq 0 ]]; then
|
||||
# qm/pct status exits non-zero (and prints "does not exist") for a free
|
||||
# ID on that resource type -- but a VMID could exist as the OTHER
|
||||
# resource type (e.g. requested a CT id that's actually a VM), so check
|
||||
# both. Any success here means something is already using this ID --
|
||||
# refuse to go anywhere near it. (Reconfiguring an existing resource is
|
||||
# --modify's job, not this one's.)
|
||||
if ssh "$ssh_target" "qm status ${vmid}" >/dev/null 2>&1 \
|
||||
|| ssh "$ssh_target" "pct status ${vmid}" >/dev/null 2>&1; then
|
||||
echo "ERROR: VMID ${vmid} already exists on ${node}. Refusing to touch an" >&2
|
||||
echo "existing resource here -- use --modify to reconfigure it, pick a" >&2
|
||||
echo "different --vmid, or omit it to auto-assign." >&2
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
|
||||
# --- resolve the remote path -- fixed naming (not the nix store's own
|
||||
# derivation-hash-based filename), so a later run can check for it by name.
|
||||
# lxc uploads as a CT *template* (Proxmox's "vztmpl" content type, under
|
||||
# iso_storage) -- config.system.build.tarball is a plain rootfs tarball,
|
||||
# not a vzdump backup archive, so it's created with `pct create ... vztmpl`,
|
||||
# not restored with `pct restore` (that expects backup-archive metadata
|
||||
# this tarball doesn't have, and fails with "archive contains no
|
||||
# configuration file").
|
||||
remote_dir="/var/lib/vz/import"
|
||||
remote_filename="${flake_target}.raw"
|
||||
if [[ "$type" == "lxc" ]]; then
|
||||
remote_dir="/var/lib/vz/template/cache"
|
||||
remote_filename="${flake_target}.tar.xz"
|
||||
fi
|
||||
remote_path="${remote_dir}/${remote_filename}"
|
||||
|
||||
# --- build (or reuse an image already on the node) ------------------------
|
||||
echo
|
||||
local_image=""
|
||||
image_already_remote=0
|
||||
|
||||
if [[ -n "$image" ]]; then
|
||||
[[ -f "$image" ]] || { echo "ERROR: --image '${image}' not found." >&2; exit 1; }
|
||||
local_image="$image"
|
||||
echo "Using provided image: ${local_image}"
|
||||
elif [[ "$force_rebuild" -eq 1 ]]; then
|
||||
echo "--force-rebuild: skipping the existing-image check on ${node}."
|
||||
else
|
||||
echo "==> Checking whether ${node} already has ${remote_path}..."
|
||||
if [[ "$dry_run" -eq 1 ]]; then
|
||||
echo "[dry-run] would check: ssh ${ssh_target} -- test -f ${remote_path}"
|
||||
elif ssh "$ssh_target" "test -f '${remote_path}'" 2>/dev/null; then
|
||||
echo "Found it -- reusing, skipping build and upload (use --force-rebuild to override)."
|
||||
image_already_remote=1
|
||||
else
|
||||
echo "Not found -- will build."
|
||||
fi
|
||||
fi
|
||||
|
||||
if [[ "$image_already_remote" -eq 0 && -z "$local_image" ]]; then
|
||||
# Mirrors the real build commands' "${NIX_OPTS[@]}" below -- nix_extra_opts
|
||||
# (called earlier, once) has already decided whether nix-cache is in play,
|
||||
# and the dry-run preview needs to reflect that decision instead of always
|
||||
# printing the same command regardless of outcome.
|
||||
nix_opts_display=""
|
||||
if [[ ${#NIX_OPTS[@]} -gt 0 ]]; then
|
||||
printf -v nix_opts_display '%q ' "${NIX_OPTS[@]}"
|
||||
nix_opts_display=" ${nix_opts_display% }"
|
||||
fi
|
||||
if [[ "$type" == "lxc" ]]; then
|
||||
if [[ "$dry_run" -eq 1 ]]; then
|
||||
echo "[dry-run] would build: NIXOS_HOST_KEYS_DIR=${repo_root}/host-keys nix build --impure \\"
|
||||
echo "[dry-run] --no-use-registries --no-accept-flake-config${nix_opts_display} \\"
|
||||
echo "[dry-run] .#nixosConfigurations.${flake_target}.config.system.build.tarball"
|
||||
local_image="<built-tarball>"
|
||||
else
|
||||
echo "==> Building LXC tarball for ${flake_target}..."
|
||||
NIXOS_HOST_KEYS_DIR="${repo_root}/host-keys" nix build --impure \
|
||||
--no-use-registries --no-accept-flake-config "${NIX_OPTS[@]}" \
|
||||
".#nixosConfigurations.${flake_target}.config.system.build.tarball" \
|
||||
--out-link "${repo_root}/result-${flake_target}"
|
||||
local_image="$(find "${repo_root}/result-${flake_target}/tarball" -maxdepth 1 -type f | head -1)"
|
||||
echo "Built: ${local_image}"
|
||||
fi
|
||||
else
|
||||
if [[ "$dry_run" -eq 1 ]]; then
|
||||
echo "[dry-run] would build: nix build --no-use-registries --no-accept-flake-config${nix_opts_display} \\"
|
||||
echo "[dry-run] .#nixosConfigurations.${flake_target}.config.system.build.diskoImagesScript"
|
||||
echo "[dry-run] would run: sudo ./result-${flake_target} \\"
|
||||
echo "[dry-run] --pre-format-files host-keys/${flake_target}_ssh_host_ed25519_key /etc/ssh/ssh_host_ed25519_key \\"
|
||||
echo "[dry-run] --pre-format-files host-keys/${flake_target}_ssh_host_ed25519_key.pub /etc/ssh/ssh_host_ed25519_key.pub \\"
|
||||
echo "[dry-run] --build-memory 2048"
|
||||
local_image="<built-image>.raw"
|
||||
else
|
||||
echo "==> Building Disko image script for ${flake_target}..."
|
||||
nix build --no-use-registries --no-accept-flake-config "${NIX_OPTS[@]}" \
|
||||
".#nixosConfigurations.${flake_target}.config.system.build.diskoImagesScript" \
|
||||
--out-link "${repo_root}/result-${flake_target}"
|
||||
echo "==> Running it (builds the .raw image in a temporary QEMU VM, needs sudo)..."
|
||||
( cd "$repo_root" && sudo "./result-${flake_target}" \
|
||||
--pre-format-files "host-keys/${flake_target}_ssh_host_ed25519_key" /etc/ssh/ssh_host_ed25519_key \
|
||||
--pre-format-files "host-keys/${flake_target}_ssh_host_ed25519_key.pub" /etc/ssh/ssh_host_ed25519_key.pub \
|
||||
--build-memory 2048 )
|
||||
local_image="$(find "$repo_root" -maxdepth 1 -name "*.raw" -newer "${repo_root}/result-${flake_target}" | head -1)"
|
||||
if [[ -z "$local_image" ]]; then
|
||||
echo "ERROR: expected a .raw image after the build but didn't find one in ${repo_root}." >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "Built: ${local_image}"
|
||||
fi
|
||||
fi
|
||||
fi
|
||||
|
||||
# --- upload (skip entirely if reusing an image already on the node) ------
|
||||
echo
|
||||
if [[ "$image_already_remote" -eq 1 ]]; then
|
||||
: # nothing to upload
|
||||
elif [[ "$dry_run" -eq 1 ]]; then
|
||||
echo "[dry-run] would upload: scp ${local_image} ${ssh_target}:${remote_path}"
|
||||
else
|
||||
echo "==> Uploading to ${node}:${remote_path}..."
|
||||
ssh "$ssh_target" "mkdir -p ${remote_dir}"
|
||||
scp "$local_image" "${ssh_target}:${remote_path}"
|
||||
fi
|
||||
|
||||
# --- create -----------------------------------------------------------------
|
||||
echo
|
||||
if [[ "$type" == "lxc" ]]; then
|
||||
echo "==> Creating LXC container ${vmid} (${name})..."
|
||||
local_disk_size="${disk_size:-$PROXMOX_DEFAULT_LXC_DISK_GB}"
|
||||
# --memory doesn't touch swap -- it silently stays at Proxmox's own
|
||||
# 512M default otherwise (confirmed live: --memory 2048 left swap at
|
||||
# 512). Default to matching whatever --memory resolved to above.
|
||||
local_swap="${swap:-$memory}"
|
||||
# --unprivileged 1: modules/platforms/lxc.nix sets proxmoxLXC.privileged
|
||||
# = false, so the NixOS config inside the image assumes it's running as
|
||||
# an unprivileged container (cgroup/capability/mount expectations baked
|
||||
# in at boot). `pct create`'s own CLI default for this flag is
|
||||
# privileged (unlike the web UI, which defaults its checkbox the other
|
||||
# way) -- leaving it unset creates a privileged container running a
|
||||
# NixOS config that assumes unprivileged, a real mismatch.
|
||||
#
|
||||
# --features nesting=1,keyctl=1: required for a modern (v247+) systemd
|
||||
# guest to actually boot unprivileged -- confirmed live: without this,
|
||||
# AppArmor denies the nested user namespaces and credential mounts
|
||||
# systemd routinely uses (even plain getty units), and every getty
|
||||
# crash-loops on a denied mount every ~3s (visible as garbage on the
|
||||
# console) while core services like nsncd fail the same way.
|
||||
create_cmd="pct create ${vmid} ${iso_storage}:vztmpl/${remote_filename} --unprivileged 1 --features ${PROXMOX_DEFAULT_LXC_FEATURES} --rootfs ${storage}:${local_disk_size} --hostname ${name} --cores ${cores} --memory ${memory} --swap ${local_swap} --net0 name=eth0,bridge=${bridge},ip=dhcp"
|
||||
remote "$create_cmd"
|
||||
remote "pct start ${vmid}"
|
||||
else
|
||||
echo "==> Creating VM ${vmid} (${name})..."
|
||||
# pre-enrolled-keys=0 disables OVMF's Secure Boot key pre-enrollment --
|
||||
# required, or systemd-boot (unsigned) can't be trusted by the firmware.
|
||||
remote "qm create ${vmid} --name ${name} --memory ${memory} --cores ${cores} \
|
||||
--net0 virtio,bridge=${bridge} --bios ovmf --machine q35 --scsihw virtio-scsi-pci \
|
||||
--efidisk0 ${storage}:1,efitype=4m,pre-enrolled-keys=0"
|
||||
|
||||
if [[ "$dry_run" -eq 1 ]]; then
|
||||
echo "[dry-run] ssh ${ssh_target} -- qm importdisk ${vmid} ${remote_path} ${storage}"
|
||||
echo "[dry-run] (would parse the resulting disk identifier from that output)"
|
||||
echo "[dry-run] ssh ${ssh_target} -- qm set ${vmid} --scsi0 ${storage}:<parsed-disk-id>"
|
||||
else
|
||||
importdisk_output="$(ssh "$ssh_target" "qm importdisk ${vmid} ${remote_path} ${storage}")"
|
||||
echo "$importdisk_output"
|
||||
disk_id="$(echo "$importdisk_output" | grep -oP "(?<=Successfully imported disk as ')[^']+" | sed 's/^unused[0-9]*://')"
|
||||
if [[ -z "$disk_id" ]]; then
|
||||
echo "ERROR: couldn't parse the imported disk identifier from qm importdisk's output above." >&2
|
||||
echo "The VM shell (${vmid}) and imported disk both exist -- finish attaching it by hand:" >&2
|
||||
echo " ssh ${ssh_target} -- qm set ${vmid} --scsi0 ${storage}:<disk-id-from-output-above>" >&2
|
||||
echo " ssh ${ssh_target} -- qm set ${vmid} --boot order=scsi0" >&2
|
||||
exit 1
|
||||
fi
|
||||
remote "qm set ${vmid} --scsi0 ${disk_id}"
|
||||
fi
|
||||
remote "qm set ${vmid} --boot order=scsi0"
|
||||
remote "qm start ${vmid}"
|
||||
fi
|
||||
|
||||
echo
|
||||
if [[ "$dry_run" -eq 1 ]]; then
|
||||
echo "[dry-run] Nothing was built, uploaded, or created."
|
||||
else
|
||||
echo "Done. ${name} (VMID ${vmid}) should be booting on ${node}."
|
||||
fi
|
||||
+13
-57
@@ -3,39 +3,21 @@
|
||||
# second copy of these values in every script:
|
||||
# source "$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)/env.sh"
|
||||
# Every variable can still be overridden per-invocation via the
|
||||
# environment (e.g. PROXMOX_STORAGE=tank-nvme ./scripts/proxmox/create-proxmox-resource.sh ...)
|
||||
# environment (e.g. PROXMOX_STORAGE=tank-nvme ./scripts/create-proxmox-resource.sh ...)
|
||||
# since each one only sets a default if unset.
|
||||
|
||||
# Two SSH-reachable Proxmox nodes exist on the LAN:
|
||||
# - pve1.sweet.home -- production. Real, live VMs/containers.
|
||||
# - pve-test.sweet.home -- sandbox/test node, for scratch VMs/containers
|
||||
# that don't belong on production.
|
||||
#
|
||||
# PROXMOX_HOST is what scripts/proxmox/create-proxmox-resource.sh actually
|
||||
# targets by default -- overridable per-invocation with --node <hostname>,
|
||||
# or per-variable as usual (e.g. PROXMOX_HOST=$PVE_TEST_HOST). It defaults
|
||||
# to production, matching this repo's behavior before pve-test existed --
|
||||
# see CLAUDE.md's "Two Proxmox nodes" section for the policy on which
|
||||
# situations should target which node (in particular: Claude defaults to
|
||||
# pve-test, not this variable's own default, unless explicitly told
|
||||
# otherwise).
|
||||
: "${PVE1_HOST:=pve1.sweet.home}"
|
||||
: "${PVE_TEST_HOST:=pve-test.sweet.home}"
|
||||
: "${PROXMOX_HOST:=$PVE1_HOST}"
|
||||
: "${PROXMOX_SSH_USER:=wayne}"
|
||||
|
||||
# Where this flake repo lives on the Proxmox node itself.
|
||||
# scripts/proxmox/create-proxmox-resource.sh builds images directly on the node
|
||||
# instead of transferring them over the network -- it clones the repo here
|
||||
# (from this checkout's own `origin` remote) the first time it doesn't
|
||||
# find it, installing build tooling via scripts/codex-setup.sh, then
|
||||
# `git pull`s it before every subsequent build.
|
||||
: "${PROXMOX_REMOTE_REPO_DIR:=/home/${PROXMOX_SSH_USER}/nixos}"
|
||||
# SSH-reachable Proxmox node that scripts/create-proxmox-resource.sh runs
|
||||
# pct/qm on. Matches the Proxmox web UI hostname already used in
|
||||
# hosts/nixos/home.nix's desktop shortcuts (pve.<homeDomain> from
|
||||
# variables.nix) -- change this if that's not actually reachable over SSH,
|
||||
# or if you're targeting a different node in a multi-node cluster.
|
||||
: "${PROXMOX_HOST:=pve.sweet.home}"
|
||||
: "${PROXMOX_SSH_USER:=root}"
|
||||
|
||||
# Storage pool names -- Proxmox's own stock-install defaults, but this
|
||||
# varies a lot by setup (ZFS pool name, custom LVM-thin volume, etc.).
|
||||
# Verify with `pvesm status` on the node and correct these if wrong.
|
||||
: "${PROXMOX_STORAGE:=local-zfs}" # VM disks / CT rootfs
|
||||
: "${PROXMOX_STORAGE:=local-lvm}" # VM disks / CT rootfs
|
||||
: "${PROXMOX_ISO_STORAGE:=local}" # uploaded images/ISOs/CT templates
|
||||
|
||||
: "${PROXMOX_BRIDGE:=vmbr0}"
|
||||
@@ -63,42 +45,16 @@
|
||||
# crash-loops on a denied `/run/credentials/*` mount every ~3s (visible
|
||||
# as garbage on the console) and core services like nsncd fail the same
|
||||
# way on userns_create; system.build.tarball never finishes activating.
|
||||
#
|
||||
# mount=nfs;nfs4: without it, AppArmor blanket-denies the `nfs`/
|
||||
# `rpc_pipefs` mount syscalls any NFS client share needs -- confirmed
|
||||
# live on lxc-docker (which mounts several, see modules/docker/mount-data.nix
|
||||
# and modules/raspi/mount-data.nix): `mount: /var/lib/nfs/rpc_pipefs:
|
||||
# permission denied`. Harmless to grant on lxc targets that don't mount
|
||||
# NFS at all -- it only widens what the container is *allowed* to mount,
|
||||
# nothing here forces a mount to happen.
|
||||
: "${PROXMOX_DEFAULT_LXC_FEATURES:=nesting=1,keyctl=1,mount=nfs;nfs4}"
|
||||
: "${PROXMOX_DEFAULT_LXC_FEATURES:=nesting=1,keyctl=1}"
|
||||
|
||||
export PVE1_HOST PVE_TEST_HOST PROXMOX_HOST PROXMOX_SSH_USER PROXMOX_STORAGE \
|
||||
PROXMOX_ISO_STORAGE PROXMOX_BRIDGE PROXMOX_DEFAULT_CORES \
|
||||
PROXMOX_DEFAULT_MEMORY_MB PROXMOX_DEFAULT_LXC_DISK_GB \
|
||||
PROXMOX_DEFAULT_LXC_FEATURES PROXMOX_REMOTE_REPO_DIR
|
||||
export PROXMOX_HOST PROXMOX_SSH_USER PROXMOX_STORAGE PROXMOX_ISO_STORAGE \
|
||||
PROXMOX_BRIDGE PROXMOX_DEFAULT_CORES PROXMOX_DEFAULT_MEMORY_MB \
|
||||
PROXMOX_DEFAULT_LXC_DISK_GB PROXMOX_DEFAULT_LXC_FEATURES
|
||||
|
||||
# Matches variables.nix's nixCacheHost -- update both if it ever changes.
|
||||
: "${NIX_CACHE_HOST:=nix-cache}"
|
||||
export NIX_CACHE_HOST
|
||||
|
||||
# Matches variables.nix's lanDomain (the Gitea host this flake's own repo
|
||||
# is served from -- see scripts/installer/auto-install.sh's FLAKE_BASE_URL)
|
||||
# -- update both if it ever changes.
|
||||
: "${LAN_DOMAIN:=gitea.lan.ddnsgeek.com}"
|
||||
export LAN_DOMAIN
|
||||
|
||||
# Matches variables.nix's homeDomain -- the base LAN domain for service
|
||||
# subdomains, FreeIPA Kerberos realm, and host FQDNs.
|
||||
: "${HOME_DOMAIN:=sweet.home}"
|
||||
export HOME_DOMAIN
|
||||
|
||||
# Matches variables.nix's ipaServer -- the FreeIPA server hostname.
|
||||
# scripts/ipa/create-nixos-ipa-host-account.sh SSHes here to run
|
||||
# ipa host-add and ipa-getkeytab.
|
||||
: "${IPA_SERVER:=domain-controller.sweet.home}"
|
||||
export IPA_SERVER
|
||||
|
||||
# nix_extra_opts: call as a plain statement (NOT inside $(...)/<(...) --
|
||||
# that forks a subshell, and the whole point is exporting a decision back
|
||||
# into *this* shell) to populate the global NIX_OPTS array with whatever
|
||||
|
||||
@@ -1,194 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# acceptance-tests.sh — HA cluster acceptance tests (T1–T7)
|
||||
#
|
||||
# Run from a host with SSH access to both HA nodes (or from node1 itself).
|
||||
# All 7 tests must pass before considering the cluster production-ready.
|
||||
# Test values below must match variables.nix haServer* values.
|
||||
set -euo pipefail
|
||||
|
||||
# ── Configuration ─────────────────────────────────────────────────────────
|
||||
# All values override-able via environment variables; defaults match variables.nix.
|
||||
NODE1="${NODE1:-ha-server-1}"
|
||||
NODE2="${NODE2:-ha-server-2}"
|
||||
NODE1_IP="${NODE1_IP:-192.168.2.228}" # vars.haServer1Ip
|
||||
NODE2_IP="${NODE2_IP:-192.168.2.227}" # vars.haServer2Ip
|
||||
VIP="${VIP:-192.168.2.229}" # vars.haServerVip
|
||||
XFS_MOUNT="${XFS_MOUNT:-/srv/ha-data}" # vars.haStorageRoot
|
||||
ISCSI_IQN="${ISCSI_IQN:-iqn.2026-01.home.sweet:ha-storage}" # vars.haIscsiIqn
|
||||
# ──────────────────────────────────────────────────────────────────────────
|
||||
|
||||
PASS=0
|
||||
FAIL=0
|
||||
RESULTS=()
|
||||
|
||||
# Use PASS=$((PASS+1)) instead of ((PASS++)) — the latter evaluates to 0 when
|
||||
# PASS=0, which triggers set -e and kills the script after the very first PASS.
|
||||
pass() { echo " PASS: $1"; PASS=$((PASS+1)); RESULTS+=("PASS $1"); }
|
||||
fail() { echo " FAIL: $1"; FAIL=$((FAIL+1)); RESULTS+=("FAIL $1"); }
|
||||
|
||||
HA_USER="nixos"
|
||||
n1() { ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no -o ConnectTimeout=5 "${HA_USER}@${NODE1_IP}" sudo "$@" 2>/dev/null; }
|
||||
n2() { ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no -o ConnectTimeout=5 "${HA_USER}@${NODE2_IP}" sudo "$@" 2>/dev/null; }
|
||||
|
||||
echo "════════════════════════════════════════════════════"
|
||||
echo " HA Cluster Acceptance Tests — $(date '+%Y-%m-%d %H:%M:%S')"
|
||||
echo "════════════════════════════════════════════════════"
|
||||
|
||||
# ── Detect Active/Standby nodes ────────────────────────────────────────────
|
||||
# Pacemaker can promote either node; determine which is currently Active
|
||||
# (DRBD Primary / holds ha-group resources) before running tests.
|
||||
echo ""
|
||||
echo "Detecting Active/Standby nodes..."
|
||||
if n1 "drbdadm role ha-data" 2>/dev/null | grep -q "^Primary"; then
|
||||
ACTIVE_NODE="$NODE1"; ACTIVE_IP="$NODE1_IP"
|
||||
STANDBY_NODE="$NODE2"; STANDBY_IP="$NODE2_IP"
|
||||
na() { n1 "$@"; }
|
||||
ns() { n2 "$@"; }
|
||||
else
|
||||
ACTIVE_NODE="$NODE2"; ACTIVE_IP="$NODE2_IP"
|
||||
STANDBY_NODE="$NODE1"; STANDBY_IP="$NODE1_IP"
|
||||
na() { n2 "$@"; }
|
||||
ns() { n1 "$@"; }
|
||||
fi
|
||||
echo " Active: $ACTIVE_NODE ($ACTIVE_IP)"
|
||||
echo " Standby: $STANDBY_NODE ($STANDBY_IP)"
|
||||
|
||||
# ── T1: Corosync quorum established ──────────────────────────────────────
|
||||
echo ""
|
||||
echo "[T1] Corosync quorum"
|
||||
if na "corosync-quorumtool -s" 2>/dev/null | grep -q "Quorate:.*Yes"; then
|
||||
pass "cluster has quorum"
|
||||
else
|
||||
fail "cluster does not have quorum — check corosync on both nodes"
|
||||
fi
|
||||
|
||||
# ── T2: DRBD Primary on Active node, Secondary on Standby ────────────────
|
||||
echo ""
|
||||
echo "[T2] DRBD roles"
|
||||
DRBD_ROLE=$(na "drbdadm role ha-data" 2>/dev/null || echo "unknown")
|
||||
if [[ "$DRBD_ROLE" == "Primary/Secondary" || "$DRBD_ROLE" == "Primary" ]]; then
|
||||
pass "DRBD Primary on $ACTIVE_NODE ($DRBD_ROLE)"
|
||||
else
|
||||
fail "unexpected DRBD role on $ACTIVE_NODE: $DRBD_ROLE (expected Primary/Secondary)"
|
||||
fi
|
||||
|
||||
DRBD_DSTATE=$(na "drbdadm dstate ha-data" 2>/dev/null || echo "unknown")
|
||||
if echo "$DRBD_DSTATE" | grep -q "UpToDate"; then
|
||||
pass "DRBD disk state UpToDate ($DRBD_DSTATE)"
|
||||
else
|
||||
fail "DRBD disk not UpToDate: $DRBD_DSTATE"
|
||||
fi
|
||||
|
||||
# ── T3: XFS mounted at haStorageRoot on the Active node ──────────────────
|
||||
echo ""
|
||||
echo "[T3] XFS mount"
|
||||
if na "mountpoint -q '${XFS_MOUNT}'" 2>/dev/null; then
|
||||
pass "XFS mounted at ${XFS_MOUNT} on $ACTIVE_NODE"
|
||||
else
|
||||
fail "XFS not mounted at ${XFS_MOUNT} on $ACTIVE_NODE"
|
||||
fi
|
||||
|
||||
if ns "mountpoint -q '${XFS_MOUNT}'" 2>/dev/null; then
|
||||
fail "XFS unexpectedly mounted on $STANDBY_NODE (should only be on Active node)"
|
||||
else
|
||||
pass "XFS not mounted on $STANDBY_NODE (correct — Standby)"
|
||||
fi
|
||||
|
||||
# ── T4: iSCSI target visible on Active node ───────────────────────────────
|
||||
echo ""
|
||||
echo "[T4] iSCSI target"
|
||||
IQN_COUNT=$(na "ls /sys/kernel/config/target/iscsi/ 2>/dev/null | grep -c iqn" || echo "0")
|
||||
if [[ "$IQN_COUNT" -ge 1 ]]; then
|
||||
pass "iSCSI IQN active on $ACTIVE_NODE ($IQN_COUNT target(s))"
|
||||
else
|
||||
fail "no iSCSI IQN active on $ACTIVE_NODE"
|
||||
fi
|
||||
|
||||
# iSCSI port reachable from Standby node via VIP.
|
||||
# Use bash TCP probe (no iscsiadm needed — just checks port 3260 is open).
|
||||
if ns "bash -c 'echo >/dev/tcp/${VIP}/3260' 2>/dev/null"; then
|
||||
pass "iSCSI port 3260 reachable from $STANDBY_NODE via VIP ${VIP}"
|
||||
else
|
||||
fail "iSCSI port 3260 not reachable from $STANDBY_NODE via ${VIP}"
|
||||
fi
|
||||
|
||||
# ── T5: Failover — standby Active node, verify resources move to Standby ──
|
||||
echo ""
|
||||
echo "[T5] Failover (standby $ACTIVE_NODE)"
|
||||
ACTIVE_CRMD_NAME=$(na "crm_node -n" 2>/dev/null || echo "")
|
||||
na "crm_standby -N '${ACTIVE_CRMD_NAME}' -v on" 2>/dev/null || true
|
||||
echo " Waiting up to 120 s for resources to move to $STANDBY_NODE..."
|
||||
MOVED=false
|
||||
for i in $(seq 1 120); do
|
||||
if ns "mountpoint -q '${XFS_MOUNT}'" 2>/dev/null; then
|
||||
MOVED=true
|
||||
echo " Resources moved in ${i}s"
|
||||
break
|
||||
fi
|
||||
sleep 1
|
||||
done
|
||||
|
||||
if $MOVED; then
|
||||
pass "XFS mounted on $STANDBY_NODE after failover"
|
||||
IQN_ON_STANDBY=$(ns "ls /sys/kernel/config/target/iscsi/ 2>/dev/null | grep -c iqn" || echo "0")
|
||||
[[ "$IQN_ON_STANDBY" -ge 1 ]] \
|
||||
&& pass "iSCSI target active on $STANDBY_NODE after failover" \
|
||||
|| fail "iSCSI target NOT active on $STANDBY_NODE after failover"
|
||||
else
|
||||
fail "XFS did not mount on $STANDBY_NODE within 120 s — failover incomplete"
|
||||
fi
|
||||
|
||||
# ── T6: Data integrity — file written post-failover readable ─────────────
|
||||
echo ""
|
||||
echo "[T6] Data integrity"
|
||||
# Write a test file on the new Active (former Standby) and verify it.
|
||||
# Use `echo | sudo tee` for the write: "echo ... > file" via bash -c has the
|
||||
# redirect interpreted by the remote nixos shell (not sudo), so the file open
|
||||
# runs as nixos and fails with EACCES on the root-owned XFS mount. Piping
|
||||
# through sudo tee lets tee (running as root) open the file instead.
|
||||
TEST_FILE="${XFS_MOUNT}/.acceptance-test-$$"
|
||||
TEST_CONTENT="ha-acceptance-test-$(date +%s)"
|
||||
echo "${TEST_CONTENT}" | ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no -o ConnectTimeout=5 "${HA_USER}@${STANDBY_IP}" sudo tee "${TEST_FILE}" > /dev/null 2>/dev/null || true
|
||||
READBACK=$(ns cat "${TEST_FILE}" 2>/dev/null || echo "")
|
||||
if [[ "$READBACK" == "$TEST_CONTENT" ]]; then
|
||||
pass "test file written and read back correctly on $STANDBY_NODE"
|
||||
else
|
||||
fail "data integrity check failed (wrote: '$TEST_CONTENT', read: '$READBACK')"
|
||||
fi
|
||||
ns rm -f "${TEST_FILE}" 2>/dev/null || true
|
||||
|
||||
# ── T7: Node rejoin — un-standby original Active, verify cluster is healthy ─
|
||||
echo ""
|
||||
echo "[T7] Node rejoin"
|
||||
na "crm_standby -N '${ACTIVE_CRMD_NAME}' -v off" 2>/dev/null || true
|
||||
na "crm_resource --cleanup" 2>/dev/null || true
|
||||
sleep 5
|
||||
|
||||
if na "corosync-quorumtool -s 2>/dev/null | grep -q 'Quorate:.*Yes'"; then
|
||||
pass "$ACTIVE_NODE rejoined — cluster has quorum"
|
||||
else
|
||||
fail "$ACTIVE_NODE did not rejoin with quorum"
|
||||
fi
|
||||
|
||||
DRBD_ROLE_AFTER=$(na "drbdadm role ha-data" 2>/dev/null || echo "unknown")
|
||||
if echo "$DRBD_ROLE_AFTER" | grep -q "Secondary"; then
|
||||
pass "$ACTIVE_NODE is DRBD Secondary after rejoin ($DRBD_ROLE_AFTER)"
|
||||
else
|
||||
fail "unexpected DRBD role on $ACTIVE_NODE after rejoin: $DRBD_ROLE_AFTER"
|
||||
fi
|
||||
|
||||
# ── Summary ───────────────────────────────────────────────────────────────
|
||||
echo ""
|
||||
echo "════════════════════════════════════════════════════"
|
||||
echo " Results: ${PASS} PASS, ${FAIL} FAIL"
|
||||
echo "════════════════════════════════════════════════════"
|
||||
for r in "${RESULTS[@]}"; do echo " $r"; done
|
||||
echo ""
|
||||
|
||||
if [[ "$FAIL" -eq 0 ]]; then
|
||||
echo "ALL PASS — cluster is production-ready."
|
||||
exit 0
|
||||
else
|
||||
echo "SOME TESTS FAILED — investigate before deploying."
|
||||
exit 1
|
||||
fi
|
||||
@@ -1,86 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# cluster-enable-stonith.sh — enable STONITH fence agent after the fence SSH
|
||||
# key is deployed to both nodes and authorised on the Proxmox host.
|
||||
#
|
||||
# Run from ha-server-1 as root AFTER:
|
||||
# - /etc/pacemaker/fence_pve_ssh exists on both nodes (chmod +x)
|
||||
# (copy from scripts/ha/fence-pve-ssh.py)
|
||||
# - /etc/fence-pve-ssh-key (SSH private key) exists on both nodes
|
||||
# - The corresponding public key is in authorized_keys on PVE_HOST
|
||||
# - VMID_NODE1 / VMID_NODE2 filled in below
|
||||
set -euo pipefail
|
||||
|
||||
# ── Configuration ─────────────────────────────────────────────────────────
|
||||
NODE1="ha-server-1"
|
||||
NODE2="ha-server-2"
|
||||
VMID_NODE1="" # FILL IN: Proxmox VMID for ha-server-1
|
||||
VMID_NODE2="" # FILL IN: Proxmox VMID for ha-server-2
|
||||
PVE_HOST="pve1.sweet.home"
|
||||
PVE_USER="wayne"
|
||||
FENCE_KEY="/etc/fence-pve-ssh-key"
|
||||
FENCE_SCRIPT="/etc/pacemaker/fence_pve_ssh"
|
||||
# ──────────────────────────────────────────────────────────────────────────
|
||||
|
||||
log() { echo "[stonith-setup] $*"; }
|
||||
die() { echo "[stonith-setup] ERROR: $*" >&2; exit 1; }
|
||||
|
||||
[[ $(id -u) -eq 0 ]] || die "must run as root"
|
||||
[[ -n "$VMID_NODE1" ]] || die "VMID_NODE1 not set — edit this script"
|
||||
[[ -n "$VMID_NODE2" ]] || die "VMID_NODE2 not set — edit this script"
|
||||
[[ -f "$FENCE_KEY" ]] || die "fence key not found at $FENCE_KEY"
|
||||
[[ -f "$FENCE_SCRIPT" ]] || die "fence script not found at $FENCE_SCRIPT"
|
||||
|
||||
log "Verifying fence agent can reach ${PVE_HOST}..."
|
||||
ssh -i "$FENCE_KEY" -o BatchMode=yes -o ConnectTimeout=10 \
|
||||
-o StrictHostKeyChecking=no "${PVE_USER}@${PVE_HOST}" \
|
||||
"sudo /usr/sbin/qm list" &>/dev/null \
|
||||
|| die "Cannot SSH to ${PVE_USER}@${PVE_HOST} — check authorized_keys and sudo"
|
||||
log "Fence agent SSH connectivity confirmed"
|
||||
|
||||
log "Creating Pacemaker STONITH resources..."
|
||||
cibadmin --create --scope resources --xml-text "
|
||||
<primitive id=\"stonith-${NODE1}\" class=\"stonith\" type=\"external/fence_pve_ssh\">
|
||||
<instance_attributes id=\"stonith-${NODE1}-attrs\">
|
||||
<nvpair id=\"stonith-${NODE1}-plug\" name=\"plug\" value=\"${NODE1}\"/>
|
||||
<nvpair id=\"stonith-${NODE1}-pve-host\" name=\"pve_host\" value=\"${PVE_HOST}\"/>
|
||||
<nvpair id=\"stonith-${NODE1}-pve-user\" name=\"pve_user\" value=\"${PVE_USER}\"/>
|
||||
<nvpair id=\"stonith-${NODE1}-key-file\" name=\"key_file\" value=\"${FENCE_KEY}\"/>
|
||||
<nvpair id=\"stonith-${NODE1}-vmid1\" name=\"vmid_node1\" value=\"${VMID_NODE1}\"/>
|
||||
<nvpair id=\"stonith-${NODE1}-vmid2\" name=\"vmid_node2\" value=\"${VMID_NODE2}\"/>
|
||||
<nvpair id=\"stonith-${NODE1}-host-list\" name=\"pcmk_host_list\" value=\"${NODE1}\"/>
|
||||
</instance_attributes>
|
||||
<operations>
|
||||
<op id=\"stonith-${NODE1}-monitor\" name=\"monitor\" interval=\"30s\" timeout=\"30s\"/>
|
||||
</operations>
|
||||
</primitive>
|
||||
" 2>/dev/null || true
|
||||
|
||||
cibadmin --create --scope resources --xml-text "
|
||||
<primitive id=\"stonith-${NODE2}\" class=\"stonith\" type=\"external/fence_pve_ssh\">
|
||||
<instance_attributes id=\"stonith-${NODE2}-attrs\">
|
||||
<nvpair id=\"stonith-${NODE2}-plug\" name=\"plug\" value=\"${NODE2}\"/>
|
||||
<nvpair id=\"stonith-${NODE2}-pve-host\" name=\"pve_host\" value=\"${PVE_HOST}\"/>
|
||||
<nvpair id=\"stonith-${NODE2}-pve-user\" name=\"pve_user\" value=\"${PVE_USER}\"/>
|
||||
<nvpair id=\"stonith-${NODE2}-key-file\" name=\"key_file\" value=\"${FENCE_KEY}\"/>
|
||||
<nvpair id=\"stonith-${NODE2}-vmid1\" name=\"vmid_node1\" value=\"${VMID_NODE1}\"/>
|
||||
<nvpair id=\"stonith-${NODE2}-vmid2\" name=\"vmid_node2\" value=\"${VMID_NODE2}\"/>
|
||||
<nvpair id=\"stonith-${NODE2}-host-list\" name=\"pcmk_host_list\" value=\"${NODE2}\"/>
|
||||
</instance_attributes>
|
||||
<operations>
|
||||
<op id=\"stonith-${NODE2}-monitor\" name=\"monitor\" interval=\"30s\" timeout=\"30s\"/>
|
||||
</operations>
|
||||
</primitive>
|
||||
" 2>/dev/null || true
|
||||
|
||||
log "Enabling STONITH and restoring quorum policy..."
|
||||
crm_attribute -t crm_config -n stonith-enabled -v true
|
||||
crm_attribute -t crm_config -n no-quorum-policy -v stop
|
||||
|
||||
log "DRBD fencing mode must also be updated to resource-only (already the"
|
||||
log "default in cluster-config.nix; confirm with: cat /etc/drbd.d/ha-data.conf)"
|
||||
|
||||
log "Testing fence agent..."
|
||||
stonith_admin --list-devices && log "Fence devices listed successfully." \
|
||||
|| warn "stonith_admin --list-devices failed — check config"
|
||||
|
||||
log "STONITH enabled. Cluster is now fully HA."
|
||||
@@ -1,334 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# cluster-init.sh — one-time HA cluster initialisation script
|
||||
#
|
||||
# Run ONCE from ha-server-1 as root AFTER both VMs are booted and have SSH
|
||||
# access. It:
|
||||
# 1. Generates and distributes the corosync authkey
|
||||
# 2. Waits for corosync quorum and pacemaker
|
||||
# 3. Initialises DRBD metadata, promotes node1 to primary
|
||||
# 4. Creates XFS on /dev/drbd0 and mounts it
|
||||
# 5. Creates the directory tree and iSCSI LUN backing file
|
||||
# 6. Configures LIO iSCSI target (file-backed LUN)
|
||||
# 7. Configures Pacemaker resources: DRBD → XFS → iSCSI → NFS → VIP
|
||||
#
|
||||
# Prerequisites:
|
||||
# - Both VMs booted with the ha-server config (nixos-rebuild done)
|
||||
# - SSH key access from node1 to root@NODE2_IP
|
||||
# - VMID_NODE1 / VMID_NODE2 filled in below (needed for STONITH setup;
|
||||
# cluster starts without STONITH, which you enable separately via
|
||||
# scripts/ha/cluster-enable-stonith.sh)
|
||||
# - Run as root on ha-server-1
|
||||
set -euo pipefail
|
||||
|
||||
# ── Configuration ─────────────────────────────────────────────────────────
|
||||
# All values override-able via environment variables; defaults match variables.nix.
|
||||
NODE1="${NODE1:-ha-server-1}"
|
||||
NODE2="${NODE2:-ha-server-2}"
|
||||
NODE1_IP="${NODE1_IP:-192.168.2.228}" # vars.haServer1Ip
|
||||
NODE2_IP="${NODE2_IP:-192.168.2.227}" # vars.haServer2Ip
|
||||
VIP="${VIP:-192.168.2.229}" # vars.haServerVip
|
||||
XFS_MOUNT="${XFS_MOUNT:-/srv/ha-data}" # vars.haStorageRoot
|
||||
ISCSI_IQN="${ISCSI_IQN:-iqn.2026-01.home.sweet:ha-storage}" # vars.haIscsiIqn
|
||||
ISCSI_LUN_FILE="${XFS_MOUNT}/iscsi-lun.img"
|
||||
ISCSI_LUN_SIZE="10G"
|
||||
DRBD_DEVICE="/dev/drbd0"
|
||||
VMID_NODE1="${VMID_NODE1:-}" # set by deploy.sh; needed for STONITH
|
||||
VMID_NODE2="${VMID_NODE2:-}"
|
||||
PVE_HOST="${PVE_HOST:-pve1.sweet.home}"
|
||||
PVE_USER="${PVE_USER:-wayne}"
|
||||
# Inter-node SSH: HA_USER is the user to SSH as on NODE2; HA_KEY is the private
|
||||
# key to use. Default is root-to-root (no key arg). deploy.sh sets HA_USER=nixos
|
||||
# and HA_KEY=/tmp/cluster-init-key so the script works even when root-to-root SSH
|
||||
# is not available.
|
||||
HA_USER="${HA_USER:-root}"
|
||||
HA_KEY="${HA_KEY:-}"
|
||||
|
||||
# NFS dataset subdirectories to create under XFS_MOUNT.
|
||||
# Must mirror vars.nfsShares subpath values in variables.nix.
|
||||
NFS_SUBDIRS=(
|
||||
"docker/config"
|
||||
"docker/volumes"
|
||||
"docker/databases"
|
||||
"docker/nextcloud-data"
|
||||
"raspi/volumes"
|
||||
"proxmox/iso"
|
||||
"proxmox/lxc"
|
||||
"pxe-boot/images"
|
||||
)
|
||||
# ──────────────────────────────────────────────────────────────────────────
|
||||
|
||||
log() { echo "[cluster-init] $*"; }
|
||||
die() { echo "[cluster-init] ERROR: $*" >&2; exit 1; }
|
||||
warn() { echo "[cluster-init] WARNING: $*" >&2; }
|
||||
|
||||
[[ $(id -u) -eq 0 ]] || die "must run as root"
|
||||
[[ "$(hostname)" == "$NODE1" ]] || die "must run on $NODE1"
|
||||
|
||||
# NixOS may not include xfsprogs in root's PATH even when it's in the store.
|
||||
# If mkfs.xfs is missing, search the Nix store for it.
|
||||
if ! command -v mkfs.xfs &>/dev/null; then
|
||||
_xfs_bin=$(find /nix/store -maxdepth 3 -name mkfs.xfs 2>/dev/null | head -1 | xargs dirname 2>/dev/null || true)
|
||||
[[ -n "$_xfs_bin" ]] && export PATH="$_xfs_bin:$PATH" \
|
||||
|| die "mkfs.xfs not found — add xfsprogs to ha-server.nix environment.systemPackages and rebuild"
|
||||
fi
|
||||
|
||||
# Inter-node SSH/SCP helpers — abstract over root-to-root vs nixos+sudo.
|
||||
_SSH_OPTS="-o StrictHostKeyChecking=no -o ConnectTimeout=10"
|
||||
[[ -n "$HA_KEY" ]] && _SSH_OPTS="-i $HA_KEY $_SSH_OPTS"
|
||||
if [[ "$HA_USER" == "root" ]]; then
|
||||
n2_ssh() { ssh $_SSH_OPTS "root@${NODE2_IP}" "$@"; }
|
||||
n2_scp() { scp $_SSH_OPTS "$1" "root@${NODE2_IP}:$2"; }
|
||||
else
|
||||
# Non-root user with passwordless sudo; wrap each command with sudo.
|
||||
n2_ssh() { ssh $_SSH_OPTS "${HA_USER}@${NODE2_IP}" sudo "$@"; }
|
||||
n2_scp() {
|
||||
# SCP to a tmp path, then sudo-move to the real destination as the remote user.
|
||||
local src="$1" dst="$2"
|
||||
local tmp="/tmp/_cluster_init_scp_$$"
|
||||
scp $_SSH_OPTS "$src" "${HA_USER}@${NODE2_IP}:${tmp}"
|
||||
ssh $_SSH_OPTS "${HA_USER}@${NODE2_IP}" sudo mv "${tmp}" "${dst}"
|
||||
}
|
||||
fi
|
||||
|
||||
# ── 0. Corosync authkey ───────────────────────────────────────────────────
|
||||
AUTHKEY="/etc/corosync/authkey"
|
||||
mkdir -p /etc/corosync
|
||||
if [[ ! -f "$AUTHKEY" ]]; then
|
||||
log "Generating corosync authkey..."
|
||||
corosync-keygen -k "$AUTHKEY"
|
||||
chmod 0400 "$AUTHKEY"
|
||||
fi
|
||||
log "Distributing authkey to $NODE2..."
|
||||
n2_ssh "mkdir -p /etc/corosync"
|
||||
n2_scp "$AUTHKEY" "$AUTHKEY"
|
||||
n2_ssh "chmod 0400 '${AUTHKEY}'"
|
||||
|
||||
log "Restarting corosync on both nodes..."
|
||||
systemctl restart corosync
|
||||
n2_ssh "systemctl restart corosync"
|
||||
sleep 3
|
||||
|
||||
# ── 1. Corosync quorum ────────────────────────────────────────────────────
|
||||
log "Waiting for corosync quorum..."
|
||||
for i in $(seq 1 30); do
|
||||
if corosync-quorumtool -s 2>/dev/null | grep -q 'Quorate:.*Yes'; then
|
||||
log "Quorum established"
|
||||
break
|
||||
fi
|
||||
[[ $i -eq 30 ]] && die "corosync quorum not established after 60 s"
|
||||
sleep 2
|
||||
done
|
||||
|
||||
log "Waiting for pacemaker..."
|
||||
for i in $(seq 1 30); do
|
||||
if crm_mon -1 &>/dev/null; then
|
||||
log "Pacemaker running"
|
||||
break
|
||||
fi
|
||||
[[ $i -eq 30 ]] && die "pacemaker not running after 60 s"
|
||||
sleep 2
|
||||
done
|
||||
|
||||
# ── 2. DRBD initialisation ────────────────────────────────────────────────
|
||||
log "Initialising DRBD metadata on $NODE1..."
|
||||
if ! drbdadm dstate ha-data 2>/dev/null | grep -q "UpToDate\|Inconsistent\|Diskless"; then
|
||||
drbdadm create-md ha-data --force
|
||||
fi
|
||||
|
||||
log "Initialising DRBD metadata on $NODE2..."
|
||||
# Use grep -E for ERE alternation inside the remote bash -c string (avoids \| quoting issues).
|
||||
n2_ssh "bash -c 'drbdadm dstate ha-data 2>/dev/null | grep -qE \"UpToDate|Inconsistent|Diskless\" || drbdadm create-md ha-data --force'"
|
||||
|
||||
log "Bringing up DRBD on both nodes..."
|
||||
drbdadm up ha-data 2>/dev/null || true
|
||||
n2_ssh "drbdadm up ha-data" 2>/dev/null || true
|
||||
|
||||
log "Forcing $NODE1 to DRBD Primary for initial sync..."
|
||||
drbdadm primary ha-data --force
|
||||
|
||||
log "Waiting for DRBD to finish initial sync (this may take several minutes)..."
|
||||
for i in $(seq 1 300); do
|
||||
state=$(drbdadm dstate ha-data 2>/dev/null || echo "unknown")
|
||||
if echo "$state" | grep -q "UpToDate/UpToDate"; then
|
||||
log "DRBD sync complete: $state"
|
||||
break
|
||||
fi
|
||||
[[ $i -eq 300 ]] && warn "DRBD not UpToDate after 300 s — continuing anyway (check drbdadm status)"
|
||||
sleep 1
|
||||
done
|
||||
|
||||
# ── 3. XFS filesystem ─────────────────────────────────────────────────────
|
||||
log "Creating XFS on ${DRBD_DEVICE}..."
|
||||
if ! xfs_info "${DRBD_DEVICE}" &>/dev/null; then
|
||||
mkfs.xfs -f "${DRBD_DEVICE}"
|
||||
fi
|
||||
|
||||
log "Mounting ${DRBD_DEVICE} at ${XFS_MOUNT}..."
|
||||
mkdir -p "${XFS_MOUNT}"
|
||||
mountpoint -q "${XFS_MOUNT}" || mount "${DRBD_DEVICE}" "${XFS_MOUNT}"
|
||||
|
||||
# ── 4. NFS dataset directories ────────────────────────────────────────────
|
||||
log "Creating NFS dataset directories..."
|
||||
for subdir in "${NFS_SUBDIRS[@]}"; do
|
||||
mkdir -p "${XFS_MOUNT}/${subdir}"
|
||||
done
|
||||
|
||||
# ── 5. iSCSI LUN backing file ─────────────────────────────────────────────
|
||||
log "Creating iSCSI LUN backing file ${ISCSI_LUN_FILE} (${ISCSI_LUN_SIZE})..."
|
||||
if [[ ! -f "${ISCSI_LUN_FILE}" ]]; then
|
||||
fallocate -l "${ISCSI_LUN_SIZE}" "${ISCSI_LUN_FILE}"
|
||||
fi
|
||||
|
||||
# ── 6. LIO iSCSI target ───────────────────────────────────────────────────
|
||||
log "Configuring LIO iSCSI target via targetcli..."
|
||||
# Note: do NOT bind portal to ${VIP} here — the VIP isn't assigned yet (Pacemaker
|
||||
# creates it). The default portal (all IPs, port 3260) is correct; Pacemaker's
|
||||
# VIP resource will make the target reachable at the VIP address.
|
||||
#
|
||||
# Clear any existing LIO state first (idempotent: re-run after a partial failure).
|
||||
# Use specific delete commands — clearconfig does not reliably clear kernel state.
|
||||
if ls /sys/kernel/config/target/iscsi/ 2>/dev/null | grep -q "${ISCSI_IQN}"; then
|
||||
log "Clearing existing LIO target ${ISCSI_IQN} before reconfiguration..."
|
||||
targetcli "/iscsi delete ${ISCSI_IQN}" 2>/dev/null || true
|
||||
fi
|
||||
if ls /sys/kernel/config/target/core/ 2>/dev/null | grep -q "fileio"; then
|
||||
log "Clearing existing LIO backstore ha-lun0 before reconfiguration..."
|
||||
targetcli "/backstores/fileio delete ha-lun0" 2>/dev/null || true
|
||||
fi
|
||||
targetcli <<EOF
|
||||
/backstores/fileio create name=ha-lun0 file_or_dev=${ISCSI_LUN_FILE} size=0 write_back=false
|
||||
/iscsi create ${ISCSI_IQN}
|
||||
/iscsi/${ISCSI_IQN}/tpg1/luns create /backstores/fileio/ha-lun0
|
||||
/iscsi/${ISCSI_IQN}/tpg1 set attribute authentication=0
|
||||
/iscsi/${ISCSI_IQN}/tpg1 set attribute demo_mode_write_protect=0
|
||||
saveconfig /etc/target/saveconfig.json
|
||||
EOF
|
||||
|
||||
log "Tearing down LIO kernel objects — Pacemaker will restore via targetctl on the Active node..."
|
||||
# LIO holds the backing file open; clear kernel state now so the XFS unmount succeeds.
|
||||
# Use specific delete commands (clearconfig does not reliably clear kernel configfs state).
|
||||
targetcli "/iscsi delete ${ISCSI_IQN}" 2>/dev/null || warn "LIO iscsi delete failed — umount may fail"
|
||||
targetcli "/backstores/fileio delete ha-lun0" 2>/dev/null || warn "LIO backstore delete failed"
|
||||
|
||||
log "Distributing iSCSI saveconfig to $NODE2..."
|
||||
n2_scp /etc/target/saveconfig.json /etc/target/saveconfig.json
|
||||
|
||||
|
||||
log "Unmounting ${XFS_MOUNT} — Pacemaker manages it..."
|
||||
umount "${XFS_MOUNT}" || { sync; umount -l "${XFS_MOUNT}"; }
|
||||
|
||||
log "Demoting DRBD to Secondary — Pacemaker manages primary role..."
|
||||
[[ "$(drbdadm role ha-data 2>/dev/null)" == "Primary/Secondary" ]] && drbdadm secondary ha-data || true
|
||||
|
||||
# ── 7. Pacemaker resources ────────────────────────────────────────────────
|
||||
log "Configuring Pacemaker cluster properties..."
|
||||
crm_attribute -t crm_config -n stonith-enabled -v false
|
||||
crm_attribute -t crm_config -n no-quorum-policy -v ignore
|
||||
|
||||
log "Creating Pacemaker resources via cibadmin..."
|
||||
# Use cibadmin --replace with pacemaker-4.0-compatible XML.
|
||||
# Key schema rules for pacemaker-4.0:
|
||||
# - globally-unique must be in <meta_attributes>, not a direct <clone> attribute
|
||||
# - promoted-max / promoted-node-max (not master-max / master-node-max)
|
||||
# - constraint with-rsc-role="Promoted" (not "Master")
|
||||
cibadmin --replace --scope resources --xml-text '<resources>
|
||||
<clone id="ms-drbd0">
|
||||
<meta_attributes id="ms-drbd0-meta">
|
||||
<nvpair id="ms-drbd0-globally-unique" name="globally-unique" value="false"/>
|
||||
<nvpair id="ms-drbd0-promotable" name="promotable" value="true"/>
|
||||
<nvpair id="ms-drbd0-promoted-max" name="promoted-max" value="1"/>
|
||||
<nvpair id="ms-drbd0-promoted-node-max" name="promoted-node-max" value="1"/>
|
||||
<nvpair id="ms-drbd0-clone-max" name="clone-max" value="2"/>
|
||||
<nvpair id="ms-drbd0-clone-node-max" name="clone-node-max" value="1"/>
|
||||
<nvpair id="ms-drbd0-notify" name="notify" value="true"/>
|
||||
<nvpair id="ms-drbd0-interleave" name="interleave" value="true"/>
|
||||
</meta_attributes>
|
||||
<primitive id="drbd0" class="ocf" type="drbd" provider="linbit">
|
||||
<instance_attributes id="drbd0-attrs">
|
||||
<nvpair id="drbd0-resource" name="drbd_resource" value="ha-data"/>
|
||||
</instance_attributes>
|
||||
<operations>
|
||||
<op id="drbd0-start" name="start" interval="0" timeout="240s"/>
|
||||
<op id="drbd0-stop" name="stop" interval="0" timeout="120s"/>
|
||||
<op id="drbd0-promote" name="promote" interval="0" timeout="240s"/>
|
||||
<op id="drbd0-demote" name="demote" interval="0" timeout="90s"/>
|
||||
<op id="drbd0-monitor-promoted" name="monitor" interval="20s" timeout="20s" role="Promoted"/>
|
||||
<op id="drbd0-monitor-unpromoted" name="monitor" interval="30s" timeout="20s" role="Unpromoted"/>
|
||||
</operations>
|
||||
</primitive>
|
||||
</clone>
|
||||
<group id="ha-group">
|
||||
<primitive id="xfs-data" class="ocf" type="Filesystem" provider="heartbeat">
|
||||
<instance_attributes id="xfs-data-attrs">
|
||||
<nvpair id="xfs-data-device" name="device" value="/dev/drbd0"/>
|
||||
<nvpair id="xfs-data-directory" name="directory" value="/srv/ha-data"/>
|
||||
<nvpair id="xfs-data-fstype" name="fstype" value="xfs"/>
|
||||
<nvpair id="xfs-data-options" name="options" value="defaults"/>
|
||||
<nvpair id="xfs-data-force_unmount" name="force_unmount" value="true"/>
|
||||
</instance_attributes>
|
||||
<operations>
|
||||
<op id="xfs-data-start" name="start" interval="0" timeout="60s"/>
|
||||
<op id="xfs-data-stop" name="stop" interval="0" timeout="60s"/>
|
||||
<op id="xfs-data-monitor" name="monitor" interval="20s" timeout="40s"/>
|
||||
</operations>
|
||||
</primitive>
|
||||
<primitive id="iscsi-target" class="systemd" type="targetctl">
|
||||
<operations>
|
||||
<op id="iscsi-start" name="start" interval="0" timeout="60s"/>
|
||||
<op id="iscsi-stop" name="stop" interval="0" timeout="60s"/>
|
||||
<op id="iscsi-monitor" name="monitor" interval="20s" timeout="40s"/>
|
||||
</operations>
|
||||
</primitive>
|
||||
<primitive id="nfs-server" class="systemd" type="nfs-server">
|
||||
<operations>
|
||||
<op id="nfs-start" name="start" interval="0" timeout="60s"/>
|
||||
<op id="nfs-stop" name="stop" interval="0" timeout="60s"/>
|
||||
<op id="nfs-monitor" name="monitor" interval="30s" timeout="40s"/>
|
||||
</operations>
|
||||
</primitive>
|
||||
<primitive id="vip" class="ocf" type="IPaddr2" provider="heartbeat">
|
||||
<instance_attributes id="vip-attrs">
|
||||
<nvpair id="vip-ip" name="ip" value="192.168.2.229"/>
|
||||
<nvpair id="vip-cidr" name="cidr_netmask" value="24"/>
|
||||
</instance_attributes>
|
||||
<operations>
|
||||
<op id="vip-start" name="start" interval="0" timeout="20s"/>
|
||||
<op id="vip-stop" name="stop" interval="0" timeout="20s"/>
|
||||
<op id="vip-monitor" name="monitor" interval="10s" timeout="20s"/>
|
||||
</operations>
|
||||
</primitive>
|
||||
</group>
|
||||
</resources>'
|
||||
|
||||
log "Adding Pacemaker ordering and colocation constraints..."
|
||||
cibadmin --replace --scope constraints --xml-text '<constraints>
|
||||
<rsc_order id="order-drbd-group" first="ms-drbd0" first-action="promote" then="ha-group" then-action="start" kind="Mandatory"/>
|
||||
<rsc_colocation id="coloc-group-with-drbd" score="INFINITY" rsc="ha-group" with-rsc="ms-drbd0" with-rsc-role="Promoted"/>
|
||||
</constraints>'
|
||||
|
||||
log "Waiting for resources to start..."
|
||||
for i in $(seq 1 60); do
|
||||
if crm_resource -r vip --locate 2>/dev/null | grep -q "running on"; then
|
||||
log "VIP is up: $(crm_resource -r vip --locate)"
|
||||
break
|
||||
fi
|
||||
[[ $i -eq 60 ]] && { warn "VIP not up after 120 s — check: crm_mon -1"; break; }
|
||||
sleep 2
|
||||
done
|
||||
|
||||
log ""
|
||||
log "═══════════════════════════════════════════════════════════════"
|
||||
log " HA cluster initialised."
|
||||
log ""
|
||||
log " crm_mon -1 — cluster status"
|
||||
log " iscsiadm -m discovery -t st -p ${VIP} — verify iSCSI target"
|
||||
log " showmount -e ${VIP} — verify NFS exports"
|
||||
log ""
|
||||
log " To enable STONITH (after deploying fence SSH key):"
|
||||
log " 1. Fill in VMID_NODE1 / VMID_NODE2 in cluster-enable-stonith.sh"
|
||||
log " 2. Copy scripts/ha/fence-pve-ssh.py to /etc/pacemaker/fence_pve_ssh"
|
||||
log " on both nodes (chmod +x)"
|
||||
log " 3. Generate and distribute the fence SSH key"
|
||||
log " (see docs or cluster-enable-stonith.sh header)"
|
||||
log " 4. bash scripts/ha/cluster-enable-stonith.sh"
|
||||
log "═══════════════════════════════════════════════════════════════"
|
||||
@@ -1,462 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# deploy.sh — Full lifecycle management for the HA file-server cluster.
|
||||
#
|
||||
# Handles everything from zero (no VMs, no secrets) through a running,
|
||||
# tested cluster, and optionally tears it back down.
|
||||
#
|
||||
# Usage:
|
||||
# scripts/ha/deploy.sh [options]
|
||||
# scripts/ha/deploy.sh --destroy [options]
|
||||
#
|
||||
# Phases (all run by default; skip any with --skip-*):
|
||||
# 1. ensure-bridge Create storage bridge (vmbr1) on the Proxmox node if absent.
|
||||
# 2. sync-keys Generate SSH host keys and register age keys for both nodes.
|
||||
# 3. create-vms Build disk images and create both VMs via create-proxmox-resource.sh.
|
||||
# 4. add-hardware Attach storage NIC (vmbr1) and DRBD data disk to each VM.
|
||||
# 5. boot-wait Start VMs, wait for SSH on both nodes.
|
||||
# 6. cluster-init Form the cluster: DRBD, corosync, Pacemaker, NFS, VIP.
|
||||
# Also encrypts the generated corosync authkey into the repo.
|
||||
# 7. run-tests Run acceptance tests (T1–T7).
|
||||
#
|
||||
# Options:
|
||||
# --node <host> Proxmox host to deploy on (default: pve1.sweet.home)
|
||||
# --vmid1 <n> VMID for ha-server-1 (default: 200)
|
||||
# --vmid2 <n> VMID for ha-server-2 (default: 201)
|
||||
# --storage <pool> Proxmox storage pool (default: local-zfs)
|
||||
# --storage-bridge <br> Bridge for HA storage network (default: vmbr1)
|
||||
# --drbd-disk-gb <n> DRBD data disk size in GB (default: 32)
|
||||
# --memory <MB> RAM per node (default: 4096)
|
||||
# --cores <n> vCPUs per node (default: 4)
|
||||
# --skip-ensure-bridge Skip storage bridge creation/check
|
||||
# --skip-sync-keys Skip sync-host-keys.sh (clan vars already exist)
|
||||
# --skip-create-vms Skip VM creation (VMs already exist)
|
||||
# --skip-add-hardware Skip net1/scsi1 attachment (already attached)
|
||||
# --skip-boot-wait Skip boot/SSH wait (VMs already running)
|
||||
# --skip-cluster-init Skip cluster formation (cluster already configured)
|
||||
# --skip-tests Skip acceptance tests
|
||||
# --force-rebuild Pass --force-rebuild to create-proxmox-resource.sh
|
||||
# --destroy Stop and delete both VMs (skip all other phases)
|
||||
# --dry-run Print what would run without executing
|
||||
# -h|--help Show this message
|
||||
#
|
||||
# Prerequisites:
|
||||
# - SSH access to the Proxmox node as $PROXMOX_SSH_USER (wayne).
|
||||
# - sops age key in the standard location (used by sync-host-keys.sh).
|
||||
# - For --skip-sync-keys: clan vars already in vars/per-machine/proxmox-ha-server-{1,2}/.
|
||||
# - For full tests: secrets/common.yaml decryptable on both nodes (run
|
||||
# `sops updatekeys secrets/common.yaml` after sync-keys).
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
REPO_ROOT="$(cd "$SCRIPT_DIR/../.." && pwd)"
|
||||
|
||||
# shellcheck source=../env.sh
|
||||
source "${REPO_ROOT}/scripts/env.sh"
|
||||
|
||||
# ── Defaults ──────────────────────────────────────────────────────────────────
|
||||
|
||||
NODE="${PROXMOX_HOST:-$PVE1_HOST}"
|
||||
VMID1=200
|
||||
VMID2=201
|
||||
STORAGE="${PROXMOX_STORAGE:-local-zfs}"
|
||||
STORAGE_BRIDGE="vmbr1"
|
||||
DRBD_DISK_GB=32
|
||||
MEMORY_MB=4096
|
||||
CORES=4
|
||||
|
||||
SKIP_ENSURE_BRIDGE=false
|
||||
SKIP_SYNC_KEYS=false
|
||||
SKIP_CREATE_VMS=false
|
||||
SKIP_ADD_HARDWARE=false
|
||||
SKIP_BOOT_WAIT=false
|
||||
SKIP_CLUSTER_INIT=false
|
||||
SKIP_TESTS=false
|
||||
FORCE_REBUILD=false
|
||||
DESTROY=false
|
||||
DRY_RUN=false
|
||||
|
||||
# ── Variables from repo ───────────────────────────────────────────────────────
|
||||
|
||||
NODE1_HOST="ha-server-1"
|
||||
NODE2_HOST="ha-server-2"
|
||||
NODE1_IP="192.168.2.228"
|
||||
NODE2_IP="192.168.2.227"
|
||||
STORAGE_IP1="192.168.4.228"
|
||||
STORAGE_IP2="192.168.4.227"
|
||||
STORAGE_CIDR="192.168.4.0/29"
|
||||
SSH_USER="${PROXMOX_SSH_USER:-wayne}"
|
||||
|
||||
# ── Argument parsing ──────────────────────────────────────────────────────────
|
||||
|
||||
usage() {
|
||||
sed -n '/^# Usage:/,/^[^#]/{ /^#/{ s/^# \?//; p } }' "$0"
|
||||
exit "${1:-0}"
|
||||
}
|
||||
|
||||
while [[ $# -gt 0 ]]; do
|
||||
case "$1" in
|
||||
--node) NODE="$2"; shift 2 ;;
|
||||
--vmid1) VMID1="$2"; shift 2 ;;
|
||||
--vmid2) VMID2="$2"; shift 2 ;;
|
||||
--storage) STORAGE="$2"; shift 2 ;;
|
||||
--storage-bridge) STORAGE_BRIDGE="$2"; shift 2 ;;
|
||||
--drbd-disk-gb) DRBD_DISK_GB="$2"; shift 2 ;;
|
||||
--memory) MEMORY_MB="$2"; shift 2 ;;
|
||||
--cores) CORES="$2"; shift 2 ;;
|
||||
--skip-ensure-bridge) SKIP_ENSURE_BRIDGE=true; shift ;;
|
||||
--skip-sync-keys) SKIP_SYNC_KEYS=true; shift ;;
|
||||
--skip-create-vms) SKIP_CREATE_VMS=true; shift ;;
|
||||
--skip-add-hardware) SKIP_ADD_HARDWARE=true; shift ;;
|
||||
--skip-boot-wait) SKIP_BOOT_WAIT=true; shift ;;
|
||||
--skip-cluster-init) SKIP_CLUSTER_INIT=true; shift ;;
|
||||
--skip-tests) SKIP_TESTS=true; shift ;;
|
||||
--force-rebuild) FORCE_REBUILD=true; shift ;;
|
||||
--destroy) DESTROY=true; shift ;;
|
||||
--dry-run) DRY_RUN=true; shift ;;
|
||||
-h|--help) usage 0 ;;
|
||||
*) echo "Unknown option: $1" >&2; usage 1 ;;
|
||||
esac
|
||||
done
|
||||
|
||||
# ── Helpers ───────────────────────────────────────────────────────────────────
|
||||
|
||||
log() { echo "==> $*"; }
|
||||
logn() { echo " $*"; }
|
||||
err() { echo "ERROR: $*" >&2; exit 1; }
|
||||
|
||||
run() {
|
||||
if $DRY_RUN; then
|
||||
echo "[dry-run] $*"
|
||||
else
|
||||
"$@"
|
||||
fi
|
||||
}
|
||||
|
||||
pve() {
|
||||
# Run a command on the Proxmox node via SSH.
|
||||
if $DRY_RUN; then
|
||||
echo "[dry-run] ssh ${SSH_USER}@${NODE} sudo $*"
|
||||
else
|
||||
ssh -i ~/.ssh/id_ed25519 "${SSH_USER}@${NODE}" "sudo $*"
|
||||
fi
|
||||
}
|
||||
|
||||
pve_check() {
|
||||
# Run a read-only probe on the Proxmox node — always executes even in dry-run.
|
||||
ssh -i ~/.ssh/id_ed25519 "${SSH_USER}@${NODE}" "sudo $*"
|
||||
}
|
||||
|
||||
HA_USER="nixos"
|
||||
|
||||
n1() {
|
||||
# Run a command on ha-server-1 via SSH as nixos with sudo.
|
||||
ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no -o ConnectTimeout=5 "${HA_USER}@${NODE1_IP}" sudo "$@" 2>/dev/null
|
||||
}
|
||||
|
||||
n2() {
|
||||
# Run a command on ha-server-2 via SSH as nixos with sudo.
|
||||
ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no -o ConnectTimeout=5 "${HA_USER}@${NODE2_IP}" sudo "$@" 2>/dev/null
|
||||
}
|
||||
|
||||
wait_for_ssh() {
|
||||
local ip="$1" label="$2"
|
||||
if $DRY_RUN; then
|
||||
logn "[dry-run] Skipping SSH wait for ${label} (${ip})"
|
||||
return 0
|
||||
fi
|
||||
local deadline=$(( $(date +%s) + 300 ))
|
||||
log "Waiting for SSH on ${label} (${ip}) — up to 5 min..."
|
||||
while [[ $(date +%s) -lt $deadline ]]; do
|
||||
if ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no -o ConnectTimeout=3 \
|
||||
-o BatchMode=yes "${HA_USER}@${ip}" true 2>/dev/null; then
|
||||
logn "${label} is up."
|
||||
return 0
|
||||
fi
|
||||
sleep 5
|
||||
done
|
||||
err "Timed out waiting for SSH on ${label} (${ip})"
|
||||
}
|
||||
|
||||
# ── Destroy mode ─────────────────────────────────────────────────────────────
|
||||
|
||||
if $DESTROY; then
|
||||
log "Destroying HA cluster VMs (${VMID1}=${NODE1_HOST}, ${VMID2}=${NODE2_HOST}) on ${NODE}"
|
||||
for vmid in "$VMID1" "$VMID2"; do
|
||||
STATUS=$(pve "qm status ${vmid} 2>/dev/null" 2>/dev/null || true)
|
||||
if echo "$STATUS" | grep -q "running"; then
|
||||
log "Stopping VMID ${vmid}..."
|
||||
pve "qm stop ${vmid} --skiplock 1"
|
||||
sleep 5
|
||||
fi
|
||||
if $DRY_RUN || pve "qm config ${vmid} >/dev/null 2>&1"; then
|
||||
log "Deleting VMID ${vmid}..."
|
||||
run pve "qm destroy ${vmid} --purge 1"
|
||||
else
|
||||
logn "VMID ${vmid} not found — already gone."
|
||||
fi
|
||||
done
|
||||
log "Done — cluster VMs destroyed."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# ── Phase 1: Storage bridge ───────────────────────────────────────────────────
|
||||
|
||||
if ! $SKIP_ENSURE_BRIDGE; then
|
||||
log "Phase 1: Ensuring storage bridge ${STORAGE_BRIDGE} on ${NODE}"
|
||||
if pve_check "test -d /sys/class/net/${STORAGE_BRIDGE}" &>/dev/null; then
|
||||
logn "${STORAGE_BRIDGE} already exists — skipping."
|
||||
else
|
||||
logn "Creating isolated internal bridge ${STORAGE_BRIDGE} (no upstream port, ${STORAGE_CIDR})"
|
||||
BRIDGE_CONF="auto ${STORAGE_BRIDGE}
|
||||
iface ${STORAGE_BRIDGE} inet manual
|
||||
bridge-ports none
|
||||
bridge-stp off
|
||||
bridge-fd 0"
|
||||
if $DRY_RUN; then
|
||||
echo "[dry-run] Would write /etc/network/interfaces.d/${STORAGE_BRIDGE}.conf and ifup it"
|
||||
else
|
||||
ssh -i ~/.ssh/id_ed25519 "${SSH_USER}@${NODE}" \
|
||||
"echo '${BRIDGE_CONF}' | sudo tee /etc/network/interfaces.d/${STORAGE_BRIDGE}.conf > /dev/null && sudo ifup ${STORAGE_BRIDGE}"
|
||||
logn "${STORAGE_BRIDGE} created and brought up."
|
||||
fi
|
||||
fi
|
||||
fi
|
||||
|
||||
# ── Phase 2: Sync host keys ───────────────────────────────────────────────────
|
||||
|
||||
if ! $SKIP_SYNC_KEYS; then
|
||||
log "Phase 2: Syncing SSH host keys for both HA targets"
|
||||
for target in proxmox-ha-server-1 proxmox-ha-server-2; do
|
||||
CLAN_DIR="${REPO_ROOT}/vars/per-machine/${target}/openssh"
|
||||
if [[ -d "$CLAN_DIR" ]]; then
|
||||
logn "Clan vars for ${target} already exist — skipping."
|
||||
else
|
||||
logn "Generating host keys for ${target}..."
|
||||
run bash "${REPO_ROOT}/scripts/secrets/sync-host-keys.sh" "$target"
|
||||
fi
|
||||
done
|
||||
fi
|
||||
|
||||
# ── Phase 2.5: Prepare Proxmox node for building ─────────────────────────────
|
||||
if ! $SKIP_CREATE_VMS && ! $DRY_RUN; then
|
||||
CURRENT_BRANCH="$(git -C "$REPO_ROOT" rev-parse --abbrev-ref HEAD)"
|
||||
|
||||
# Fix /nix ownership if it exists but belongs to a different UID.
|
||||
# pve1's IPA-enrolled wayne (UID 50002) can't write to a store created by
|
||||
# another UID — passwordless sudo corrects it once.
|
||||
# Use direct SSH (no sudo) for the writability check so we test wayne's own
|
||||
# access, not root's.
|
||||
local_ssh() { ssh -i ~/.ssh/id_ed25519 "${SSH_USER}@${NODE}" "$*"; }
|
||||
if local_ssh "test -d /nix" &>/dev/null && ! local_ssh "test -w /nix" &>/dev/null; then
|
||||
logn "/nix exists but not writable by ${SSH_USER} — fixing ownership with sudo (one-time)..."
|
||||
local_ssh "sudo chown -R ${SSH_USER} /nix"
|
||||
logn "Done."
|
||||
fi
|
||||
unset -f local_ssh
|
||||
|
||||
# Ensure the remote clone is on the correct branch so create-proxmox-resource.sh
|
||||
# builds from the same commits we're deploying.
|
||||
REMOTE_REPO="/home/${SSH_USER}/nixos"
|
||||
if pve_check "test -d ${REMOTE_REPO}/.git" &>/dev/null; then
|
||||
REMOTE_BRANCH=$(ssh -i ~/.ssh/id_ed25519 "${SSH_USER}@${NODE}" \
|
||||
"cd ${REMOTE_REPO} && git rev-parse --abbrev-ref HEAD 2>/dev/null")
|
||||
if [[ "$REMOTE_BRANCH" != "$CURRENT_BRANCH" ]]; then
|
||||
logn "Remote clone is on '${REMOTE_BRANCH}', switching to '${CURRENT_BRANCH}'..."
|
||||
ssh -i ~/.ssh/id_ed25519 "${SSH_USER}@${NODE}" \
|
||||
"cd ${REMOTE_REPO} && git fetch origin && git checkout '${CURRENT_BRANCH}' && git pull --ff-only"
|
||||
logn "Done."
|
||||
fi
|
||||
fi
|
||||
fi
|
||||
|
||||
# ── Phase 3: Create VMs ───────────────────────────────────────────────────────
|
||||
|
||||
if ! $SKIP_CREATE_VMS; then
|
||||
log "Phase 3: Building and creating VMs on ${NODE}"
|
||||
|
||||
REBUILD_FLAG=""
|
||||
$FORCE_REBUILD && REBUILD_FLAG="--force-rebuild"
|
||||
|
||||
CREATE="${REPO_ROOT}/scripts/proxmox/create-proxmox-resource.sh"
|
||||
|
||||
for spec in "${VMID1}:ha-server-1:proxmox-ha-server-1" "${VMID2}:ha-server-2:proxmox-ha-server-2"; do
|
||||
IFS=: read -r vmid host_name flake_target <<< "$spec"
|
||||
log "Creating ${flake_target} (VMID ${vmid}) on ${NODE}..."
|
||||
run bash "$CREATE" \
|
||||
--type vm \
|
||||
--host "$host_name" \
|
||||
--vmid "$vmid" \
|
||||
--node "$NODE" \
|
||||
--storage "$STORAGE" \
|
||||
--memory "$MEMORY_MB" \
|
||||
--cores "$CORES" \
|
||||
${REBUILD_FLAG}
|
||||
done
|
||||
fi
|
||||
|
||||
# ── Phase 4: Add storage NIC and DRBD disk ────────────────────────────────────
|
||||
|
||||
if ! $SKIP_ADD_HARDWARE; then
|
||||
log "Phase 4: Attaching storage NIC (${STORAGE_BRIDGE}) and DRBD disk (${DRBD_DISK_GB}G) to each VM"
|
||||
for vmid in "$VMID1" "$VMID2"; do
|
||||
log " VMID ${vmid}: stopping to add hardware..."
|
||||
pve "qm stop ${vmid} --skiplock 1 2>/dev/null; sleep 3" || true
|
||||
|
||||
logn "Adding net1 (${STORAGE_BRIDGE})..."
|
||||
pve "qm set ${vmid} --net1 virtio,bridge=${STORAGE_BRIDGE},firewall=0"
|
||||
|
||||
logn "Adding scsi1 (${STORAGE}:${DRBD_DISK_GB}G for DRBD)..."
|
||||
pve "qm set ${vmid} --scsi1 ${STORAGE}:${DRBD_DISK_GB},format=raw"
|
||||
|
||||
logn "Starting VMID ${vmid}..."
|
||||
pve "qm start ${vmid}"
|
||||
done
|
||||
fi
|
||||
|
||||
# ── Phase 5: Wait for SSH ─────────────────────────────────────────────────────
|
||||
|
||||
if ! $SKIP_BOOT_WAIT; then
|
||||
log "Phase 5: Waiting for both nodes to come up"
|
||||
wait_for_ssh "$NODE1_IP" "$NODE1_HOST"
|
||||
wait_for_ssh "$NODE2_IP" "$NODE2_HOST"
|
||||
logn "Both nodes are SSHable."
|
||||
# Give systemd a few seconds to settle after activation
|
||||
sleep 10
|
||||
fi
|
||||
|
||||
# ── Phase 6: Cluster init ─────────────────────────────────────────────────────
|
||||
|
||||
if ! $SKIP_CLUSTER_INIT; then
|
||||
log "Phase 6: Initialising HA cluster"
|
||||
|
||||
CLUSTER_INIT="${REPO_ROOT}/scripts/ha/cluster-init.sh"
|
||||
[[ -x "$CLUSTER_INIT" ]] || chmod +x "$CLUSTER_INIT"
|
||||
|
||||
if $DRY_RUN; then
|
||||
logn "[dry-run] Would generate temp key, authorise on ${NODE2_HOST}, scp cluster-init.sh to ${NODE1_HOST}, and run it as root via sudo"
|
||||
else
|
||||
# Generate a temp keypair so cluster-init.sh can SSH node1→node2 as ${HA_USER}.
|
||||
# Root on node1 has no keys; a temp key bridging node1→node2 nixos solves this.
|
||||
TEMP_KEY="${REPO_ROOT}/.tmp-cluster-init-key"
|
||||
TEMP_KEY_PUB="${TEMP_KEY}.pub"
|
||||
rm -f "$TEMP_KEY" "$TEMP_KEY_PUB"
|
||||
ssh-keygen -t ed25519 -f "$TEMP_KEY" -N "" -C "cluster-init-temp-$(date +%s)" -q
|
||||
TEMP_PUBKEY=$(cat "$TEMP_KEY_PUB")
|
||||
|
||||
logn "Authorising temp key on ${NODE2_HOST} for ${HA_USER}..."
|
||||
ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no "${HA_USER}@${NODE2_IP}" \
|
||||
"mkdir -p ~/.ssh && chmod 700 ~/.ssh && echo '${TEMP_PUBKEY}' >> ~/.ssh/authorized_keys"
|
||||
|
||||
logn "Placing temp key on ${NODE1_HOST} for root..."
|
||||
scp -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no \
|
||||
"$TEMP_KEY" "${HA_USER}@${NODE1_IP}:/tmp/cluster-init-key"
|
||||
ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no "${HA_USER}@${NODE1_IP}" \
|
||||
"sudo mkdir -p /root/.ssh && sudo cp /tmp/cluster-init-key /root/.ssh/cluster-init-key && \
|
||||
sudo chmod 600 /root/.ssh/cluster-init-key && rm -f /tmp/cluster-init-key"
|
||||
|
||||
# Fix targetctl.service on both nodes: the iscsi-target.nix module bakes
|
||||
# pkgs.targetcli-fb for the targetctl binary, but targetctl is actually in
|
||||
# rtslib-fb (a different store path). Apply a runtime dropin that corrects
|
||||
# both ExecStart and ExecStop before Pacemaker ever touches the service.
|
||||
# The fixed iscsi-target.nix module will make this redundant on next rebuild.
|
||||
logn "Patching targetctl.service on both nodes..."
|
||||
_patch_targetctl() {
|
||||
local ip="$1"
|
||||
ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no "${HA_USER}@${ip}" sudo bash << 'PATCH'
|
||||
set -euo pipefail
|
||||
TC=$(find /nix/store -maxdepth 4 -path '*/python3*env/bin/targetctl' 2>/dev/null | head -1)
|
||||
PY=$(find /nix/store -maxdepth 4 -path '*/python3*env/bin/python3' -name 'python3' 2>/dev/null | \
|
||||
while IFS= read -r p; do "$p" -c "import rtslib_fb" 2>/dev/null && echo "$p" && break; done | head -1)
|
||||
[[ -n "$TC" && -n "$PY" ]] || { echo "targetctl or python3+rtslib_fb not found"; exit 1; }
|
||||
# Write stop script that saves LIO config then tears down kernel state
|
||||
"$PY" - "$TC" "$PY" << 'PYEOF'
|
||||
import sys, os, stat
|
||||
tc, py = sys.argv[1], sys.argv[2]
|
||||
script = f"""#!{py}
|
||||
import subprocess, sys, rtslib_fb
|
||||
root = rtslib_fb.RTSRoot()
|
||||
targets = list(root.targets)
|
||||
if targets:
|
||||
r = subprocess.run(["{tc}", "save", "/etc/target/saveconfig.json"], capture_output=True)
|
||||
print(f"saved {{len(targets)}} target(s); rc={{r.returncode}}")
|
||||
else:
|
||||
print("no active LIO targets")
|
||||
for t in targets:
|
||||
try:
|
||||
for tpg in list(t.tpgs): tpg.enable = False
|
||||
t.delete()
|
||||
except Exception as e: print(f"warn: {{e}}", file=sys.stderr)
|
||||
for so in list(root.storage_objects):
|
||||
try: so.delete()
|
||||
except Exception as e: print(f"warn: {{e}}", file=sys.stderr)
|
||||
print("LIO kernel target cleared")
|
||||
"""
|
||||
path = "/run/ha-targetctl-stop.py"
|
||||
with open(path, "w") as f: f.write(script)
|
||||
os.chmod(path, 0o755)
|
||||
print(f"wrote {path}")
|
||||
PYEOF
|
||||
mkdir -p /run/systemd/system/targetctl.service.d
|
||||
printf '[Service]\nExecStart=\nExecStart=%s restore /etc/target/saveconfig.json\nExecStop=\nExecStop=/run/ha-targetctl-stop.py\n' \
|
||||
"$TC" > /run/systemd/system/targetctl.service.d/fix-exec.conf
|
||||
systemctl daemon-reload
|
||||
echo "patched on $(hostname)"
|
||||
PATCH
|
||||
}
|
||||
_patch_targetctl "${NODE1_IP}"
|
||||
_patch_targetctl "${NODE2_IP}"
|
||||
|
||||
logn "Uploading cluster-init.sh to ${NODE1_HOST}..."
|
||||
scp -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no \
|
||||
"$CLUSTER_INIT" "${HA_USER}@${NODE1_IP}:/tmp/cluster-init.sh"
|
||||
|
||||
logn "Running cluster-init.sh on ${NODE1_HOST}..."
|
||||
ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no "${HA_USER}@${NODE1_IP}" \
|
||||
"sudo env NODE1=${NODE1_HOST} NODE2=${NODE2_HOST} \
|
||||
NODE1_IP=${NODE1_IP} NODE2_IP=${NODE2_IP} \
|
||||
VIP=192.168.2.229 XFS_MOUNT=/srv/ha-data \
|
||||
ISCSI_IQN=iqn.2026-01.home.sweet:ha-storage \
|
||||
VMID_NODE1=${VMID1} VMID_NODE2=${VMID2} \
|
||||
HA_USER=${HA_USER} HA_KEY=/root/.ssh/cluster-init-key \
|
||||
bash /tmp/cluster-init.sh"
|
||||
|
||||
logn "Cleaning up temp key from both nodes..."
|
||||
ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no "${HA_USER}@${NODE2_IP}" \
|
||||
"sed -i '/cluster-init-temp/d' ~/.ssh/authorized_keys" 2>/dev/null || true
|
||||
ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no "${HA_USER}@${NODE1_IP}" \
|
||||
"sudo rm -f /root/.ssh/cluster-init-key" 2>/dev/null || true
|
||||
rm -f "$TEMP_KEY" "$TEMP_KEY_PUB"
|
||||
|
||||
# Encrypt the corosync authkey generated by cluster-init and commit it.
|
||||
log " Encrypting corosync authkey into secrets/ha-corosync-authkey..."
|
||||
AUTHKEY_TMP="${REPO_ROOT}/secrets/ha-corosync-authkey.tmp"
|
||||
ssh -i ~/.ssh/id_ed25519 -o StrictHostKeyChecking=no "${HA_USER}@${NODE1_IP}" \
|
||||
"sudo cat /etc/corosync/authkey" > "$AUTHKEY_TMP"
|
||||
if [[ ! -s "$AUTHKEY_TMP" ]]; then
|
||||
err "corosync authkey on node1 is empty — cluster-init may have failed."
|
||||
fi
|
||||
mv "$AUTHKEY_TMP" "${REPO_ROOT}/secrets/ha-corosync-authkey"
|
||||
(cd "${REPO_ROOT}" && nix run nixpkgs#sops -- -e --input-type binary -i secrets/ha-corosync-authkey)
|
||||
logn "Authkey encrypted. Committing..."
|
||||
(cd "${REPO_ROOT}" && git add secrets/ha-corosync-authkey && \
|
||||
git commit -m "secrets(ha): encrypt corosync authkey generated by cluster-init")
|
||||
logn "Committed."
|
||||
fi
|
||||
fi
|
||||
|
||||
# ── Phase 7: Acceptance tests ─────────────────────────────────────────────────
|
||||
|
||||
if ! $SKIP_TESTS; then
|
||||
log "Phase 7: Running acceptance tests (T1–T7)"
|
||||
if $DRY_RUN; then
|
||||
logn "[dry-run] Would run acceptance-tests.sh against ${NODE1_HOST}/${NODE2_HOST}"
|
||||
else
|
||||
NODE1="$NODE1_HOST" NODE2="$NODE2_HOST" \
|
||||
NODE1_IP="$NODE1_IP" NODE2_IP="$NODE2_IP" \
|
||||
VIP="192.168.2.229" \
|
||||
bash "${REPO_ROOT}/scripts/ha/acceptance-tests.sh"
|
||||
fi
|
||||
fi
|
||||
|
||||
log "Deploy complete."
|
||||
@@ -1,179 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
fence_pve_ssh - Proxmox VE SSH fence agent for Pacemaker.
|
||||
|
||||
Uses SSH to reach the Proxmox host and run 'qm stop/start <vmid>'.
|
||||
Deploy to /etc/pacemaker/fence_pve_ssh on both HA nodes (chmod +x).
|
||||
|
||||
Configuration (as pacemaker stonith resource attributes):
|
||||
pve_host Proxmox host to SSH to (default: pve1.sweet.home)
|
||||
pve_user SSH user (default: wayne)
|
||||
key_file SSH private key path (default: /etc/fence-pve-ssh-key)
|
||||
vmid_node1 VMID for ha-server-1
|
||||
vmid_node2 VMID for ha-server-2
|
||||
plug Node name to act on (set by pacemaker: ha-server-1 or ha-server-2)
|
||||
action Action: off|on|reboot|status|list|metadata
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import subprocess
|
||||
import sys
|
||||
import os
|
||||
|
||||
|
||||
METADATA = """<?xml version="1.0" ?>
|
||||
<resource-agent name="fence_pve_ssh" shortdesc="Proxmox VE SSH fence agent (test lab)">
|
||||
<longdesc>Fences a VM on a Proxmox VE host by SSHing to the PVE host and
|
||||
running qm stop/start. For test use only.</longdesc>
|
||||
<vendor-url>https://proxmox.com</vendor-url>
|
||||
<parameters>
|
||||
<parameter name="action" required="1" unique="0">
|
||||
<getopt mixed="-a, --action=[action]"/>
|
||||
<content type="string" default="reboot"/>
|
||||
<shortdesc lang="en">Fencing action: off|on|reboot|status|list</shortdesc>
|
||||
</parameter>
|
||||
<parameter name="plug" required="0" unique="0">
|
||||
<getopt mixed="-n, --plug=[nodename]"/>
|
||||
<content type="string"/>
|
||||
<shortdesc lang="en">Cluster node name to fence</shortdesc>
|
||||
</parameter>
|
||||
<parameter name="pve_host" required="0" unique="0">
|
||||
<getopt mixed="--pve-host=[host]"/>
|
||||
<content type="string" default="pve1.sweet.home"/>
|
||||
<shortdesc lang="en">Proxmox VE host to SSH to</shortdesc>
|
||||
</parameter>
|
||||
<parameter name="pve_user" required="0" unique="0">
|
||||
<getopt mixed="--pve-user=[user]"/>
|
||||
<content type="string" default="wayne"/>
|
||||
<shortdesc lang="en">SSH user on the Proxmox host</shortdesc>
|
||||
</parameter>
|
||||
<parameter name="key_file" required="0" unique="0">
|
||||
<getopt mixed="--key-file=[path]"/>
|
||||
<content type="string" default="/etc/fence-pve-ssh-key"/>
|
||||
<shortdesc lang="en">SSH private key file path</shortdesc>
|
||||
</parameter>
|
||||
<parameter name="vmid_node1" required="1" unique="0">
|
||||
<getopt mixed="--vmid-node1=[vmid]"/>
|
||||
<content type="string"/>
|
||||
<shortdesc lang="en">VMID for ha-test-node1</shortdesc>
|
||||
</parameter>
|
||||
<parameter name="vmid_node2" required="1" unique="0">
|
||||
<getopt mixed="--vmid-node2=[vmid]"/>
|
||||
<content type="string"/>
|
||||
<shortdesc lang="en">VMID for ha-test-node2</shortdesc>
|
||||
</parameter>
|
||||
</parameters>
|
||||
<actions>
|
||||
<action name="off" timeout="60s"/>
|
||||
<action name="on" timeout="60s"/>
|
||||
<action name="reboot" timeout="60s"/>
|
||||
<action name="status" timeout="30s"/>
|
||||
<action name="list" timeout="10s"/>
|
||||
<action name="metadata" timeout="5s"/>
|
||||
</actions>
|
||||
</resource-agent>
|
||||
"""
|
||||
|
||||
|
||||
def parse_args():
|
||||
p = argparse.ArgumentParser(add_help=False)
|
||||
p.add_argument("-a", "--action", default="reboot")
|
||||
p.add_argument("-n", "--plug")
|
||||
p.add_argument("--pve-host", default="pve1.sweet.home")
|
||||
p.add_argument("--pve-user", default="wayne")
|
||||
p.add_argument("--key-file", default="/etc/fence-pve-ssh-key")
|
||||
p.add_argument("--vmid-node1")
|
||||
p.add_argument("--vmid-node2")
|
||||
# Allow remaining unknown args (pacemaker may pass extra ones)
|
||||
return p.parse_known_args()[0]
|
||||
|
||||
|
||||
def ssh(pve_host, pve_user, key_file, cmd):
|
||||
result = subprocess.run(
|
||||
[
|
||||
"ssh",
|
||||
"-i", key_file,
|
||||
"-o", "StrictHostKeyChecking=no",
|
||||
"-o", "BatchMode=yes",
|
||||
"-o", "ConnectTimeout=10",
|
||||
f"{pve_user}@{pve_host}",
|
||||
cmd,
|
||||
],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=30,
|
||||
)
|
||||
return result
|
||||
|
||||
|
||||
def get_vmid(args):
|
||||
node = args.plug
|
||||
if not node:
|
||||
print("ERROR: --plug not specified", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
mapping = {
|
||||
"ha-server-1": args.vmid_node1,
|
||||
"ha-server-2": args.vmid_node2,
|
||||
}
|
||||
vmid = mapping.get(node)
|
||||
if not vmid:
|
||||
print(f"ERROR: unknown node '{node}'", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
return vmid
|
||||
|
||||
|
||||
def main():
|
||||
args = parse_args()
|
||||
action = args.action.lower()
|
||||
|
||||
if action == "metadata":
|
||||
print(METADATA)
|
||||
sys.exit(0)
|
||||
|
||||
if action == "list":
|
||||
if args.vmid_node1:
|
||||
print("ha-server-1")
|
||||
if args.vmid_node2:
|
||||
print("ha-server-2")
|
||||
sys.exit(0)
|
||||
|
||||
vmid = get_vmid(args)
|
||||
|
||||
if not os.path.exists(args.key_file):
|
||||
print(f"ERROR: SSH key not found at {args.key_file}", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
if action in ("off", "reboot"):
|
||||
print(f"Stopping VM {vmid} ({args.plug}) on {args.pve_host}...")
|
||||
r = ssh(args.pve_host, args.pve_user, args.key_file,
|
||||
f"sudo /usr/sbin/qm stop {vmid}")
|
||||
if r.returncode != 0:
|
||||
print(f"ERROR stopping VM: {r.stderr}", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
print(f"VM {vmid} stopped")
|
||||
|
||||
if action in ("on", "reboot"):
|
||||
print(f"Starting VM {vmid} ({args.plug}) on {args.pve_host}...")
|
||||
r = ssh(args.pve_host, args.pve_user, args.key_file,
|
||||
f"sudo /usr/sbin/qm start {vmid}")
|
||||
if r.returncode != 0:
|
||||
print(f"ERROR starting VM: {r.stderr}", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
print(f"VM {vmid} started")
|
||||
|
||||
if action == "status":
|
||||
r = ssh(args.pve_host, args.pve_user, args.key_file,
|
||||
f"sudo /usr/sbin/qm status {vmid}")
|
||||
if r.returncode != 0:
|
||||
print(f"ERROR querying VM status: {r.stderr}", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
# qm status returns "status: running" or "status: stopped"
|
||||
status_line = r.stdout.strip()
|
||||
print(status_line)
|
||||
if "stopped" in status_line:
|
||||
sys.exit(2) # pacemaker interprets exit 2 as "off"
|
||||
sys.exit(0) # running = exit 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,206 +0,0 @@
|
||||
#!/usr/bin/env nix-shell
|
||||
#!nix-shell -i bash -p jq disko nixos-install-tools zfs
|
||||
# shellcheck shell=bash
|
||||
# The only genuinely external tools this script calls directly: `jq`
|
||||
# (parsing the `nix eval` host list), `disko`/`nixos-install` (the
|
||||
# install itself), and `zpool` (exporting a ZFS root pool before reboot,
|
||||
# see the comment above that call below). Everything disko shells out to
|
||||
# internally (parted/sgdisk/mkfs.*/zfs/...) is self-contained -- disko's
|
||||
# own generated scripts hardcode absolute Nix store paths for those, they
|
||||
# don't rely on this script's PATH at all (confirmed by inspecting a
|
||||
# generated system.build.formatScript). The built installer image
|
||||
# (modules/installer/common.nix, plus the upstream
|
||||
# installation-cd-minimal.nix it imports via iso.nix) already has all
|
||||
# four in environment.systemPackages, so this nix-shell wrapper is a
|
||||
# fast no-op there; it's what makes the script also work standalone
|
||||
# (e.g. run directly from a checkout on a stock ISO), where they aren't
|
||||
# guaranteed.
|
||||
set -eux
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
# shellcheck source=../env.sh
|
||||
source "$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)/env.sh"
|
||||
|
||||
export FLAKE_BASE_URL="git+https://${LAN_DOMAIN}/beatzaplenty/nixos.git"
|
||||
|
||||
echo "Fetching available NixOS hosts from flake..."
|
||||
# Two categories deliberately excluded from the menu:
|
||||
# lxc-* — these build a config.system.build.tarball meant for
|
||||
# `pct restore` on Proxmox directly, not an install.
|
||||
# Running nixos-install against one here would
|
||||
# bind-mount / onto /mnt and then refuse to touch the
|
||||
# filesystem it's currently running on — see
|
||||
# docs/auto-installer.md.
|
||||
# installer — this *is* the installer image's own flake target,
|
||||
# not a deployable host; "installing" it means
|
||||
# nixos-install-ing a copy of the installer into
|
||||
# itself.
|
||||
mapfile -t options < <(
|
||||
nix eval --json --no-use-registries --no-accept-flake-config --extra-experimental-features "flakes nix-command" \
|
||||
"${FLAKE_BASE_URL}#nixosConfigurations" \
|
||||
--apply builtins.attrNames \
|
||||
| jq -r '.[]
|
||||
| select(startswith("lxc-") | not)
|
||||
| select(. != "installer")'
|
||||
)
|
||||
|
||||
if [[ ${#options[@]} -eq 0 ]]; then
|
||||
echo "ERROR: No NixOS hosts found in ${FLAKE_BASE_URL}#nixosConfigurations" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "Note: lxc-* targets aren't installed this way — build them with"
|
||||
echo " nix build .#nixosConfigurations.<name>.config.system.build.tarball"
|
||||
echo "and 'pct restore' the result on Proxmox directly. See docs/auto-installer.md."
|
||||
|
||||
echo "Choose the flake profile to install:"
|
||||
select choice in "${options[@]}"; do
|
||||
if [[ -n "$choice" ]]; then
|
||||
echo "You selected: $choice"
|
||||
break
|
||||
else
|
||||
echo "Invalid selection. Try again."
|
||||
fi
|
||||
done
|
||||
|
||||
echo "Starting install with flake: ${FLAKE_BASE_URL}#${choice}"
|
||||
|
||||
# Optional: confirm before proceeding
|
||||
read -rp "Proceed with installation? (y/N): " confirm
|
||||
if [[ ! "$confirm" =~ ^[Yy]$ ]]; then
|
||||
echo "Aborted."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# A nix-cache host is *the* substituter/remote-builder for every other
|
||||
# host once installed (its own config explicitly excludes itself from
|
||||
# using either — see buildType != "nix-cache" in the nixos flake.nix).
|
||||
# Installing one shouldn't depend on a nix-cache substituter either,
|
||||
# for the same reason — plus in practice "nix-cache" only resolves over
|
||||
# Tailscale, which a fresh installer environment was never connected to
|
||||
# anyway, so it's dead weight even for non-nix-cache installs until
|
||||
# that's sorted out. Override it away here specifically for nix-cache
|
||||
# targets to keep install-time behaviour consistent with run-time.
|
||||
nix_extra_opts=()
|
||||
if [[ "${choice}" == *-nix-cache ]]; then
|
||||
echo "Installing a nix-cache host — skipping the nix-cache substituter."
|
||||
nix_extra_opts+=(--option substituters "https://cache.nixos.org/")
|
||||
fi
|
||||
|
||||
# Every host reachable through this menu has a Disko config (lxc-*
|
||||
# is filtered out above, and is the only category that doesn't —
|
||||
# see docs/auto-installer.md), so this can run unconditionally: no
|
||||
# need to probe the flake first and branch on whether Disko applies.
|
||||
disko --mode destroy,format,mount \
|
||||
--flake "${FLAKE_BASE_URL}#${choice}" "${nix_extra_opts[@]}" --yes-wipe-all-disks
|
||||
|
||||
# sops-nix derives this host's decryption key from its own SSH host key
|
||||
# at *activation* time, which runs before systemd would otherwise
|
||||
# generate one on first boot. Without pre-seeding it here, secrets
|
||||
# (including the login password) fail to decrypt on first boot.
|
||||
# Generate the key with scripts/secrets/prepare-host-key.sh first.
|
||||
#
|
||||
# Two places a key can come from, checked in order:
|
||||
# /etc/host-keys — baked into this image at build time (see
|
||||
# modules/installer/host-keys.nix; only present
|
||||
# if built with NIXOS_HOST_KEYS_DIR set)
|
||||
# /root/host-keys — scp'd in manually after boot (older fallback,
|
||||
# still supported for images built without keys)
|
||||
mkdir -p /root/host-keys
|
||||
if [[ -f "/etc/host-keys/${choice}_ssh_host_ed25519_key" ]]; then
|
||||
echo "Found baked-in SSH host key for ${choice}, installing to target..."
|
||||
install -D -m 0600 "/etc/host-keys/${choice}_ssh_host_ed25519_key" /mnt/etc/ssh/ssh_host_ed25519_key
|
||||
install -D -m 0644 "/etc/host-keys/${choice}_ssh_host_ed25519_key.pub" /mnt/etc/ssh/ssh_host_ed25519_key.pub
|
||||
elif [[ -f "/root/host-keys/${choice}_ssh_host_ed25519_key" ]]; then
|
||||
echo "Found pre-seeded SSH host key for ${choice}, installing to target..."
|
||||
install -D -m 0600 "/root/host-keys/${choice}_ssh_host_ed25519_key" /mnt/etc/ssh/ssh_host_ed25519_key
|
||||
install -D -m 0644 "/root/host-keys/${choice}_ssh_host_ed25519_key.pub" /mnt/etc/ssh/ssh_host_ed25519_key.pub
|
||||
else
|
||||
# Third place a key can come from: an arbitrary path the operator
|
||||
# points at interactively (e.g. a USB stick, a mount from another
|
||||
# machine) -- only offered when there's an actual human at the other
|
||||
# end of stdin to ask, never in a non-interactive run.
|
||||
key_copied=0
|
||||
if [[ -t 0 ]]; then
|
||||
echo "No SSH host key found for ${choice} (checked /etc/host-keys and /root/host-keys)."
|
||||
read -rp "Path to a directory containing ${choice}_ssh_host_ed25519_key(.pub) (blank to skip): " key_src_dir
|
||||
if [[ -n "$key_src_dir" && -f "${key_src_dir}/${choice}_ssh_host_ed25519_key" && -f "${key_src_dir}/${choice}_ssh_host_ed25519_key.pub" ]]; then
|
||||
cp "${key_src_dir}/${choice}_ssh_host_ed25519_key" "${key_src_dir}/${choice}_ssh_host_ed25519_key.pub" /root/host-keys/
|
||||
key_copied=1
|
||||
elif [[ -n "$key_src_dir" ]]; then
|
||||
echo "WARNING: ${choice}_ssh_host_ed25519_key(.pub) not found in ${key_src_dir}."
|
||||
fi
|
||||
fi
|
||||
|
||||
if [[ "$key_copied" -eq 1 ]]; then
|
||||
echo "Copied SSH host key for ${choice} from ${key_src_dir}, installing to target..."
|
||||
install -D -m 0600 "/root/host-keys/${choice}_ssh_host_ed25519_key" /mnt/etc/ssh/ssh_host_ed25519_key
|
||||
install -D -m 0644 "/root/host-keys/${choice}_ssh_host_ed25519_key.pub" /mnt/etc/ssh/ssh_host_ed25519_key.pub
|
||||
else
|
||||
echo "WARNING: no SSH host key found for ${choice} (checked /etc/host-keys and /root/host-keys)"
|
||||
echo "sops-nix secrets (including the login password) will NOT decrypt on first boot."
|
||||
echo "Run scripts/secrets/prepare-host-key.sh for host ${choice} on your admin workstation first,"
|
||||
echo "then either rebuild this image with NIXOS_HOST_KEYS_DIR set, scp the result to"
|
||||
echo "/root/host-keys/ on this machine, or point at it when prompted above."
|
||||
read -rp "Continue without a pre-seeded key anyway? (y/N): " skip_key
|
||||
if [[ ! "$skip_key" =~ ^[Yy]$ ]]; then
|
||||
echo "Aborted."
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
fi
|
||||
|
||||
mkdir -p /mnt/install-tmp
|
||||
export TMPDIR=/mnt/install-tmp
|
||||
|
||||
nixos-install \
|
||||
--flake "${FLAKE_BASE_URL}#${choice}" \
|
||||
"${nix_extra_opts[@]}" \
|
||||
--no-root-password
|
||||
|
||||
|
||||
rm -rf /mnt/install-tmp
|
||||
# Redundant copy of the host's private key — the real one is now at
|
||||
# /etc/ssh/ssh_host_ed25519_key. Nothing NixOS-managed ever cleans this
|
||||
# up on its own since it was written imperatively, not declaratively.
|
||||
rm -rf /root/host-keys
|
||||
|
||||
# disko's --mode ...,mount left any ZFS root pool imported (that's what
|
||||
# let nixos-install write into /mnt). If we reboot with it still
|
||||
# imported, it isn't just "not exported" -- it's stamped with *this*
|
||||
# live installer environment's hostid, which almost never matches the
|
||||
# target's own networking.hostId (see hosts/*/host.nix; the installer
|
||||
# itself sets none). modules/services/zfs/enable-service.nix and
|
||||
# modules/common/configuration.nix both set boot.zfs.forceImportRoot =
|
||||
# false deliberately (the safe option per that setting's own docs), so
|
||||
# the freshly-installed system's first real boot sees a pool "in use by
|
||||
# another system" and refuses to import it without -f -- which is what
|
||||
# makes boot stall waiting on the ZFS import. Exporting here (a no-op
|
||||
# if the chosen host has no ZFS root, e.g. proxmox-*/linode-*) clears
|
||||
# that in-use state so the next import, from any hostid, succeeds.
|
||||
#
|
||||
# Anything still mounted under /mnt -- nixos-install's own leftover
|
||||
# chroot bind mounts for running the target's activation script
|
||||
# (/mnt/dev, /mnt/proc, /mnt/sys, /mnt/run), and disko's own /mnt/boot
|
||||
# ESP mount (modules/disko/baremetal.nix) -- blocks ZFS from unmounting
|
||||
# its root dataset at /mnt, the same way any nested mount blocks
|
||||
# unmounting its parent. Confirmed live: zpool export failed with
|
||||
# "cannot unmount '/mnt': pool or dataset busy" even after handling the
|
||||
# chroot mounts alone, because /mnt/boot was still mounted too. Because
|
||||
# of this script's `set -e`, that killed the script before it ever
|
||||
# reached reboot, silently defeating the whole point of exporting first.
|
||||
# Unmounting everything under /mnt up front (recursively, so nested
|
||||
# mounts like /mnt/dev/pts come along for free) sidesteps needing to
|
||||
# enumerate every mount disko/nixos-install might leave behind.
|
||||
if mountpoint -q /mnt; then
|
||||
umount -R /mnt
|
||||
fi
|
||||
|
||||
if [[ -n "$(zpool list -H -o name 2>/dev/null)" ]]; then
|
||||
echo "Exporting ZFS pool(s) before reboot..."
|
||||
zpool export -a
|
||||
fi
|
||||
|
||||
sleep 10
|
||||
reboot
|
||||
@@ -1,303 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Add a NixOS host to the FreeIPA domain and produce a sops-encrypted keytab
|
||||
# at secrets/<hostname>.keytab, ready for modules/ipa/client.nix.
|
||||
#
|
||||
# One command replaces three error-prone manual steps:
|
||||
# 1. ipa host-add on the domain controller
|
||||
# 2. ipa-getkeytab on the domain controller + SCP back
|
||||
# 3. sops encrypt in-place (must be at secrets/<hostname>.keytab for
|
||||
# the creation rule to match -- the common mistake that breaks sops)
|
||||
#
|
||||
# Usage:
|
||||
# scripts/ipa/create-nixos-ipa-host-account.sh [options] <hostname>
|
||||
#
|
||||
# Arguments:
|
||||
# <hostname> Short hostname, e.g. "tailscale-router". The FQDN is
|
||||
# derived as <hostname>.<HOME_DOMAIN>.
|
||||
#
|
||||
# Options:
|
||||
# --ip <addr> Register this IP with the IPA host record (optional).
|
||||
# --dc <host> SSH to this host to run IPA commands.
|
||||
# Default: $IPA_SERVER (from env.sh / environment).
|
||||
# --dc-user <u> SSH user on the domain controller. Default: wayne.
|
||||
# --dry-run Print what would be done without making any changes.
|
||||
# -h, --help Show this message.
|
||||
#
|
||||
# Prereqs:
|
||||
# 1. Run from the repo root (so .sops.yaml and secrets/ are found).
|
||||
# 2. SSH access to the domain controller as --dc-user (default: wayne)
|
||||
# with passwordless sudo (or sudo cached). IPA commands and kinit run
|
||||
# as root via sudo so the Kerberos ticket is in root's cache where all
|
||||
# ipa tools expect it. If there's no valid ticket, the script runs
|
||||
# `sudo kinit admin` interactively — you'll be prompted for the IPA
|
||||
# admin password once. The password never touches this script.
|
||||
# 3. The host's age key(s) must already be in .sops.yaml. Run
|
||||
# scripts/secrets/sync-host-keys.sh <flake-target> first so the host
|
||||
# can decrypt its own keytab on boot. This script adds the .sops.yaml
|
||||
# creation rule for secrets/<hostname>.keytab automatically, but the
|
||||
# host age key anchor (&lxc-<hostname> etc.) must already exist —
|
||||
# otherwise only the admin key can decrypt the keytab and the deployed
|
||||
# host will fail to read it.
|
||||
# 4. sops in PATH, or Nix available to run it via `nix run`.
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
REPO_ROOT="$(cd "$SCRIPT_DIR/../.." && pwd)"
|
||||
|
||||
# shellcheck source=../env.sh
|
||||
source "${SCRIPT_DIR}/../env.sh"
|
||||
|
||||
# --- Argument parsing ---
|
||||
|
||||
DC_HOST="${IPA_SERVER}"
|
||||
DC_USER="wayne"
|
||||
IP_ADDR=""
|
||||
DRY_RUN=false
|
||||
TARGET=""
|
||||
|
||||
usage() {
|
||||
sed -n '/^# Usage:/,/^[^#]/{ /^#/{ s/^# \?//; p } }' "$0"
|
||||
exit "${1:-0}"
|
||||
}
|
||||
|
||||
while [[ $# -gt 0 ]]; do
|
||||
case "$1" in
|
||||
--ip) IP_ADDR="$2"; shift 2 ;;
|
||||
--dc) DC_HOST="$2"; shift 2 ;;
|
||||
--dc-user) DC_USER="$2"; shift 2 ;;
|
||||
--dry-run) DRY_RUN=true; shift ;;
|
||||
-h|--help) usage 0 ;;
|
||||
-*) echo "Unknown flag: $1" >&2; usage 1 ;;
|
||||
*)
|
||||
if [[ -n "${TARGET}" ]]; then echo "Unexpected argument: $1" >&2; usage 1; fi
|
||||
TARGET="$1"; shift
|
||||
;;
|
||||
esac
|
||||
done
|
||||
|
||||
if [[ -z "${TARGET}" ]]; then
|
||||
echo "Error: hostname required." >&2
|
||||
usage 1
|
||||
fi
|
||||
|
||||
# Reject FQDNs passed by mistake — the script appends HOME_DOMAIN itself.
|
||||
# "nixos.sweet.home" → FQDN would become "nixos.sweet.home.sweet.home".
|
||||
if [[ "${TARGET}" == *"."* ]]; then
|
||||
echo "Error: <hostname> must be the short name (e.g. 'nixos'), not a FQDN." >&2
|
||||
echo " The FQDN is derived automatically as ${TARGET}.${HOME_DOMAIN}." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
FQDN="${TARGET}.${HOME_DOMAIN}"
|
||||
KEYTAB_SECRET="${REPO_ROOT}/secrets/${TARGET}.keytab"
|
||||
# Temp path on the domain controller — use a name that won't collide.
|
||||
DC_TMP="/tmp/nixos-keytab-${TARGET}-$$.keytab"
|
||||
|
||||
# --- Helpers ---
|
||||
|
||||
log() { echo "==> $*"; }
|
||||
logn() { echo " $*"; }
|
||||
|
||||
run() {
|
||||
if $DRY_RUN; then
|
||||
echo "[dry-run] $*"
|
||||
else
|
||||
"$@"
|
||||
fi
|
||||
}
|
||||
|
||||
dc_run() {
|
||||
# Run a command string on the domain controller via SSH.
|
||||
if $DRY_RUN; then
|
||||
echo "[dry-run] ssh ${DC_USER}@${DC_HOST} $*"
|
||||
else
|
||||
ssh "${DC_USER}@${DC_HOST}" "$@"
|
||||
fi
|
||||
}
|
||||
|
||||
# --- Locate sops ---
|
||||
|
||||
if command -v sops &>/dev/null; then
|
||||
SOPS_CMD=(sops)
|
||||
else
|
||||
log "sops not in PATH — will use 'nix run nixpkgs#sops'"
|
||||
SOPS_CMD=(nix run "nixpkgs#sops" --)
|
||||
fi
|
||||
|
||||
# --- Preflight checks ---
|
||||
|
||||
cd "${REPO_ROOT}"
|
||||
|
||||
[[ -f .sops.yaml ]] || { echo "Error: .sops.yaml not found — run from repo root." >&2; exit 1; }
|
||||
[[ -d secrets ]] || { echo "Error: secrets/ not found — run from repo root." >&2; exit 1; }
|
||||
|
||||
# --- Step 1: Ensure .sops.yaml has a creation rule for this keytab ---
|
||||
#
|
||||
# sops matches creation rules against the PATH of the file being encrypted,
|
||||
# not the output path. To match secrets/<hostname>.keytab, the file must
|
||||
# already be at that path when sops -e -i is called. The creation rule must
|
||||
# also exist at that point or sops will refuse with "no matching creation
|
||||
# rules found."
|
||||
|
||||
log "Checking .sops.yaml for creation rule: secrets/${TARGET}.keytab"
|
||||
|
||||
RULE_EXISTS=false
|
||||
# Match "path_regex: secrets/<hostname>...keytab" — using .*keytab rather
|
||||
# than \.keytab because the file stores the regex verbatim (\.keytab = two
|
||||
# chars: backslash + dot), which a BRE \. (= escaped literal dot) won't span.
|
||||
if grep -q "path_regex: secrets/${TARGET}.*keytab" .sops.yaml 2>/dev/null; then
|
||||
RULE_EXISTS=true
|
||||
logn "Rule already exists — skipping addition."
|
||||
fi
|
||||
|
||||
if ! $RULE_EXISTS; then
|
||||
# Collect which platform-variant age anchors exist in .sops.yaml for this
|
||||
# hostname. The keytab is platform-agnostic (same FQDN regardless of
|
||||
# whether lxc/proxmox/linode variant is deployed), so all platform anchors
|
||||
# that have been registered get added as recipients.
|
||||
RECIPIENTS=("*admin")
|
||||
for platform in lxc proxmox linode; do
|
||||
anchor="${platform}-${TARGET}"
|
||||
if grep -q "^ - &${anchor} " .sops.yaml; then
|
||||
RECIPIENTS+=("*${anchor}")
|
||||
fi
|
||||
done
|
||||
|
||||
if [[ ${#RECIPIENTS[@]} -eq 1 ]]; then
|
||||
echo "Warning: no platform age keys found for '${TARGET}' in .sops.yaml." >&2
|
||||
echo " Run scripts/secrets/sync-host-keys.sh <flake-target> first," >&2
|
||||
echo " otherwise only the admin key can decrypt the keytab and the" >&2
|
||||
echo " deployed host won't be able to read it at boot." >&2
|
||||
echo " Continuing with admin-only encryption..." >&2
|
||||
fi
|
||||
|
||||
# Build the indented recipient list for the YAML block.
|
||||
RECIPIENT_YAML=""
|
||||
for r in "${RECIPIENTS[@]}"; do
|
||||
RECIPIENT_YAML+=" - ${r}"$'\n'
|
||||
done
|
||||
RECIPIENT_YAML="${RECIPIENT_YAML%$'\n'}" # strip trailing newline
|
||||
|
||||
NEW_RULE="
|
||||
# Host keytab for ${TARGET} FreeIPA enrollment (binary sops file).
|
||||
# Generated by scripts/ipa/create-nixos-ipa-host-account.sh.
|
||||
- path_regex: secrets/${TARGET}\\.keytab\$
|
||||
key_groups:
|
||||
- age:
|
||||
${RECIPIENT_YAML}"
|
||||
|
||||
if $DRY_RUN; then
|
||||
echo "[dry-run] Would append to .sops.yaml:"
|
||||
echo "${NEW_RULE}"
|
||||
else
|
||||
logn "Adding creation rule (recipients: ${RECIPIENTS[*]})"
|
||||
printf '%s\n' "${NEW_RULE}" >> .sops.yaml
|
||||
logn "Added."
|
||||
fi
|
||||
fi
|
||||
|
||||
# --- Step 2: Add IPA host account (idempotent) ---
|
||||
|
||||
log "Adding FreeIPA host account: ${FQDN}"
|
||||
|
||||
# Ensure there's a valid admin Kerberos ticket on the DC.
|
||||
# ipa host-add and ipa-getkeytab both need one. All IPA commands run via
|
||||
# sudo so the ticket must be in root's cache — check and refresh as root.
|
||||
# ssh -t allocates a PTY so kinit (and sudo if needed) can prompt normally;
|
||||
# no password ever touches this script or the shell history.
|
||||
if ! $DRY_RUN; then
|
||||
if ! ssh "${DC_USER}@${DC_HOST}" "sudo klist -s" &>/dev/null; then
|
||||
log "No valid Kerberos ticket on ${DC_HOST} — running sudo kinit admin"
|
||||
ssh -t "${DC_USER}@${DC_HOST}" "sudo kinit admin"
|
||||
if ! ssh "${DC_USER}@${DC_HOST}" "sudo klist -s" &>/dev/null; then
|
||||
echo "Error: kinit admin failed or produced no valid ticket." >&2
|
||||
exit 1
|
||||
fi
|
||||
else
|
||||
logn "Kerberos ticket on ${DC_HOST} is valid."
|
||||
fi
|
||||
fi
|
||||
|
||||
IP_FLAG=""
|
||||
[[ -n "${IP_ADDR}" ]] && IP_FLAG="--ip-address=${IP_ADDR}"
|
||||
|
||||
# --force: create the host record even if DNS doesn't resolve it yet.
|
||||
if $DRY_RUN; then
|
||||
echo "[dry-run] ssh ${DC_USER}@${DC_HOST} sudo ipa host-add '${FQDN}' ${IP_FLAG} --force"
|
||||
else
|
||||
HOST_ADD_OUT=$(ssh "${DC_USER}@${DC_HOST}" "sudo ipa host-add '${FQDN}' ${IP_FLAG} --force 2>&1") \
|
||||
&& HOST_ADD_RC=0 || HOST_ADD_RC=$?
|
||||
if [[ $HOST_ADD_RC -eq 0 ]]; then
|
||||
echo "${HOST_ADD_OUT}"
|
||||
elif echo "${HOST_ADD_OUT}" | grep -q "already exists"; then
|
||||
logn "(host already registered)"
|
||||
else
|
||||
echo "Error: ipa host-add failed (exit ${HOST_ADD_RC}):" >&2
|
||||
echo "${HOST_ADD_OUT}" >&2
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
|
||||
# --- Step 3: Fetch the keytab from the domain controller ---
|
||||
|
||||
log "Fetching keytab for host/${FQDN}"
|
||||
|
||||
# Remove the plaintext keytab if the script aborts before encryption completes.
|
||||
# The trap is cleared at the end of step 4 once sops has encrypted it in-place.
|
||||
trap 'rm -f "${KEYTAB_SECRET}"' EXIT
|
||||
|
||||
dc_run "sudo ipa-getkeytab -s '${IPA_SERVER}' -p 'host/${FQDN}' -k '${DC_TMP}'"
|
||||
|
||||
if $DRY_RUN; then
|
||||
echo "[dry-run] Would stream ${DC_USER}@${DC_HOST}:${DC_TMP} → secrets/${TARGET}.keytab"
|
||||
else
|
||||
logn "Streaming keytab from ${DC_HOST}:${DC_TMP} → secrets/${TARGET}.keytab"
|
||||
# scp can't read a root-owned temp file as ${DC_USER}; pipe through sudo cat instead.
|
||||
ssh "${DC_USER}@${DC_HOST}" "sudo cat '${DC_TMP}'" > "${KEYTAB_SECRET}"
|
||||
|
||||
logn "Removing temp file on ${DC_HOST}"
|
||||
dc_run "sudo rm -f '${DC_TMP}'"
|
||||
fi
|
||||
|
||||
# --- Step 4: Encrypt in-place ---
|
||||
#
|
||||
# The file must already be at secrets/<hostname>.keytab (done above) so
|
||||
# sops matches the creation rule by path. Using -i (in-place) rather than
|
||||
# stdout redirect keeps the path intact through the encrypt call.
|
||||
|
||||
log "Encrypting secrets/${TARGET}.keytab in-place with sops"
|
||||
run "${SOPS_CMD[@]}" -e --input-type binary -i "${KEYTAB_SECRET}"
|
||||
|
||||
# Encryption succeeded — the file is now sops-encrypted; cancel the cleanup trap.
|
||||
trap - EXIT
|
||||
|
||||
# --- Done ---
|
||||
|
||||
if ! $DRY_RUN; then
|
||||
echo ""
|
||||
echo "Done. secrets/${TARGET}.keytab is sops-encrypted and ready."
|
||||
echo ""
|
||||
echo "Next steps:"
|
||||
echo " 1. Verify: grep '\"data\": \"ENC' secrets/${TARGET}.keytab"
|
||||
echo " 2. Stage and commit:"
|
||||
echo " git add secrets/${TARGET}.keytab .sops.yaml"
|
||||
echo " git commit -m 'secrets: add IPA keytab for ${TARGET}'"
|
||||
echo " 3. Add to hosts/${TARGET}/host.nix (networking block and imports):"
|
||||
echo ""
|
||||
echo " networking = {"
|
||||
echo " hostName = \"${TARGET}\";"
|
||||
echo " domain = vars.homeDomain; # required for Kerberos FQDN"
|
||||
echo " nameservers = [ vars.domainControllerIp ]; # IPA DNS"
|
||||
echo " ..."
|
||||
echo " };"
|
||||
echo ""
|
||||
echo " imports = ["
|
||||
echo " (import ../../modules/ipa/client.nix {"
|
||||
echo " keytabSopsFile = ../../secrets/${TARGET}.keytab;"
|
||||
echo " caCertFile = ../../certs/ipa-ca.crt;"
|
||||
echo " })"
|
||||
echo " ];"
|
||||
echo ""
|
||||
echo " 4. Deploy: nixos-rebuild switch (or create-proxmox-resource.sh)"
|
||||
fi
|
||||
@@ -1,106 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Clan vars helpers: manage SSH host keys stored as clan vars (sops-encrypted
|
||||
# binary files under vars/per-machine/<target>/openssh/) instead of the
|
||||
# gitignored host-keys/ directory.
|
||||
#
|
||||
# Layout (per clan's convention):
|
||||
# vars/per-machine/<target>/openssh/ssh_host_ed25519_key/secret -- sops binary (admin-encrypted)
|
||||
# vars/per-machine/<target>/openssh/ssh_host_ed25519_key.pub/value -- plaintext SSH pubkey
|
||||
#
|
||||
# Sourced by create-proxmox-resource.sh and sync-host-keys.sh.
|
||||
# Depends on sops-age.sh and ssh-host-keys.sh being sourced first (for
|
||||
# sops_yaml_admin_pubkey, ssh_pubkey_to_age, and NIX_OPTS).
|
||||
|
||||
if ! declare -p NIX_OPTS >/dev/null 2>&1; then
|
||||
declare -a NIX_OPTS=()
|
||||
fi
|
||||
|
||||
# clan_ssh_key_exists <target> <repo_root>
|
||||
# Returns 0 if clan vars hold a SSH host key for <target>, 1 otherwise.
|
||||
clan_ssh_key_exists() {
|
||||
local target="$1" repo_root="$2"
|
||||
[[ -f "${repo_root}/vars/per-machine/${target}/openssh/ssh_host_ed25519_key/secret" ]]
|
||||
}
|
||||
|
||||
# clan_ssh_pubkey_path <target> <repo_root>
|
||||
# Prints the path to the plaintext SSH public key value file.
|
||||
clan_ssh_pubkey_path() {
|
||||
local target="$1" repo_root="$2"
|
||||
echo "${repo_root}/vars/per-machine/${target}/openssh/ssh_host_ed25519_key.pub/value"
|
||||
}
|
||||
|
||||
# clan_decrypt_ssh_key <target> <repo_root> <dest_dir>
|
||||
# Decrypts the sops-encrypted SSH host private key for <target> into <dest_dir>,
|
||||
# naming it <target>_ssh_host_ed25519_key (to match NIXOS_HOST_KEYS_DIR
|
||||
# conventions that lxc.nix and the disko build already expect). Also copies
|
||||
# the plaintext public key. The caller is responsible for protecting and
|
||||
# cleaning up <dest_dir>.
|
||||
clan_decrypt_ssh_key() {
|
||||
local target="$1" repo_root="$2" dest_dir="$3"
|
||||
local secret="${repo_root}/vars/per-machine/${target}/openssh/ssh_host_ed25519_key/secret"
|
||||
local pubval="${repo_root}/vars/per-machine/${target}/openssh/ssh_host_ed25519_key.pub/value"
|
||||
local dest_priv="${dest_dir}/${target}_ssh_host_ed25519_key"
|
||||
local dest_pub="${dest_dir}/${target}_ssh_host_ed25519_key.pub"
|
||||
|
||||
nix-shell "${NIX_OPTS[@]}" -p sops --run \
|
||||
"sops -d --output-type binary '${secret}'" > "$dest_priv"
|
||||
chmod 0600 "$dest_priv"
|
||||
cp "$pubval" "$dest_pub"
|
||||
}
|
||||
|
||||
# clan_generate_ssh_key <target> <repo_root>
|
||||
# Generates a new SSH host key pair and stores it in clan vars format:
|
||||
# - private key: sops binary-encrypted for the admin age key
|
||||
# - public key: plaintext value file
|
||||
# Idempotent: if the secret already exists, prints a note and returns 0.
|
||||
# Requires sops_yaml_admin_pubkey (from sops-age.sh) to be available.
|
||||
clan_generate_ssh_key() {
|
||||
local target="$1" repo_root="$2"
|
||||
local var_base="${repo_root}/vars/per-machine/${target}/openssh"
|
||||
local secret_dir="${var_base}/ssh_host_ed25519_key"
|
||||
local pubval_dir="${var_base}/ssh_host_ed25519_key.pub"
|
||||
|
||||
if [[ -f "${secret_dir}/secret" ]]; then
|
||||
echo "Clan SSH host key for ${target} already exists -- skipping generation."
|
||||
return 0
|
||||
fi
|
||||
|
||||
# Resolve admin age public key from .sops.yaml
|
||||
local admin_pubkey
|
||||
admin_pubkey="$(sops_yaml_admin_pubkey "${repo_root}/.sops.yaml")"
|
||||
if [[ -z "$admin_pubkey" ]]; then
|
||||
echo "ERROR: Could not find &admin age key in ${repo_root}/.sops.yaml" >&2
|
||||
return 1
|
||||
fi
|
||||
|
||||
# Generate the SSH key pair in a secure temp directory
|
||||
local tmpdir
|
||||
tmpdir="$(mktemp -d)"
|
||||
local priv_tmp="${tmpdir}/ssh_host_ed25519_key"
|
||||
|
||||
# shellcheck disable=SC2064
|
||||
trap "rm -rf '${tmpdir}'" RETURN
|
||||
|
||||
nix-shell "${NIX_OPTS[@]}" -p openssh --run \
|
||||
"ssh-keygen -t ed25519 -N '' -C '${target}' -f '${priv_tmp}'" >/dev/null
|
||||
|
||||
# Create a minimal sops config that uses only the admin age key -- this
|
||||
# prevents sops from merging in ALL recipients from .sops.yaml (which
|
||||
# would unnecessarily encrypt for every host's key, not just admin).
|
||||
local sops_cfg="${tmpdir}/sops-config.json"
|
||||
printf '{"creation_rules":[{"key_groups":[{"age":["%s"]}]}]}\n' \
|
||||
"$admin_pubkey" > "$sops_cfg"
|
||||
|
||||
# Encrypt the private key in sops binary format (admin-only recipient)
|
||||
mkdir -p "$secret_dir" "$pubval_dir"
|
||||
nix-shell "${NIX_OPTS[@]}" -p sops --run \
|
||||
"sops -e --config '${sops_cfg}' --input-type binary '${priv_tmp}'" \
|
||||
> "${secret_dir}/secret"
|
||||
|
||||
# Store the public key as a plaintext value file
|
||||
cp "${priv_tmp}.pub" "${pubval_dir}/value"
|
||||
|
||||
echo "Generated and stored clan SSH host key for ${target}."
|
||||
echo " Private key: ${secret_dir}/secret (sops binary, admin-key encrypted)"
|
||||
echo " Public key: ${pubval_dir}/value"
|
||||
}
|
||||
@@ -1,21 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Shared "type X to confirm" prompt for scripts/proxmox/create-proxmox-resource.sh
|
||||
# (--modify, and replacing an existing --allow-duplicate-host resource) and
|
||||
# scripts/secrets/sync-host-keys.sh (--regenerate-all-keys) -- three destructive
|
||||
# confirmations that all work the same way (echo the expected value back
|
||||
# exactly), kept in one place so the prompt/comparison logic can't drift.
|
||||
# Source alongside env.sh:
|
||||
# source "$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)/lib/confirm.sh"
|
||||
#
|
||||
# Deliberately does NOT print anything on mismatch or decide exit-vs-return
|
||||
# -- callers vary on both (a top-level script exits, a subcommand function
|
||||
# returns; wording differs too), so that stays at the call site.
|
||||
|
||||
# confirm_typed <expected> <prompt>
|
||||
# Prints <prompt> via `read -rp`, then reports (via exit status) whether the
|
||||
# typed input matched <expected> exactly.
|
||||
confirm_typed() {
|
||||
local expected="$1" prompt="$2" input
|
||||
read -rp "$prompt" input
|
||||
[[ "$input" == "$expected" ]]
|
||||
}
|
||||
@@ -1,20 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Shared Nix bootstrap for scripts/codex-setup.sh and
|
||||
# scripts/codex-maintenance.sh: the nix.conf settings both need in effect
|
||||
# before a single `nix` command runs (flakes enabled, never honor a flake
|
||||
# input's own nixConfig, no "dirty tree" warning spam), plus a helper to
|
||||
# pull an already-installed Nix's daemon/profile script onto PATH if it
|
||||
# isn't there yet. Source this instead of copying it -- see CLAUDE.md.
|
||||
export NIX_CONFIG="${NIX_CONFIG:-}
|
||||
experimental-features = nix-command flakes
|
||||
accept-flake-config = false
|
||||
warn-dirty = false
|
||||
"
|
||||
|
||||
ensure_nix_profile() {
|
||||
if [ -f /nix/var/nix/profiles/default/etc/profile.d/nix-daemon.sh ]; then
|
||||
. /nix/var/nix/profiles/default/etc/profile.d/nix-daemon.sh
|
||||
elif [ -f "$HOME/.nix-profile/etc/profile.d/nix.sh" ]; then
|
||||
. "$HOME/.nix-profile/etc/profile.d/nix.sh"
|
||||
fi
|
||||
}
|
||||
@@ -1,54 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Shared flake-introspection helpers for scripts/*.sh. Source alongside
|
||||
# env.sh:
|
||||
# source "$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)/lib/nix-eval.sh"
|
||||
#
|
||||
# NIX_EVAL_FLAGS: --no-use-registries so a call here never resolves through
|
||||
# the user's global flake registry (every call targets this repo's own
|
||||
# flake, or an explicit github: ref, not a registry alias); --no-accept-flake-config
|
||||
# so a flake input's own nixConfig (e.g. a dependency's substituters) is
|
||||
# never honored -- matches accept-flake-config = false already set repo-wide
|
||||
# (see lib/nix-bootstrap.sh / CLAUDE.md). Reuse this array rather than
|
||||
# retyping the two flags at each call site.
|
||||
declare -a NIX_EVAL_FLAGS=(--no-use-registries --no-accept-flake-config)
|
||||
|
||||
# list_flake_targets <flake_ref>
|
||||
# Prints the attribute names under <flake_ref>#nixosConfigurations, one per
|
||||
# line, e.g.:
|
||||
# list_flake_targets . # from inside the repo
|
||||
# list_flake_targets "$repo_root" # from anywhere
|
||||
list_flake_targets() {
|
||||
local flake_ref="$1"
|
||||
nix eval --json "${NIX_EVAL_FLAGS[@]}" \
|
||||
"${flake_ref}#nixosConfigurations" --apply builtins.attrNames \
|
||||
| jq -r '.[]'
|
||||
}
|
||||
|
||||
# flake_target_hostname <flake_ref> <target>
|
||||
# Prints one nixosConfigurations target's config.networking.hostName.
|
||||
# Empty (not an error under set -e) if the target doesn't exist or the
|
||||
# eval otherwise fails -- callers that need to distinguish "empty" from
|
||||
# "eval failed" should check $? themselves instead of relying on this.
|
||||
flake_target_hostname() {
|
||||
local flake_ref="$1" target="$2"
|
||||
nix eval --raw "${NIX_EVAL_FLAGS[@]}" \
|
||||
"${flake_ref}#nixosConfigurations.${target}.config.networking.hostName" 2>/dev/null
|
||||
}
|
||||
|
||||
# flake_target_lxc_privileged <flake_ref> <target>
|
||||
# Prints "true" or "false" for one lxc-* target's config.proxmoxLXC.privileged
|
||||
# (modules/platforms/lxc.nix is the single source of truth -- e.g.
|
||||
# lxc-docker sets this true so it can NFS-mount; every other lxc-* host
|
||||
# stays unprivileged). Only meaningful for lxc-* targets -- the option
|
||||
# doesn't exist for linode-*/proxmox-* (nixpkgs' proxmox-lxc.nix, which
|
||||
# declares it, is only ever imported by modules/platforms/lxc.nix). Empty
|
||||
# (not an error under set -e) if the eval fails.
|
||||
flake_target_lxc_privileged() {
|
||||
local flake_ref="$1" target="$2"
|
||||
# Not --raw: the option is a Nix boolean, and --raw can only coerce
|
||||
# strings ("cannot coerce a Boolean to a string"). Plain `nix eval`
|
||||
# prints a bare `true`/`false` for a boolean, which is exactly the
|
||||
# string this needs.
|
||||
nix eval "${NIX_EVAL_FLAGS[@]}" \
|
||||
"${flake_ref}#nixosConfigurations.${target}.config.proxmoxLXC.privileged" 2>/dev/null
|
||||
}
|
||||
@@ -1,95 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Shared parallel-nix-invocation helper for scripts/codex-maintenance.sh.
|
||||
# Source alongside nix-eval.sh:
|
||||
# source "$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)/lib/nix-parallel.sh"
|
||||
#
|
||||
# The per-host/per-package `nix eval`/`nix build --dry-run` calls in
|
||||
# codex-maintenance.sh are independent of each other, so running them one at
|
||||
# a time leaves most cores idle for most of the sweep -- run_nix_parallel
|
||||
# fans a batch of them out across up to NIX_PARALLEL_JOBS processes instead.
|
||||
|
||||
# NIX_PARALLEL_JOBS: how many `nix` invocations run_nix_parallel runs at
|
||||
# once. Defaults to core count capped by available memory (~1GB/job,
|
||||
# floor 1) rather than plain `nproc` -- each concurrent `nix eval` here
|
||||
# evaluates a whole NixOS system closure from scratch, and on a small/
|
||||
# memory-constrained CI runner, `nproc` concurrent evals can OOM-kill each
|
||||
# other (confirmed empirically: on a 4GB/6-core box, 5-6 concurrent evals
|
||||
# started getting killed while 3-4 ran clean and were still ~2x faster than
|
||||
# serial). Override via env if a given machine/CI runner has room to spare
|
||||
# or needs a tighter cap.
|
||||
default_nix_parallel_jobs() {
|
||||
local cores mem_avail_kb mem_cap
|
||||
cores="$(nproc 2>/dev/null || echo 4)"
|
||||
mem_avail_kb="$(awk '/^MemAvailable:/ {print $2}' /proc/meminfo 2>/dev/null)"
|
||||
if [[ -z "$mem_avail_kb" ]]; then
|
||||
echo "$cores"
|
||||
return
|
||||
fi
|
||||
mem_cap=$((mem_avail_kb / 1024 / 1024))
|
||||
((mem_cap < 1)) && mem_cap=1
|
||||
((mem_cap < cores)) && echo "$mem_cap" || echo "$cores"
|
||||
}
|
||||
NIX_PARALLEL_JOBS="${NIX_PARALLEL_JOBS:-$(default_nix_parallel_jobs)}"
|
||||
|
||||
# Separator between a job's label and its flake attr in the arrays
|
||||
# run_nix_parallel takes -- a control character so it can't collide with
|
||||
# anything a label or attr path would plausibly contain.
|
||||
NIX_PARALLEL_SEP=$'\x1f'
|
||||
|
||||
# run_nix_parallel <jobs_array_name> <nix subcommand + flags...>
|
||||
#
|
||||
# jobs_array_name: name of an already-populated bash array whose entries are
|
||||
# "<label>${NIX_PARALLEL_SEP}<attr>" pairs, e.g.
|
||||
# jobs=("proxmox-docker${NIX_PARALLEL_SEP}.#nixosConfigurations.proxmox-docker...drvPath")
|
||||
# Remaining args are passed to `nix` before the attr, e.g.:
|
||||
# run_nix_parallel jobs eval --raw "${NIX_EVAL_FLAGS[@]}"
|
||||
# run_nix_parallel jobs build --dry-run --no-link "${NIX_EVAL_FLAGS[@]}"
|
||||
#
|
||||
# Prints "==> <label>" followed by that job's stdout+stderr for every job,
|
||||
# in submission order (not completion order) so a run stays readable and
|
||||
# diffable across invocations even though the work itself doesn't finish in
|
||||
# that order. Returns non-zero if any job failed, only after every job has
|
||||
# finished and been printed -- same "surface everything, then fail" contract
|
||||
# a `set -e` caller gets, just parallelized instead of stopping at the first
|
||||
# failure.
|
||||
run_nix_parallel() {
|
||||
local -n jobs_ref="$1"
|
||||
shift
|
||||
local -a nix_args=("$@")
|
||||
|
||||
local n=${#jobs_ref[@]}
|
||||
[[ $n -eq 0 ]] && return 0
|
||||
|
||||
local tmp_dir
|
||||
tmp_dir="$(mktemp -d)"
|
||||
|
||||
local i=0 running=0
|
||||
for job in "${jobs_ref[@]}"; do
|
||||
local attr="${job#*"${NIX_PARALLEL_SEP}"}"
|
||||
printf '%s\n' "${job%%"${NIX_PARALLEL_SEP}"*}" >"${tmp_dir}/${i}.label"
|
||||
(
|
||||
if nix "${nix_args[@]}" "$attr" >"${tmp_dir}/${i}.out" 2>&1; then
|
||||
echo 0 >"${tmp_dir}/${i}.status"
|
||||
else
|
||||
echo 1 >"${tmp_dir}/${i}.status"
|
||||
fi
|
||||
) &
|
||||
i=$((i + 1))
|
||||
running=$((running + 1))
|
||||
if ((running >= NIX_PARALLEL_JOBS)); then
|
||||
wait -n
|
||||
running=$((running - 1))
|
||||
fi
|
||||
done
|
||||
wait
|
||||
|
||||
local failed=0 j
|
||||
for ((j = 0; j < n; j++)); do
|
||||
echo "==> $(cat "${tmp_dir}/${j}.label")"
|
||||
cat "${tmp_dir}/${j}.out"
|
||||
[[ "$(cat "${tmp_dir}/${j}.status")" -ne 0 ]] && failed=1
|
||||
done
|
||||
|
||||
rm -rf "$tmp_dir"
|
||||
return $failed
|
||||
}
|
||||
@@ -1,52 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Shared sops/age helpers for scripts/secrets/backup-admin-key.sh,
|
||||
# scripts/secrets/rotate-admin-key.sh, and scripts/secrets/sync-host-keys.sh -- all three
|
||||
# derive an age public key from a private identity file the same way, two
|
||||
# of them resolve the same sops/age default key-file path, and two of them
|
||||
# run `sops updatekeys` the same way. Kept in one place so they can't drift
|
||||
# apart. Source alongside env.sh:
|
||||
# source "$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)/lib/sops-age.sh"
|
||||
#
|
||||
# Uses NIX_OPTS (an array of extra `nix-shell` options -- see env.sh's
|
||||
# nix_extra_opts) if the caller has already set it, same convention as
|
||||
# lib/ssh-host-keys.sh. Falls back to no extra options if the caller never
|
||||
# sourced env.sh.
|
||||
if ! declare -p NIX_OPTS >/dev/null 2>&1; then
|
||||
declare -a NIX_OPTS=()
|
||||
fi
|
||||
|
||||
# sops/age's own default identity-file resolution order, minus $SOPS_AGE_KEY
|
||||
# itself (an inline identity, not a path -- callers that accept it check it
|
||||
# separately, before falling back to this).
|
||||
: "${DEFAULT_SOPS_AGE_KEY_FILE:=${SOPS_AGE_KEY_FILE:-${XDG_CONFIG_HOME:-$HOME/.config}/sops/age/keys.txt}}"
|
||||
|
||||
# age_pubkey_from_identity_file <identity-file>
|
||||
# Prints the age public key for a private identity file (age-keygen -y).
|
||||
age_pubkey_from_identity_file() {
|
||||
local identity_file="$1"
|
||||
nix-shell "${NIX_OPTS[@]}" -p age --run "age-keygen -y '${identity_file}'"
|
||||
}
|
||||
|
||||
# sops_yaml_admin_pubkey <sops-yaml-path>
|
||||
# Prints .sops.yaml's current &admin age public key, or empty (not an error
|
||||
# under set -e) if no such anchor line exists -- callers that need to treat
|
||||
# "missing" as fatal check for an empty result themselves.
|
||||
sops_yaml_admin_pubkey() {
|
||||
local sops_yaml="$1"
|
||||
grep -E '^ - &admin age1' "$sops_yaml" 2>/dev/null | awk '{print $NF}' || true
|
||||
}
|
||||
|
||||
# sops_updatekeys <secrets-file> [key-file]
|
||||
# Re-encrypts <secrets-file> for .sops.yaml's current recipient set. If
|
||||
# <key-file> is given, decrypts with that identity (SOPS_AGE_KEY_FILE)
|
||||
# instead of whatever's ambient -- needed when the ambient default key
|
||||
# doesn't match yet (e.g. mid-rotation, decrypting with the outgoing key).
|
||||
sops_updatekeys() {
|
||||
local secrets_file="$1" key_file="${2:-}"
|
||||
if [[ -n "$key_file" ]]; then
|
||||
SOPS_AGE_KEY_FILE="$key_file" nix-shell "${NIX_OPTS[@]}" -p sops --run \
|
||||
"sops updatekeys --yes '${secrets_file}'"
|
||||
else
|
||||
nix-shell "${NIX_OPTS[@]}" -p sops --run "sops updatekeys --yes '${secrets_file}'"
|
||||
fi
|
||||
}
|
||||
@@ -1,30 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Shared SSH-host-key / age-conversion helpers for scripts/secrets/sync-host-keys.sh
|
||||
# and scripts/secrets/prepare-host-key.sh -- both generate the same kind of key
|
||||
# (ed25519, no passphrase, the sops-nix age-derivation input) and convert it
|
||||
# to an age recipient the same way; kept in one place so the two can't
|
||||
# drift apart.
|
||||
#
|
||||
# Uses NIX_OPTS (an array of extra `nix-shell` options -- see env.sh's
|
||||
# nix_extra_opts) if the caller has already set it, so a decision to avoid
|
||||
# an unreachable nix-cache is reused here instead of probed again. Falls
|
||||
# back to no extra options if the caller never sourced env.sh.
|
||||
if ! declare -p NIX_OPTS >/dev/null 2>&1; then
|
||||
declare -a NIX_OPTS=()
|
||||
fi
|
||||
|
||||
# generate_host_ed25519_key <hostname> <keyfile>
|
||||
# Writes <keyfile> and <keyfile>.pub. Caller is responsible for refusing to
|
||||
# overwrite an existing keyfile -- this always runs ssh-keygen fresh.
|
||||
generate_host_ed25519_key() {
|
||||
local hostname="$1" keyfile="$2"
|
||||
nix-shell "${NIX_OPTS[@]}" -p openssh --run \
|
||||
"ssh-keygen -t ed25519 -N '' -C '${hostname}' -f '${keyfile}'" >/dev/null
|
||||
}
|
||||
|
||||
# ssh_pubkey_to_age <pubkeyfile>
|
||||
# Prints the age public key derived from an ed25519 SSH public key file.
|
||||
ssh_pubkey_to_age() {
|
||||
local pubkeyfile="$1"
|
||||
nix-shell "${NIX_OPTS[@]}" -p ssh-to-age --run "ssh-to-age -i '${pubkeyfile}'"
|
||||
}
|
||||
@@ -2,7 +2,7 @@
|
||||
# Generates a new machine's SSH host key by an arbitrary name, before it
|
||||
# necessarily has a flake target yet -- prints the .sops.yaml snippet to
|
||||
# add by hand. For any host that already has a flake target,
|
||||
# scripts/secrets/sync-host-keys.sh <target> does this same job plus the
|
||||
# scripts/sync-host-keys.sh <target> does this same job plus the
|
||||
# .sops.yaml/key_groups registration and re-encryption automatically; use
|
||||
# this script only to pre-generate a key ahead of adding the flake target
|
||||
# itself.
|
||||
@@ -22,13 +22,9 @@
|
||||
# new machine.
|
||||
set -euo pipefail
|
||||
|
||||
repo_root="$(cd "$(dirname "$0")/../.." && pwd)"
|
||||
# shellcheck source=../env.sh
|
||||
source "${repo_root}/scripts/env.sh"
|
||||
# shellcheck source=../lib/ssh-host-keys.sh
|
||||
source "${repo_root}/scripts/lib/ssh-host-keys.sh"
|
||||
repo_root="$(cd "$(dirname "$0")/.." && pwd)"
|
||||
|
||||
hostname="${1:?usage: scripts/secrets/prepare-host-key.sh <hostname>}"
|
||||
hostname="${1:?usage: scripts/prepare-host-key.sh <hostname>}"
|
||||
sops_yaml="${repo_root}/.sops.yaml"
|
||||
|
||||
if [[ ! -f "$sops_yaml" ]]; then
|
||||
@@ -41,15 +37,13 @@ mkdir -p "$keydir"
|
||||
keyfile="${keydir}/${hostname}_ssh_host_ed25519_key"
|
||||
|
||||
if [[ -f "$keyfile" ]]; then
|
||||
echo "Key already exists: ${keyfile}"
|
||||
echo "Reusing the existing key. Remove it first if you want to regenerate."
|
||||
exit 0
|
||||
echo "ERROR: $keyfile already exists. Remove it first if you want to regenerate." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
nix_extra_opts
|
||||
generate_host_ed25519_key "$hostname" "$keyfile"
|
||||
nix-shell -p openssh --run "ssh-keygen -t ed25519 -N '' -C '${hostname}' -f '${keyfile}'" >/dev/null
|
||||
|
||||
age_pub="$(ssh_pubkey_to_age "${keyfile}.pub")"
|
||||
age_pub="$(nix-shell -p ssh-to-age --run "ssh-to-age -i '${keyfile}.pub'")"
|
||||
|
||||
cat <<EOF
|
||||
|
||||
@@ -1,221 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Ad hoc clone of a single VM/CT from pve1 (production) to pve-test
|
||||
# (sandbox), via vzdump + qmrestore/pct restore -- not a general-purpose
|
||||
# backup tool, just a quick "give me a disposable copy of this thing on
|
||||
# pve-test" for testing against real-ish data without touching prod.
|
||||
#
|
||||
# Flow:
|
||||
# 1. vzdump the resource on pve1 into its "local" storage (--mode
|
||||
# snapshot by default, so the source keeps running throughout --
|
||||
# see --mode below for when that's not possible).
|
||||
# 2. Stream the resulting archive straight from pve1 to pve-test
|
||||
# (ssh pve1 cat ... | ssh pve-test cat > ...) -- this machine is
|
||||
# just the relay, no separate on-disk staging copy here.
|
||||
# 3. qmrestore / pct restore it on pve-test under --new-vmid (default:
|
||||
# same VMID as the source -- pve-test is a separate node/cluster, so
|
||||
# no collision unless that VMID is already in use there too).
|
||||
# Always restored with --unique 1 (fresh MAC addresses) since the
|
||||
# source is typically still running on the same LAN -- restoring
|
||||
# with the *same* MAC would put two live guests on the wire with
|
||||
# identical hardware addresses.
|
||||
# 4. Delete the vzdump archive from pve1's local storage and the
|
||||
# relayed copy on pve-test, so neither node accumulates ad hoc
|
||||
# backup files from this script. Only the pve1 original is
|
||||
# preserved on any failure after step 1, so a failed
|
||||
# transfer/restore can be retried without re-running the backup.
|
||||
#
|
||||
# This script's own defaults are pve1 -> pve-test, unlike
|
||||
# create-proxmox-resource.sh's --node (which defaults to production) --
|
||||
# see CLAUDE.md's "Two Proxmox nodes" section. pve1 is only ever touched
|
||||
# here after typing the source VMID back to confirm; pve-test is treated
|
||||
# as disposable, matching this repo's usual policy for that node.
|
||||
#
|
||||
# See --help for the full option list.
|
||||
set -euo pipefail
|
||||
|
||||
repo_root="$(cd "$(dirname "$0")/../.." && pwd)"
|
||||
# shellcheck source=../env.sh
|
||||
source "${repo_root}/scripts/env.sh"
|
||||
# shellcheck source=../lib/confirm.sh
|
||||
source "${repo_root}/scripts/lib/confirm.sh"
|
||||
|
||||
usage() {
|
||||
cat <<EOF
|
||||
Usage: $0 --vmid <n> [options]
|
||||
|
||||
--vmid <n> Required: VMID on the source node to clone.
|
||||
Kind (qemu VM vs LXC CT) is auto-detected.
|
||||
--new-vmid <n> VMID to restore as on the target node
|
||||
(default: same as --vmid).
|
||||
--mode snapshot|suspend|stop
|
||||
vzdump backup mode (default: snapshot -- the
|
||||
source resource keeps running throughout;
|
||||
requires snapshot-capable storage, e.g.
|
||||
ZFS/LVM-thin/Ceph/qcow2). Fall back to
|
||||
"suspend" (brief pause) or "stop" (source
|
||||
goes down for the duration) if the source's
|
||||
storage doesn't support live snapshots --
|
||||
vzdump's own error will say so.
|
||||
--source-node <host> (default: \$PVE1_HOST, ${PVE1_HOST})
|
||||
--target-node <host> (default: \$PVE_TEST_HOST, ${PVE_TEST_HOST})
|
||||
--source-storage <pool> Where vzdump writes the backup on the
|
||||
source node (default: local).
|
||||
--target-storage <pool> Where the restored disk/rootfs lands on
|
||||
the target node (default: \$PROXMOX_STORAGE, ${PROXMOX_STORAGE}).
|
||||
--keep-backup Don't delete the vzdump archive from
|
||||
either node afterward (debugging aid).
|
||||
--yes Skip the typed VMID confirmation
|
||||
before touching the source node.
|
||||
--dry-run Print the full plan and skip every
|
||||
mutating step (vzdump, transfer,
|
||||
restore, delete) and the confirm
|
||||
prompt. Still makes read-only SSH
|
||||
calls to look up the source kind
|
||||
and check the target VMID is free
|
||||
-- harmless on either node.
|
||||
-h, --help
|
||||
EOF
|
||||
}
|
||||
|
||||
vmid=""
|
||||
new_vmid=""
|
||||
mode="snapshot"
|
||||
source_node="$PVE1_HOST"
|
||||
target_node="$PVE_TEST_HOST"
|
||||
source_storage="local"
|
||||
target_storage="$PROXMOX_STORAGE"
|
||||
keep_backup=0
|
||||
skip_confirm=0
|
||||
dry_run=0
|
||||
|
||||
while [[ $# -gt 0 ]]; do
|
||||
case "$1" in
|
||||
--vmid) vmid="$2"; shift 2 ;;
|
||||
--new-vmid) new_vmid="$2"; shift 2 ;;
|
||||
--mode) mode="$2"; shift 2 ;;
|
||||
--source-node) source_node="$2"; shift 2 ;;
|
||||
--target-node) target_node="$2"; shift 2 ;;
|
||||
--source-storage) source_storage="$2"; shift 2 ;;
|
||||
--target-storage) target_storage="$2"; shift 2 ;;
|
||||
--keep-backup) keep_backup=1; shift ;;
|
||||
--yes) skip_confirm=1; shift ;;
|
||||
--dry-run) dry_run=1; shift ;;
|
||||
-h | --help) usage; exit 0 ;;
|
||||
*) echo "Unknown option: $1" >&2; usage >&2; exit 1 ;;
|
||||
esac
|
||||
done
|
||||
|
||||
if [[ -z "$vmid" ]]; then
|
||||
echo "ERROR: --vmid is required." >&2
|
||||
usage >&2
|
||||
exit 1
|
||||
fi
|
||||
if [[ "$mode" != "snapshot" && "$mode" != "suspend" && "$mode" != "stop" ]]; then
|
||||
echo "ERROR: --mode must be snapshot, suspend, or stop." >&2
|
||||
exit 1
|
||||
fi
|
||||
[[ -z "$new_vmid" ]] && new_vmid="$vmid"
|
||||
|
||||
source_target="${PROXMOX_SSH_USER}@${source_node}"
|
||||
target_target="${PROXMOX_SSH_USER}@${target_node}"
|
||||
|
||||
# No dry-run wrapper needed for the calls below: every mutating step
|
||||
# (vzdump, transfer, restore, delete) is reached only after the --dry-run
|
||||
# early-exit further down, so a plain `ssh` call is never in the dry-run
|
||||
# path.
|
||||
|
||||
# --- identify the resource kind on the source node -----------------------
|
||||
echo "==> Looking up VMID ${vmid} on ${source_node}..."
|
||||
kind=""
|
||||
if ssh "$source_target" "qm status ${vmid}" >/dev/null 2>&1; then
|
||||
kind="vm"
|
||||
elif ssh "$source_target" "pct status ${vmid}" >/dev/null 2>&1; then
|
||||
kind="lxc"
|
||||
else
|
||||
echo "ERROR: VMID ${vmid} doesn't exist on ${source_node} as either a VM or CT." >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "VMID ${vmid} on ${source_node} is a ${kind}."
|
||||
|
||||
# --- refuse to clobber an existing resource on the target node -----------
|
||||
if ssh "$target_target" "qm status ${new_vmid}" >/dev/null 2>&1 \
|
||||
|| ssh "$target_target" "pct status ${new_vmid}" >/dev/null 2>&1; then
|
||||
echo "ERROR: VMID ${new_vmid} already exists on ${target_node}. Pass --new-vmid" >&2
|
||||
echo "with a free ID, or remove the existing resource there first." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo
|
||||
echo "Plan:"
|
||||
echo " source: ${kind} VMID ${vmid} on ${source_node} (storage: ${source_storage}, mode: ${mode})"
|
||||
echo " target: VMID ${new_vmid} on ${target_node} (storage: ${target_storage}, fresh MAC via --unique)"
|
||||
[[ "$keep_backup" -eq 1 ]] && echo " backup archives are kept on both nodes afterward (--keep-backup)"
|
||||
|
||||
if [[ "$dry_run" -eq 1 ]]; then
|
||||
echo
|
||||
echo "[dry-run] No backup, transfer, restore, or delete was performed."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
if [[ "$skip_confirm" -ne 1 ]]; then
|
||||
echo
|
||||
if ! confirm_typed "$vmid" "Type the source VMID (${vmid}) to confirm backing it up from ${source_node}: "; then
|
||||
echo "Cancelled -- input didn't match ${vmid}." >&2
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
|
||||
# --- vzdump on the source node --------------------------------------------
|
||||
echo
|
||||
echo "==> Backing up VMID ${vmid} on ${source_node} (mode=${mode}, storage=${source_storage})..."
|
||||
vzdump_log="$(ssh "$source_target" \
|
||||
"vzdump ${vmid} --mode ${mode} --storage ${source_storage} --compress zstd" 2>&1)" \
|
||||
|| {
|
||||
echo "$vzdump_log" >&2
|
||||
echo "ERROR: vzdump failed on ${source_node}." >&2
|
||||
exit 1
|
||||
}
|
||||
echo "$vzdump_log"
|
||||
|
||||
archive="$(echo "$vzdump_log" | grep -oP "creating vzdump archive '\K[^']+" | tail -n1)"
|
||||
if [[ -z "$archive" ]]; then
|
||||
echo "ERROR: couldn't find the archive path in vzdump's output above." >&2
|
||||
exit 1
|
||||
fi
|
||||
archive_basename="$(basename "$archive")"
|
||||
target_tmp_archive="/var/tmp/${archive_basename}"
|
||||
echo "Archive: ${archive}"
|
||||
|
||||
# Always clean up the relayed copy on the target node, success or failure
|
||||
# -- it's only ever a working copy, restored or not.
|
||||
cleanup_target_tmp() {
|
||||
if [[ "$keep_backup" -ne 1 ]]; then
|
||||
ssh "$target_target" "rm -f '${target_tmp_archive}'" >/dev/null 2>&1 || true
|
||||
fi
|
||||
}
|
||||
trap cleanup_target_tmp EXIT
|
||||
|
||||
# --- relay the archive from source to target ------------------------------
|
||||
echo
|
||||
echo "==> Transferring archive to ${target_node}..."
|
||||
ssh "$source_target" "cat '${archive}'" | ssh "$target_target" "cat > '${target_tmp_archive}'"
|
||||
|
||||
# --- restore on the target node --------------------------------------------
|
||||
echo
|
||||
echo "==> Restoring as VMID ${new_vmid} on ${target_node} (storage=${target_storage})..."
|
||||
if [[ "$kind" == "vm" ]]; then
|
||||
ssh "$target_target" "qmrestore '${target_tmp_archive}' ${new_vmid} --storage ${target_storage} --unique 1"
|
||||
else
|
||||
ssh "$target_target" "pct restore ${new_vmid} '${target_tmp_archive}' --storage ${target_storage} --unique 1"
|
||||
fi
|
||||
|
||||
# --- clean up the source backup now that the restore succeeded -----------
|
||||
if [[ "$keep_backup" -ne 1 ]]; then
|
||||
echo
|
||||
echo "==> Deleting backup archive from ${source_node}'s ${source_storage} storage..."
|
||||
ssh "$source_target" "rm -f '${archive}' '${archive}.notes' '${archive}.log'" >/dev/null 2>&1 || true
|
||||
fi
|
||||
|
||||
echo
|
||||
echo "Done. VMID ${new_vmid} (${kind}) is now on ${target_node}, cloned from" \
|
||||
"VMID ${vmid} on ${source_node}."
|
||||
@@ -1,199 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Points a non-NixOS Debian machine's Nix install at nix-cache: adds it as
|
||||
# a substituter (with cache.nixos.org kept as fallback) and, once the
|
||||
# remote-builder private key is installed, as a distributed-build machine
|
||||
# too.
|
||||
#
|
||||
# This is the non-NixOS equivalent of modules/nix-cache/client.nix +
|
||||
# modules/nix-cache/remote-builder-client.nix -- those two only apply to
|
||||
# hosts built from this flake. A plain Debian box with Nix installed has no
|
||||
# NixOS module system to pick that config up, so this edits nix.conf by hand.
|
||||
#
|
||||
# Two modes depending on who runs it:
|
||||
#
|
||||
# root (multi-user / daemon install):
|
||||
# Writes /etc/nix/nix.conf, /etc/ssh/ssh_known_hosts, restarts nix-daemon.
|
||||
# Requires /etc/nix/nix.conf to already exist (i.e. nix-daemon is set up).
|
||||
# Run as: sudo ./configure-nix-cache-client.sh [options]
|
||||
#
|
||||
# non-root (single-user install):
|
||||
# Writes ~/.config/nix/nix.conf, ~/.ssh/known_hosts. No daemon to restart.
|
||||
# Run as: ./configure-nix-cache-client.sh [options]
|
||||
#
|
||||
# The values below mirror variables.nix / modules/nix-cache/client.nix in
|
||||
# this repo -- update both if nix-cache is ever rebuilt with a new host
|
||||
# key or the cache signing key is rotated (see docs/nix-cache.md).
|
||||
#
|
||||
# REMOTE_BUILDER_KEY defaults to the running user's default SSH identity
|
||||
# (root: /root/.ssh/id_ed25519, other user: ~/.ssh/id_ed25519). That key
|
||||
# must be listed in vars.remoteBuilderAuthorizedKeys in this repo and
|
||||
# nix-cache rebuilt before remote building works.
|
||||
#
|
||||
# Usage:
|
||||
# ./configure-nix-cache-client.sh [--dry-run] [--no-remote-builder] [--no-restart]
|
||||
#
|
||||
# Env overrides (defaults match variables.nix):
|
||||
# NIX_CACHE_HOST, NIX_CACHE_HOST_KEY, REMOTE_BUILDER_USER, REMOTE_BUILDER_KEY
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
: "${NIX_CACHE_HOST:=nix-cache}"
|
||||
: "${NIX_CACHE_HOST_KEY:=ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAICuHUxGNH6ei3BZD+EfZs3l4X8uJNcjQiOsM/G4yo4O/ lxc-nix-cache}"
|
||||
: "${REMOTE_BUILDER_USER:=nixremote}"
|
||||
|
||||
CACHE_PUB_KEY="cache.local-1:usoWYanY3Kpq2+kDIS2nhWoLZiRxanmdysdzqCFBHW4="
|
||||
FALLBACK_URL="https://cache.nixos.org/"
|
||||
FALLBACK_PUB_KEY="cache.nixos.org-1:6NCHdD59X431o0gWypbMrAURkbJ16ZPMQFGspcDShjY="
|
||||
|
||||
MARKER_BEGIN="# BEGIN nix-cache client config (configure-nix-cache-client.sh)"
|
||||
MARKER_END="# END nix-cache client config"
|
||||
|
||||
# Mode: root uses system-wide paths and restarts the daemon; non-root uses
|
||||
# user-level paths and has no daemon to restart.
|
||||
if [[ "$EUID" -eq 0 ]]; then
|
||||
install_mode="multi"
|
||||
NIX_CONF="/etc/nix/nix.conf"
|
||||
KNOWN_HOSTS="/etc/ssh/ssh_known_hosts"
|
||||
: "${REMOTE_BUILDER_KEY:=/root/.ssh/id_ed25519}"
|
||||
else
|
||||
install_mode="single"
|
||||
NIX_CONF="${XDG_CONFIG_HOME:-$HOME/.config}/nix/nix.conf"
|
||||
KNOWN_HOSTS="$HOME/.ssh/known_hosts"
|
||||
: "${REMOTE_BUILDER_KEY:=$HOME/.ssh/id_ed25519}"
|
||||
fi
|
||||
|
||||
dry_run=0
|
||||
with_remote_builder=1
|
||||
restart_daemon=1
|
||||
|
||||
for arg in "$@"; do
|
||||
case "$arg" in
|
||||
--dry-run) dry_run=1 ;;
|
||||
--no-remote-builder) with_remote_builder=0 ;;
|
||||
--no-restart) restart_daemon=0 ;;
|
||||
-h|--help)
|
||||
sed -n '2,37p' "$0"
|
||||
exit 0
|
||||
;;
|
||||
*)
|
||||
echo "ERROR: unknown argument: $arg" >&2
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
done
|
||||
|
||||
if ! command -v nix >/dev/null 2>&1; then
|
||||
echo "ERROR: no 'nix' binary on PATH -- install the Nix package manager first." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if [[ "$install_mode" == "multi" && ! -f "$NIX_CONF" ]]; then
|
||||
echo "ERROR: $NIX_CONF not found -- expected an existing multi-user Nix install." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Single-user: create the config file if it doesn't exist yet.
|
||||
if [[ "$install_mode" == "single" && "$dry_run" -eq 0 ]]; then
|
||||
mkdir -p "$(dirname "$NIX_CONF")"
|
||||
[[ -f "$NIX_CONF" ]] || touch "$NIX_CONF"
|
||||
fi
|
||||
|
||||
builder_line=""
|
||||
if [[ "$with_remote_builder" -eq 1 ]]; then
|
||||
if [[ -f "$REMOTE_BUILDER_KEY" ]]; then
|
||||
case "$(uname -m)" in
|
||||
x86_64) nix_system="x86_64-linux" ;;
|
||||
aarch64) nix_system="aarch64-linux" ;;
|
||||
*)
|
||||
echo "WARNING: unrecognized architecture '$(uname -m)' -- skipping remote builder, keeping substituter config." >&2
|
||||
with_remote_builder=0
|
||||
;;
|
||||
esac
|
||||
if [[ "$with_remote_builder" -eq 1 ]]; then
|
||||
builder_line="builders = ssh://${REMOTE_BUILDER_USER}@${NIX_CACHE_HOST} ${nix_system} ${REMOTE_BUILDER_KEY} 4 2 big-parallel,kvm,nixos-test,benchmark"
|
||||
fi
|
||||
else
|
||||
echo "WARNING: $REMOTE_BUILDER_KEY not found -- skipping remote builder config (substituter still configured)." >&2
|
||||
echo " See docs/nix-cache.md 'Remote builder SSH keys' for how to install it, then re-run this script." >&2
|
||||
with_remote_builder=0
|
||||
fi
|
||||
fi
|
||||
|
||||
block="$(cat <<EOF
|
||||
$MARKER_BEGIN
|
||||
extra-substituters = http://${NIX_CACHE_HOST} ${FALLBACK_URL}
|
||||
extra-trusted-public-keys = ${CACHE_PUB_KEY} ${FALLBACK_PUB_KEY}
|
||||
EOF
|
||||
)"
|
||||
if [[ "$with_remote_builder" -eq 1 ]]; then
|
||||
block="${block}
|
||||
builders-use-substitutes = true
|
||||
${builder_line}"
|
||||
fi
|
||||
block="${block}
|
||||
$MARKER_END"
|
||||
|
||||
echo "== nix.conf block to install ($NIX_CONF) =="
|
||||
echo "$block"
|
||||
echo "================================"
|
||||
|
||||
if [[ "$dry_run" -eq 1 ]]; then
|
||||
echo "(--dry-run: not writing $NIX_CONF)"
|
||||
else
|
||||
tmp_conf="$(mktemp)"
|
||||
trap 'rm -f "$tmp_conf"' EXIT
|
||||
|
||||
if grep -qF "$MARKER_BEGIN" "$NIX_CONF"; then
|
||||
awk -v begin="$MARKER_BEGIN" -v end="$MARKER_END" -v block="$block" '
|
||||
$0 == begin { print block; skip = 1; next }
|
||||
$0 == end { skip = 0; next }
|
||||
skip { next }
|
||||
{ print }
|
||||
' "$NIX_CONF" > "$tmp_conf"
|
||||
else
|
||||
cp "$NIX_CONF" "$tmp_conf"
|
||||
printf '\n%s\n' "$block" >> "$tmp_conf"
|
||||
fi
|
||||
|
||||
cp "$NIX_CONF" "${NIX_CONF}.bak.$(date +%Y%m%d%H%M%S)"
|
||||
install -m 0644 "$tmp_conf" "$NIX_CONF"
|
||||
echo "Updated $NIX_CONF (backup saved alongside it)."
|
||||
fi
|
||||
|
||||
if [[ "$with_remote_builder" -eq 1 ]]; then
|
||||
known_hosts_line="${NIX_CACHE_HOST} ${NIX_CACHE_HOST_KEY}"
|
||||
if [[ "$dry_run" -eq 1 ]]; then
|
||||
echo "(--dry-run: would ensure this line is present in $KNOWN_HOSTS)"
|
||||
echo " $known_hosts_line"
|
||||
else
|
||||
mkdir -p "$(dirname "$KNOWN_HOSTS")"
|
||||
touch "$KNOWN_HOSTS"
|
||||
if ! grep -qF "$known_hosts_line" "$KNOWN_HOSTS" 2>/dev/null; then
|
||||
echo "$known_hosts_line" >> "$KNOWN_HOSTS"
|
||||
echo "Added nix-cache's SSH host key to $KNOWN_HOSTS."
|
||||
fi
|
||||
fi
|
||||
fi
|
||||
|
||||
# Only restart the daemon for multi-user installs -- single-user has no daemon.
|
||||
if [[ "$dry_run" -eq 0 && "$restart_daemon" -eq 1 && "$install_mode" == "multi" ]]; then
|
||||
if command -v systemctl >/dev/null 2>&1 && systemctl is-active --quiet nix-daemon 2>/dev/null; then
|
||||
systemctl restart nix-daemon
|
||||
echo "Restarted nix-daemon to pick up the new config."
|
||||
else
|
||||
echo "nix-daemon not managed by systemd (or not running) -- restart it manually to pick up the new config."
|
||||
fi
|
||||
fi
|
||||
|
||||
cat <<EOF
|
||||
|
||||
Done. Verify with:
|
||||
curl http://${NIX_CACHE_HOST}/nix-cache-info
|
||||
nix show-config | grep -E 'substituters|trusted-public-keys|builders'
|
||||
EOF
|
||||
if [[ "$with_remote_builder" -eq 1 ]]; then
|
||||
cat <<EOF
|
||||
ssh -i ${REMOTE_BUILDER_KEY} ${REMOTE_BUILDER_USER}@${NIX_CACHE_HOST} nix-store --version
|
||||
nix build nixpkgs#hello -L
|
||||
EOF
|
||||
fi
|
||||
@@ -1,971 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Creates new Proxmox VMs/LXC containers from this flake, and reconfigures
|
||||
# existing ones -- the manual workflows in docs/proxmox-images.md (VM) and
|
||||
# docs/auto-installer.md's "LXC hosts" section (container), automated.
|
||||
#
|
||||
# Images are built directly on the Proxmox node (PROXMOX_REMOTE_REPO_DIR /
|
||||
# --remote-repo-dir in scripts/env.sh), not on whatever machine runs this
|
||||
# script -- there's no multi-gigabyte image to transfer afterward. The first
|
||||
# time a node doesn't have that repo path yet, it's bootstrapped: cloned from
|
||||
# this checkout's own `origin` remote, then scripts/codex-setup.sh installs
|
||||
# the build tooling (Nix, etc.). Every run after that just `git pull`s it.
|
||||
# SSH host keys are stored as clan vars (vars/per-machine/<target>/openssh/,
|
||||
# committed and sops-encrypted) -- the script decrypts them locally and
|
||||
# copies only the two files for this target to the node's host-keys/ before
|
||||
# building. A target with no clan var is an error (generate one first with
|
||||
# scripts/secrets/sync-host-keys.sh <target>).
|
||||
#
|
||||
# --node (default: $PROXMOX_HOST, see scripts/env.sh) picks which of the two
|
||||
# LAN Proxmox nodes this runs against: production, pve1.sweet.home
|
||||
# ($PVE1_HOST, PROXMOX_HOST's own default), or the sandbox node,
|
||||
# pve-test.sweet.home ($PVE_TEST_HOST) -- pass --node "$PVE_TEST_HOST" (or
|
||||
# set PROXMOX_HOST=$PVE_TEST_HOST) to target the sandbox instead. See
|
||||
# CLAUDE.md's "Two Proxmox nodes" section: an agent session should default
|
||||
# to pve-test and only touch pve1 when the operator has explicitly said so
|
||||
# for the current task -- this script itself doesn't enforce that (its own
|
||||
# default is production, matching this repo's behavior before pve-test
|
||||
# existed), it's a policy for whoever/whatever is driving it.
|
||||
#
|
||||
# Usage:
|
||||
# scripts/proxmox/create-proxmox-resource.sh --type lxc|vm --host <name> [options]
|
||||
# scripts/proxmox/create-proxmox-resource.sh --type lxc|vm --list
|
||||
# scripts/proxmox/create-proxmox-resource.sh --modify --vmid <n> [--cores N] [--memory MB] [--grow-disk GB]
|
||||
#
|
||||
# SAFETY:
|
||||
# - The default (create) mode only ever creates a NEW resource -- it
|
||||
# refuses to run if the target VMID already exists on the node, or if
|
||||
# a VM/CT identified as --host already exists under any other VMID
|
||||
# (checked live against the node; --allow-duplicate-host overrides).
|
||||
# - --allow-duplicate-host distinguishes an exact match (same --type
|
||||
# *and* --host, e.g. re-running --type lxc --host docker while an
|
||||
# lxc-docker container already exists -- almost always a redeploy of
|
||||
# the same target to pick up a rebuilt image) from a cross-type match
|
||||
# (a different platform sharing the same host identity, e.g. a
|
||||
# proxmox-docker VM coexisting with lxc-docker). Only the exact match
|
||||
# is destroyed and replaced, after typing the hostname back to
|
||||
# confirm (outside --dry-run) -- a cross-type match is always left
|
||||
# untouched, matching-or-not.
|
||||
# - --modify only ever touches a resource you name explicitly via
|
||||
# --vmid, shows exactly what will change first, and (outside
|
||||
# --dry-run) always requires typing that VMID back to confirm before
|
||||
# anything is sent to the node. There is no bulk/implicit modify.
|
||||
# - Outside of --allow-duplicate-host's exact-match replace above,
|
||||
# neither mode can start/stop/delete a resource.
|
||||
#
|
||||
# See --help for the full option list.
|
||||
set -euo pipefail
|
||||
|
||||
repo_root="$(cd "$(dirname "$0")/../.." && pwd)"
|
||||
# shellcheck source=../env.sh
|
||||
source "${repo_root}/scripts/env.sh"
|
||||
# shellcheck source=../lib/nix-eval.sh
|
||||
source "${repo_root}/scripts/lib/nix-eval.sh"
|
||||
# shellcheck source=../lib/confirm.sh
|
||||
source "${repo_root}/scripts/lib/confirm.sh"
|
||||
# shellcheck source=../lib/sops-age.sh
|
||||
source "${repo_root}/scripts/lib/sops-age.sh"
|
||||
# shellcheck source=../lib/clan-vars.sh
|
||||
source "${repo_root}/scripts/lib/clan-vars.sh"
|
||||
|
||||
sync_keys="${repo_root}/scripts/secrets/sync-host-keys.sh"
|
||||
|
||||
usage() {
|
||||
cat <<EOF
|
||||
Usage: $0 --type lxc|vm --host <name> [options] (create)
|
||||
$0 --type lxc|vm --list (list --host values)
|
||||
$0 --modify --vmid <n> [options] (reconfigure)
|
||||
|
||||
Create mode (default):
|
||||
--type lxc|vm lxc = container, built as a CT template tarball.
|
||||
vm = VM, built as a Disko .raw disk image (UEFI/OVMF).
|
||||
--host <name> Which host identity to deploy -- matches
|
||||
config.networking.hostName (server, docker,
|
||||
nix-cache, nixos, pxe-boot, nix-minimal). Use
|
||||
--list to see what's available for --type.
|
||||
--name <name> Proxmox display name/hostname (default: --host's
|
||||
value, e.g. nix-cache -- for lxc this becomes the
|
||||
guest's real networking.hostName too, since
|
||||
proxmoxLXC.manageHostName pulls it from Proxmox's
|
||||
own container config, so it must match host.nix
|
||||
regardless of build type)
|
||||
--vmid <n> Numeric VMID (default: next free, via
|
||||
\`pvesh get /cluster/nextid\` on the node).
|
||||
Refuses to run if this ID already exists.
|
||||
--disk-size <GB> lxc only: rootfs size for \`pct create\`
|
||||
(default: \$PROXMOX_DEFAULT_LXC_DISK_GB, ${PROXMOX_DEFAULT_LXC_DISK_GB}).
|
||||
--image <path> Use this local image/tarball (uploaded to the
|
||||
node via scp) instead of checking the node /
|
||||
building one there from the flake.
|
||||
--force-rebuild Skip the "does the node already have this
|
||||
image" check -- always build fresh and
|
||||
overwrite what's there.
|
||||
--remote-repo-dir <path> Where this flake repo lives (or gets
|
||||
cloned) on the node, and is built from
|
||||
(default: \$PROXMOX_REMOTE_REPO_DIR, ${PROXMOX_REMOTE_REPO_DIR}).
|
||||
--allow-duplicate-host Required if a VM/CT identified as --host
|
||||
already exists on the node (checked live via
|
||||
qm/pct, not any file in this repo) --
|
||||
otherwise refused, since it'd share that
|
||||
host's hostName/hostId. An existing resource
|
||||
of this *same* --type (e.g. re-running --type
|
||||
lxc --host docker over an existing lxc-docker)
|
||||
is destroyed and replaced, after confirming --
|
||||
a different --type sharing the same --host
|
||||
(e.g. a proxmox-docker VM) is always left
|
||||
untouched.
|
||||
|
||||
Modify mode (reconfigure an EXISTING resource -- requires --modify):
|
||||
--modify Switch to modify mode.
|
||||
--vmid <n> Required: which existing resource to change.
|
||||
Type/VM-vs-CT is auto-detected on the node.
|
||||
--grow-disk <GB> Grow the primary disk by this many GB
|
||||
(qm/pct resize; Proxmox only supports
|
||||
growing, never shrinking, an existing disk).
|
||||
At least one of --cores / --memory / --grow-disk is required. Always
|
||||
prints the current -> new values and requires typing the VMID back to
|
||||
confirm, even outside --dry-run.
|
||||
|
||||
Shared:
|
||||
--cores <n> create: default \$PROXMOX_DEFAULT_CORES (${PROXMOX_DEFAULT_CORES}).
|
||||
modify: omit to leave unchanged.
|
||||
--memory <MB> create: default \$PROXMOX_DEFAULT_MEMORY_MB (${PROXMOX_DEFAULT_MEMORY_MB}).
|
||||
modify: omit to leave unchanged.
|
||||
--swap <MB> lxc only, create time: \`--memory\` doesn't
|
||||
touch swap -- it silently stays at Proxmox's
|
||||
own 512M default otherwise. (default: matches
|
||||
whatever --memory resolves to)
|
||||
--storage <pool> (default: \$PROXMOX_STORAGE, ${PROXMOX_STORAGE})
|
||||
--iso-storage <pool> (default: \$PROXMOX_ISO_STORAGE, ${PROXMOX_ISO_STORAGE})
|
||||
--bridge <bridge> (default: \$PROXMOX_BRIDGE, ${PROXMOX_BRIDGE})
|
||||
--node <host> Proxmox node to SSH into (default:
|
||||
\$PROXMOX_HOST, ${PROXMOX_HOST} --
|
||||
production; the sandbox node is
|
||||
\$PVE_TEST_HOST, ${PVE_TEST_HOST}).
|
||||
--dry-run Print the full plan; touch nothing
|
||||
local or remote, no prompts.
|
||||
-h, --help
|
||||
|
||||
Config for --storage/--bridge/--node/etc. lives in scripts/env.sh -- edit
|
||||
that instead of passing the same flag every time.
|
||||
EOF
|
||||
}
|
||||
|
||||
dry_run=0
|
||||
modify=0
|
||||
type=""
|
||||
host=""
|
||||
name=""
|
||||
vmid=""
|
||||
cores=""
|
||||
memory=""
|
||||
swap=""
|
||||
disk_size=""
|
||||
grow_disk=""
|
||||
image=""
|
||||
storage="$PROXMOX_STORAGE"
|
||||
iso_storage="$PROXMOX_ISO_STORAGE"
|
||||
bridge="$PROXMOX_BRIDGE"
|
||||
node="$PROXMOX_HOST"
|
||||
remote_repo_dir="$PROXMOX_REMOTE_REPO_DIR"
|
||||
do_list=0
|
||||
allow_duplicate_host=0
|
||||
force_rebuild=0
|
||||
|
||||
while [[ $# -gt 0 ]]; do
|
||||
case "$1" in
|
||||
--type) type="$2"; shift 2 ;;
|
||||
--host) host="$2"; shift 2 ;;
|
||||
--name) name="$2"; shift 2 ;;
|
||||
--vmid) vmid="$2"; shift 2 ;;
|
||||
--cores) cores="$2"; shift 2 ;;
|
||||
--memory) memory="$2"; shift 2 ;;
|
||||
--swap) swap="$2"; shift 2 ;;
|
||||
--disk-size) disk_size="$2"; shift 2 ;;
|
||||
--grow-disk) grow_disk="$2"; shift 2 ;;
|
||||
--image) image="$2"; shift 2 ;;
|
||||
--storage) storage="$2"; shift 2 ;;
|
||||
--iso-storage) iso_storage="$2"; shift 2 ;;
|
||||
--bridge) bridge="$2"; shift 2 ;;
|
||||
--node) node="$2"; shift 2 ;;
|
||||
--remote-repo-dir) remote_repo_dir="$2"; shift 2 ;;
|
||||
--list) do_list=1; shift ;;
|
||||
--allow-duplicate-host) allow_duplicate_host=1; shift ;;
|
||||
--force-rebuild) force_rebuild=1; shift ;;
|
||||
--modify) modify=1; shift ;;
|
||||
--dry-run) dry_run=1; shift ;;
|
||||
-h | --help) usage; exit 0 ;;
|
||||
*) echo "Unknown option: $1" >&2; usage >&2; exit 1 ;;
|
||||
esac
|
||||
done
|
||||
|
||||
ssh_target="${PROXMOX_SSH_USER}@${node}"
|
||||
|
||||
# Proxmox tools (pvesh, qm, pct) require root access to the cluster IPC
|
||||
# socket. When SSH-ing as a non-root user with sudo, prefix every remote
|
||||
# Proxmox command with sudo.
|
||||
sudo_prefix=""
|
||||
sudo_display=""
|
||||
if [[ "$PROXMOX_SSH_USER" != "root" ]]; then
|
||||
sudo_prefix="sudo"
|
||||
sudo_display="sudo "
|
||||
fi
|
||||
|
||||
remote() {
|
||||
if [[ "$dry_run" -eq 1 ]]; then
|
||||
echo "[dry-run] ssh ${ssh_target} -- $*"
|
||||
else
|
||||
ssh "$ssh_target" "$@"
|
||||
fi
|
||||
}
|
||||
|
||||
# ============================================================ modify mode
|
||||
cmd_modify() {
|
||||
if [[ -z "$vmid" ]]; then
|
||||
echo "ERROR: --modify requires --vmid." >&2
|
||||
exit 1
|
||||
fi
|
||||
if [[ -z "$cores" && -z "$memory" && -z "$grow_disk" ]]; then
|
||||
echo "ERROR: --modify needs at least one of --cores / --memory / --grow-disk." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "Looking up VMID ${vmid} on ${node}..."
|
||||
local kind current_cores current_memory disk_key
|
||||
if ssh "$ssh_target" "${sudo_prefix} qm status ${vmid}" >/dev/null 2>&1; then
|
||||
kind="vm"
|
||||
disk_key="scsi0"
|
||||
elif ssh "$ssh_target" "${sudo_prefix} pct status ${vmid}" >/dev/null 2>&1; then
|
||||
kind="lxc"
|
||||
disk_key="rootfs"
|
||||
else
|
||||
echo "ERROR: VMID ${vmid} doesn't exist on ${node} -- nothing to modify." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
local config_cmd="${sudo_prefix} qm config ${vmid}"
|
||||
[[ "$kind" == "lxc" ]] && config_cmd="${sudo_prefix} pct config ${vmid}"
|
||||
local current_config
|
||||
current_config="$(ssh "$ssh_target" "$config_cmd")"
|
||||
current_cores="$(echo "$current_config" | grep -oP '^cores:\s*\K\S+' || echo '?')"
|
||||
current_memory="$(echo "$current_config" | grep -oP '^memory:\s*\K\S+' || echo '?')"
|
||||
|
||||
echo
|
||||
echo "VMID ${vmid} is a ${kind} on ${node}. Planned changes:"
|
||||
[[ -n "$cores" ]] && echo " cores: ${current_cores} -> ${cores}"
|
||||
[[ -n "$memory" ]] && echo " memory: ${current_memory} MB -> ${memory} MB"
|
||||
[[ -n "$grow_disk" ]] && echo " ${disk_key}: grow by +${grow_disk}G (Proxmox can only grow, not shrink, an existing disk)"
|
||||
|
||||
if [[ "$dry_run" -eq 1 ]]; then
|
||||
echo
|
||||
echo "[dry-run] Nothing was changed."
|
||||
return
|
||||
fi
|
||||
|
||||
echo
|
||||
if ! confirm_typed "$vmid" "Type the VMID (${vmid}) to confirm these changes: "; then
|
||||
echo "Cancelled -- input didn't match ${vmid}."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
local set_cmd="${sudo_prefix} qm set"
|
||||
local resize_cmd="${sudo_prefix} qm resize"
|
||||
[[ "$kind" == "lxc" ]] && set_cmd="${sudo_prefix} pct set" && resize_cmd="${sudo_prefix} pct resize"
|
||||
|
||||
if [[ -n "$cores" || -n "$memory" ]]; then
|
||||
local args=""
|
||||
[[ -n "$cores" ]] && args="${args} --cores ${cores}"
|
||||
[[ -n "$memory" ]] && args="${args} --memory ${memory}"
|
||||
remote "${set_cmd} ${vmid}${args}"
|
||||
fi
|
||||
if [[ -n "$grow_disk" ]]; then
|
||||
remote "${resize_cmd} ${vmid} ${disk_key} +${grow_disk}G"
|
||||
fi
|
||||
|
||||
echo
|
||||
echo "Done. VMID ${vmid} updated."
|
||||
}
|
||||
|
||||
if [[ "$modify" -eq 1 ]]; then
|
||||
cmd_modify
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# ============================================================= create mode
|
||||
if [[ "$type" != "lxc" && "$type" != "vm" ]]; then
|
||||
echo "ERROR: --type must be 'lxc' or 'vm'." >&2
|
||||
usage >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
platform_prefix="lxc"
|
||||
[[ "$type" == "vm" ]] && platform_prefix="proxmox"
|
||||
[[ -z "$cores" ]] && cores="$PROXMOX_DEFAULT_CORES"
|
||||
[[ -z "$memory" ]] && memory="$PROXMOX_DEFAULT_MEMORY_MB"
|
||||
|
||||
if [[ "$type" == "vm" && -n "$disk_size" ]]; then
|
||||
echo "WARNING: --disk-size is LXC-only for create mode and is ignored for VMs." >&2
|
||||
echo " VM disk size comes from proxmoxImageSize in variables.nix (currently ${disk_size}G was requested)." >&2
|
||||
echo " To expand after creation, use: --modify --vmid <n> --grow-disk <GB>" >&2
|
||||
fi
|
||||
|
||||
# --- discover / resolve the flake target from --host --------------------
|
||||
# Emits "<target>\t<hostName>" pairs for every ${platform_prefix}-* flake
|
||||
# target -- the one source both --list and the --host lookup below read
|
||||
# from, so they can never see a different set of targets from each other.
|
||||
targets_for_platform() {
|
||||
local target
|
||||
for target in $(list_flake_targets "$repo_root" 2>/dev/null | grep -- "^${platform_prefix}-"); do
|
||||
printf '%s\t%s\n' "$target" "$(flake_target_hostname "$repo_root" "$target")"
|
||||
done
|
||||
}
|
||||
|
||||
list_hosts() {
|
||||
local target hostname
|
||||
while IFS=$'\t' read -r target hostname; do
|
||||
printf ' %-12s -> %s\n' "$hostname" "$target"
|
||||
done < <(targets_for_platform)
|
||||
}
|
||||
|
||||
if [[ "$do_list" -eq 1 ]]; then
|
||||
echo "Available --host values for --type ${type}:"
|
||||
list_hosts
|
||||
exit 0
|
||||
fi
|
||||
|
||||
if [[ -z "$host" ]]; then
|
||||
echo "ERROR: --host is required (or use --list to see options)." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
flake_target=""
|
||||
while IFS=$'\t' read -r target hostname; do
|
||||
if [[ "$hostname" == "$host" ]]; then
|
||||
flake_target="$target"
|
||||
break
|
||||
fi
|
||||
done < <(targets_for_platform)
|
||||
|
||||
if [[ -z "$flake_target" ]]; then
|
||||
echo "ERROR: no ${platform_prefix}-* target has hostName '${host}'." >&2
|
||||
echo "Available:" >&2
|
||||
list_hosts >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# The container/VM's real identity is --host (e.g. "nix-cache"), validated
|
||||
# above against config.networking.hostName -- not the flake target name
|
||||
# (e.g. "lxc-nix-cache"), which is build-type-specific and only exists to
|
||||
# pick which platform variant to build. Defaulting --name to the flake
|
||||
# target would make lxc's --hostname (which proxmoxLXC.manageHostName
|
||||
# feeds straight into the guest's real hostname) disagree with host.nix.
|
||||
[[ -z "$name" ]] && name="$host"
|
||||
|
||||
# For VM builds: the diskoImagesScript (run via QEMU on the node) writes the
|
||||
# raw disk image as <hostname>.raw into the CWD it was called from (the remote
|
||||
# repo dir), not to /var/lib/vz/import/ or anywhere else. Import directly from
|
||||
# there -- no intermediate mv that can fail crossing filesystem boundaries or
|
||||
# leave a stale file on error.
|
||||
vm_built_raw=""
|
||||
[[ "$type" == "vm" ]] && vm_built_raw="${remote_repo_dir}/${host}.raw"
|
||||
|
||||
# --- refuse to duplicate a host that's already live on the node ---------
|
||||
# Queries the node itself (qm/pct's own name/hostname config), not any
|
||||
# static list in this repo -- a file can't track whether a resource still
|
||||
# actually exists, and this used to be checked against variables.nix's
|
||||
# deployedTargets, which drifted stale (it kept naming a VM as "the real
|
||||
# deployment" well after that VM had been destroyed, blocking its own
|
||||
# redeploy) until that list was dropped in favour of this live check. This
|
||||
# only catches guests identified with the default --name (== --host, what
|
||||
# this script itself always uses unless --name is overridden) -- a guest
|
||||
# manually renamed on the node afterwards wouldn't match, but nothing here
|
||||
# creates guests that way.
|
||||
if [[ "$dry_run" -eq 1 ]]; then
|
||||
echo
|
||||
echo "[dry-run] would check ${node} for an existing VM/CT identified as '${host}'"
|
||||
if [[ "$allow_duplicate_host" -eq 1 ]]; then
|
||||
echo "[dry-run] --allow-duplicate-host: an existing ${type} named '${host}' would be" \
|
||||
"destroyed and replaced; a different-type match would be left untouched"
|
||||
fi
|
||||
else
|
||||
echo
|
||||
echo "==> Checking ${node} for an existing VM/CT identified as '${host}'..."
|
||||
ssh_check_status=0
|
||||
existing="$(ssh "$ssh_target" bash -s -- "$host" "$sudo_prefix" <<'REMOTE_SCRIPT'
|
||||
target="$1"
|
||||
sudo_pfx="$2"
|
||||
for id in $($sudo_pfx qm list 2>/dev/null | awk 'NR>1{print $1}'); do
|
||||
n="$($sudo_pfx qm config "$id" 2>/dev/null | grep -oP '^name:\s*\K\S+' || true)"
|
||||
[[ "$n" == "$target" ]] && echo "vm ${id} ${n}"
|
||||
done
|
||||
for id in $($sudo_pfx pct list 2>/dev/null | awk 'NR>1{print $1}'); do
|
||||
n="$($sudo_pfx pct config "$id" 2>/dev/null | grep -oP '^hostname:\s*\K\S+' || true)"
|
||||
[[ "$n" == "$target" ]] && echo "lxc ${id} ${n}"
|
||||
done
|
||||
exit 0
|
||||
REMOTE_SCRIPT
|
||||
)" || ssh_check_status=$?
|
||||
if [[ "$ssh_check_status" -ne 0 ]]; then
|
||||
echo "ERROR: couldn't reach ${node} (ssh exited ${ssh_check_status}) to check for an" >&2
|
||||
echo "existing '${host}' resource -- refusing to guess. Fix connectivity and retry," >&2
|
||||
echo "or pass --allow-duplicate-host if you're sure none exists (this skips the" >&2
|
||||
echo "check entirely)." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Split into "exact" (same resource kind as --type -- i.e. literally this
|
||||
# same host+platform combo already exists, almost always a redeploy of
|
||||
# the same target to test a rebuilt image) vs "cross-type" (a different
|
||||
# platform sharing this host identity, e.g. a stopped proxmox-docker VM
|
||||
# coexisting with an lxc-docker container -- a deliberate, valid setup
|
||||
# this script has never managed and still won't). Read via a herestring
|
||||
# (not a pipe) so the appends below survive outside the loop.
|
||||
this_kind="$type"
|
||||
exact_matches=""
|
||||
cross_matches=""
|
||||
if [[ -n "$existing" ]]; then
|
||||
while read -r kind id n; do
|
||||
[[ -z "$kind" ]] && continue
|
||||
if [[ "$kind" == "$this_kind" ]]; then
|
||||
exact_matches+="${kind} ${id} ${n}"$'\n'
|
||||
else
|
||||
cross_matches+="${kind} ${id} ${n}"$'\n'
|
||||
fi
|
||||
done <<<"$existing"
|
||||
fi
|
||||
|
||||
if [[ -n "$exact_matches" && "$allow_duplicate_host" -ne 1 ]]; then
|
||||
echo "ERROR: '${host}' already exists on ${node} as this same resource type:" >&2
|
||||
echo "$exact_matches" | while read -r kind id n; do
|
||||
[[ -z "$kind" ]] && continue
|
||||
echo " - ${kind} VMID ${id} (${n})" >&2
|
||||
done
|
||||
echo "Refusing to create a second ${this_kind} sharing this identity. Pass" >&2
|
||||
echo "--allow-duplicate-host to destroy it and create a fresh one in its place" >&2
|
||||
echo "(after confirming), or use --modify to reconfigure the existing one instead." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if [[ -n "$cross_matches" && "$allow_duplicate_host" -ne 1 ]]; then
|
||||
echo "ERROR: '${host}' already exists on ${node} as a different resource type:" >&2
|
||||
echo "$cross_matches" | while read -r kind id n; do
|
||||
[[ -z "$kind" ]] && continue
|
||||
echo " - ${kind} VMID ${id} (${n})" >&2
|
||||
done
|
||||
echo "Refusing to create a second resource sharing this identity. Pass" >&2
|
||||
echo "--allow-duplicate-host to create one anyway (it gets its own distinct" >&2
|
||||
echo "sops key and VMID -- the existing resource above is left untouched)," >&2
|
||||
echo "or use --modify to reconfigure the existing one instead." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if [[ -n "$cross_matches" ]]; then
|
||||
echo "--allow-duplicate-host: '${host}' also exists on ${node} as a different resource" \
|
||||
"type -- leaving it untouched:"
|
||||
echo "$cross_matches" | while read -r kind id n; do
|
||||
[[ -z "$kind" ]] && continue
|
||||
echo " - ${kind} VMID ${id} (${n})"
|
||||
done
|
||||
fi
|
||||
|
||||
if [[ -n "$exact_matches" ]]; then
|
||||
echo "--allow-duplicate-host: '${host}' already exists on ${node} as this same resource" \
|
||||
"type -- it will be destroyed and replaced:"
|
||||
echo "$exact_matches" | while read -r kind id n; do
|
||||
[[ -z "$kind" ]] && continue
|
||||
echo " - ${kind} VMID ${id} (${n})"
|
||||
done
|
||||
echo
|
||||
if ! confirm_typed "$host" "Type the hostname (${host}) to confirm destroying the above and replacing it: "; then
|
||||
echo "Cancelled -- input didn't match ${host}." >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "$exact_matches" | while read -r kind id n; do
|
||||
[[ -z "$kind" ]] && continue
|
||||
echo "==> Destroying ${kind} VMID ${id} (${n})..."
|
||||
if [[ "$kind" == "vm" ]]; then
|
||||
# qm destroy has no --force to stop-then-destroy in one call (pct's
|
||||
# does) -- stop explicitly first if it's running.
|
||||
if ssh "$ssh_target" "${sudo_prefix} qm status ${id}" 2>/dev/null | grep -q running; then
|
||||
ssh "$ssh_target" "${sudo_prefix} qm stop ${id}"
|
||||
fi
|
||||
ssh "$ssh_target" "${sudo_prefix} qm destroy ${id} --purge 1"
|
||||
else
|
||||
ssh "$ssh_target" "${sudo_prefix} pct destroy ${id} --force 1 --purge 1"
|
||||
fi
|
||||
done
|
||||
fi
|
||||
fi
|
||||
|
||||
echo "Target: ${flake_target} (host=${host}, type=${type}) -> Proxmox resource '${name}'"
|
||||
|
||||
# Decide on nix-cache once, here -- this is the earliest point that needs
|
||||
# it (sync-host-keys.sh below needs nix-shell packages regardless of
|
||||
# whether an image ends up getting built later), and the decision is
|
||||
# exported so that subprocess -- and this script's own later build step,
|
||||
# if it gets there -- both reuse it instead of probing again.
|
||||
nix_extra_opts
|
||||
|
||||
# --- make sure this target has a registered host key --------------------
|
||||
# sync-host-keys.sh is idempotent and generates the key (via clan vars) if
|
||||
# no key exists yet -- the old inline prepare-host-key.sh call is gone.
|
||||
echo
|
||||
echo "==> Ensuring host key exists and is registered..."
|
||||
sync_args=("$flake_target")
|
||||
[[ "$dry_run" -eq 1 ]] && sync_args+=(--dry-run)
|
||||
bash "$sync_keys" "${sync_args[@]}"
|
||||
|
||||
# If sync-host-keys.sh changed .sops.yaml, secrets/, or vars/per-machine/,
|
||||
# those changes must be committed and pushed before the remote `git pull`
|
||||
# below picks them up -- the PVE node builds from whatever HEAD is checked
|
||||
# out there, not the local working tree. Uncommitted clan vars or sops
|
||||
# recipients mean the image builds fine but the host cannot decrypt its
|
||||
# secrets on first boot. Block until the operator confirms they've pushed.
|
||||
if [[ "$dry_run" -eq 0 ]]; then
|
||||
_dirty="$(git -C "$repo_root" status --porcelain -- .sops.yaml secrets/ vars/per-machine/ 2>/dev/null || true)"
|
||||
if [[ -n "$_dirty" ]]; then
|
||||
echo
|
||||
echo "==> COMMIT + PUSH REQUIRED before the remote build can succeed:"
|
||||
echo " Uncommitted changes in .sops.yaml, secrets/, or vars/per-machine/."
|
||||
echo " The PVE node builds from the git-tracked flake, so these changes"
|
||||
echo " must be committed and pushed first -- otherwise the image build will"
|
||||
echo " succeed but the host cannot decrypt its secrets on first boot."
|
||||
echo
|
||||
git -C "$repo_root" status --short -- .sops.yaml secrets/ vars/per-machine/ || true
|
||||
echo
|
||||
read -rp " Commit and push those changes, then press Enter to continue (Ctrl-C to abort): "
|
||||
fi
|
||||
unset _dirty
|
||||
fi
|
||||
|
||||
# --- VMID: pick one, and refuse to touch anything that already exists ---
|
||||
echo
|
||||
if [[ -z "$vmid" ]]; then
|
||||
if [[ "$dry_run" -eq 1 ]]; then
|
||||
vmid="<next-free-vmid>"
|
||||
echo "[dry-run] would ask ${node} for the next free VMID (pvesh get /cluster/nextid)"
|
||||
else
|
||||
vmid="$(ssh "$ssh_target" "${sudo_prefix} pvesh get /cluster/nextid" | tr -d '[:space:]')"
|
||||
echo "Auto-assigned VMID: ${vmid}"
|
||||
fi
|
||||
else
|
||||
echo "Requested VMID: ${vmid}"
|
||||
fi
|
||||
|
||||
if [[ "$dry_run" -eq 0 ]]; then
|
||||
# qm/pct status exits non-zero (and prints "does not exist") for a free
|
||||
# ID on that resource type -- but a VMID could exist as the OTHER
|
||||
# resource type (e.g. requested a CT id that's actually a VM), so check
|
||||
# both. Any success here means something is already using this ID --
|
||||
# refuse to go anywhere near it. (Reconfiguring an existing resource is
|
||||
# --modify's job, not this one's.)
|
||||
if ssh "$ssh_target" "${sudo_prefix} qm status ${vmid}" >/dev/null 2>&1 \
|
||||
|| ssh "$ssh_target" "${sudo_prefix} pct status ${vmid}" >/dev/null 2>&1; then
|
||||
echo "ERROR: VMID ${vmid} already exists on ${node}. Refusing to touch an" >&2
|
||||
echo "existing resource here -- use --modify to reconfigure it, pick a" >&2
|
||||
echo "different --vmid, or omit it to auto-assign." >&2
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
|
||||
# --- resolve the remote path -- fixed naming (not the nix store's own
|
||||
# derivation-hash-based filename), so a later run can check for it by name.
|
||||
# lxc uploads as a CT *template* (Proxmox's "vztmpl" content type, under
|
||||
# iso_storage) -- config.system.build.tarball is a plain rootfs tarball,
|
||||
# not a vzdump backup archive, so it's created with `pct create ... vztmpl`,
|
||||
# not restored with `pct restore` (that expects backup-archive metadata
|
||||
# this tarball doesn't have, and fails with "archive contains no
|
||||
# configuration file").
|
||||
remote_dir="/var/lib/vz/import"
|
||||
remote_filename="${flake_target}.raw"
|
||||
if [[ "$type" == "lxc" ]]; then
|
||||
remote_dir="/var/lib/vz/template/cache"
|
||||
remote_filename="${flake_target}.tar.xz"
|
||||
fi
|
||||
remote_path="${remote_dir}/${remote_filename}"
|
||||
|
||||
# --- ensure the flake repo (+ tooling) exists on the node, and is current --
|
||||
# Bootstraps once (git clone from this checkout's own `origin`, then
|
||||
# scripts/codex-setup.sh installs Nix + friends) if ${remote_repo_dir}
|
||||
# doesn't exist yet on the node; otherwise just `git pull`s it, so the image
|
||||
# built there reflects what's actually committed and pushed. Only called
|
||||
# right before an actual remote build below -- reusing an image already on
|
||||
# the node, or an explicit --image, never touch the node's checkout at all.
|
||||
ensure_remote_repo() {
|
||||
echo
|
||||
echo "==> Ensuring ${remote_repo_dir} exists and is current on ${node}..."
|
||||
if [[ "$dry_run" -eq 1 ]]; then
|
||||
echo "[dry-run] would ensure ${remote_repo_dir} exists on ${node} (clone if missing, git pull if present), and would verify/bootstrap build tooling there (scripts/codex-setup.sh) if \`nix\` isn't already on PATH -- and if that bootstrap actually ran, would also configure ${node} as a nix-cache client (scripts/proxmox/configure-nix-cache-client.sh)"
|
||||
return
|
||||
fi
|
||||
|
||||
if ssh "$ssh_target" "test -d '${remote_repo_dir}/.git'"; then
|
||||
echo "Repo present -- pulling latest..."
|
||||
ssh "$ssh_target" "cd '${remote_repo_dir}' && git pull --ff-only"
|
||||
else
|
||||
local origin_url
|
||||
origin_url="$(git -C "$repo_root" remote get-url origin 2>/dev/null || true)"
|
||||
if [[ -z "$origin_url" ]]; then
|
||||
echo "ERROR: ${remote_repo_dir} doesn't exist on ${node}, and this checkout has no" >&2
|
||||
echo "'origin' remote to clone from. Set one (git remote add origin <url>) or create" >&2
|
||||
echo "${remote_repo_dir} on ${node} yourself (e.g. git clone), then re-run." >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "Not present -- cloning from ${origin_url}..."
|
||||
ssh "$ssh_target" "git clone '${origin_url}' '${remote_repo_dir}'"
|
||||
fi
|
||||
|
||||
# Trivial check, run every time (not just right after a fresh clone) --
|
||||
# confirmed live: a first bootstrap can clone the repo successfully and
|
||||
# still leave the node without a working `nix` (e.g. the node had no
|
||||
# `sudo`, which the Nix installer's root path depends on -- see the fix
|
||||
# in scripts/codex-setup.sh), and a later run with the repo already
|
||||
# present would otherwise never retry it. Sources
|
||||
# scripts/lib/nix-bootstrap.sh's ensure_nix_profile first -- a
|
||||
# single-user Nix install typically only gets sourced into login shells,
|
||||
# and ssh's non-interactive command execution is neither, so a
|
||||
# freshly-installed `nix` still wouldn't be on PATH here without it.
|
||||
#
|
||||
# Just `nix` today -- the only thing the remote build commands below
|
||||
# actually invoke -- but a list (not a single hardcoded check) so a
|
||||
# future remote step needing another tool can add itself here instead of
|
||||
# growing a parallel check.
|
||||
local remote_required_cmds=(nix)
|
||||
local tooling_check_cmd="cd '${remote_repo_dir}' && . scripts/lib/nix-bootstrap.sh && ensure_nix_profile"
|
||||
local cmd
|
||||
for cmd in "${remote_required_cmds[@]}"; do
|
||||
tooling_check_cmd="${tooling_check_cmd} && command -v ${cmd}"
|
||||
done
|
||||
|
||||
if ssh "$ssh_target" "$tooling_check_cmd" >/dev/null 2>&1; then
|
||||
echo "Build tooling already present on ${node}."
|
||||
else
|
||||
echo "==> Bootstrapping build tooling on ${node} (scripts/codex-setup.sh)..."
|
||||
ssh "$ssh_target" "cd '${remote_repo_dir}' && bash scripts/codex-setup.sh"
|
||||
|
||||
# Only on this first-time bootstrap, not every run -- a node that
|
||||
# already has tooling either already went through this once, or had
|
||||
# it configured some other way, and re-running is harmless but
|
||||
# pointless. Non-fatal: this only makes the node's own builds faster
|
||||
# (substitute from nix-cache instead of building from source) and
|
||||
# offloadable to it as a remote builder -- worth trying, not worth
|
||||
# aborting the image build over if nix-cache happens to be down right
|
||||
# now. Needs ensure_nix_profile first, same as the tooling_check_cmd
|
||||
# above -- ssh's non-interactive command execution won't have picked
|
||||
# up a freshly single-user-installed `nix` otherwise.
|
||||
echo "==> Configuring ${node} as a nix-cache substituter/remote-builder client..."
|
||||
if ! ssh "$ssh_target" "cd '${remote_repo_dir}' && . scripts/lib/nix-bootstrap.sh && ensure_nix_profile && bash scripts/proxmox/configure-nix-cache-client.sh"; then
|
||||
echo "WARNING: configure-nix-cache-client.sh failed on ${node} -- continuing without it (${node} will build from source / against cache.nixos.org only)." >&2
|
||||
fi
|
||||
fi
|
||||
}
|
||||
|
||||
# --- sync host key to the node ---------------------------------------------
|
||||
# SSH host keys are stored as clan vars (vars/per-machine/<target>/openssh/).
|
||||
# Decrypt locally and scp just the two files for this target to the node's
|
||||
# host-keys/ directory, where the remote build script picks them up via
|
||||
# NIXOS_HOST_KEYS_DIR (LXC) or --pre-format-files (VM). A target with no
|
||||
# clan var is an error -- generate one first with sync-host-keys.sh.
|
||||
sync_remote_host_keys() {
|
||||
echo
|
||||
echo "==> Syncing host key for ${flake_target} to ${node}..."
|
||||
if [[ "$dry_run" -eq 1 ]]; then
|
||||
echo "[dry-run] would decrypt clan SSH key for ${flake_target} and copy to ${ssh_target}:${remote_repo_dir}/host-keys/"
|
||||
return
|
||||
fi
|
||||
if ! clan_ssh_key_exists "$flake_target" "$repo_root"; then
|
||||
echo "ERROR: no clan SSH key found for ${flake_target}" >&2
|
||||
echo " (expected: ${repo_root}/vars/per-machine/${flake_target}/openssh/ssh_host_ed25519_key/secret)" >&2
|
||||
echo " Generate one first: bash scripts/secrets/sync-host-keys.sh ${flake_target}" >&2
|
||||
exit 1
|
||||
fi
|
||||
local tmpdir
|
||||
tmpdir="$(mktemp -d)"
|
||||
# shellcheck disable=SC2064
|
||||
trap "rm -rf '${tmpdir}'" RETURN
|
||||
echo " Decrypting clan SSH key for ${flake_target}..."
|
||||
clan_decrypt_ssh_key "$flake_target" "$repo_root" "$tmpdir"
|
||||
ssh "$ssh_target" "mkdir -p '${remote_repo_dir}/host-keys'"
|
||||
scp -p "${tmpdir}/${flake_target}_ssh_host_ed25519_key" \
|
||||
"${tmpdir}/${flake_target}_ssh_host_ed25519_key.pub" \
|
||||
"${ssh_target}:${remote_repo_dir}/host-keys/"
|
||||
}
|
||||
|
||||
# --- build (or reuse an image already on the node) ------------------------
|
||||
echo
|
||||
local_image=""
|
||||
image_already_remote=0
|
||||
|
||||
if [[ -n "$image" ]]; then
|
||||
[[ -f "$image" ]] || { echo "ERROR: --image '${image}' not found." >&2; exit 1; }
|
||||
local_image="$image"
|
||||
echo "Using provided image: ${local_image}"
|
||||
elif [[ "$force_rebuild" -eq 1 ]]; then
|
||||
echo "--force-rebuild: skipping the existing-image check on ${node}."
|
||||
else
|
||||
# VMs: check for the raw image in the remote repo dir (where disko writes it).
|
||||
# LXC: check for the tarball in iso_storage (where the LXC build stages it).
|
||||
_check_path="$remote_path"
|
||||
[[ "$type" == "vm" ]] && _check_path="$vm_built_raw"
|
||||
echo "==> Checking whether ${node} already has ${_check_path}..."
|
||||
if [[ "$dry_run" -eq 1 ]]; then
|
||||
echo "[dry-run] would check: ssh ${ssh_target} -- test -f ${_check_path}"
|
||||
elif ssh "$ssh_target" "test -f '${_check_path}'" 2>/dev/null; then
|
||||
echo "Found it -- reusing, skipping build (use --force-rebuild to override)."
|
||||
image_already_remote=1
|
||||
else
|
||||
echo "Not found -- will build."
|
||||
fi
|
||||
fi
|
||||
|
||||
if [[ "$image_already_remote" -eq 0 && -z "$local_image" ]]; then
|
||||
ensure_remote_repo
|
||||
sync_remote_host_keys
|
||||
|
||||
# Relayed into the remote build below exactly as decided by the local
|
||||
# nix_extra_opts call earlier in this script -- that decision (whether
|
||||
# nix-cache is reachable) is made once, locally, same as it always has
|
||||
# been; only *where* the resulting "${NIX_OPTS[@]}" gets used as a `nix
|
||||
# build` flag moves to the node. NIX_EXTRA_OPTS is already a %q-quoted
|
||||
# string built for exactly this eval-based reconstruction (see env.sh).
|
||||
nix_opts_display=""
|
||||
if [[ ${#NIX_OPTS[@]} -gt 0 ]]; then
|
||||
printf -v nix_opts_display '%q ' "${NIX_OPTS[@]}"
|
||||
nix_opts_display=" ${nix_opts_display% }"
|
||||
fi
|
||||
|
||||
if [[ "$type" == "lxc" ]]; then
|
||||
if [[ "$dry_run" -eq 1 ]]; then
|
||||
echo "[dry-run] would build on ${node}: NIXOS_HOST_KEYS_DIR=\$(pwd)/host-keys nix build --impure \\"
|
||||
echo "[dry-run] --no-use-registries --no-accept-flake-config${nix_opts_display} \\"
|
||||
echo "[dry-run] .#nixosConfigurations.${flake_target}.config.system.build.tarball"
|
||||
echo "[dry-run] would stage the result at ${remote_path}"
|
||||
local_image="<built-tarball>"
|
||||
else
|
||||
echo "==> Building LXC tarball for ${flake_target} on ${node}..."
|
||||
# Built as a single already-%q-quoted command string, not separate ssh
|
||||
# argv elements -- ssh joins remote command args with plain spaces and
|
||||
# hands the result to the remote shell to re-split, which would
|
||||
# otherwise scatter NIX_EXTRA_OPTS (itself several space-separated,
|
||||
# %q-quoted tokens) across the wrong positional parameters below.
|
||||
printf -v remote_cmd 'bash -s -- %q %q %q %q %q %q' \
|
||||
"$remote_repo_dir" "$flake_target" "$remote_dir" "$remote_filename" "$NIX_EXTRA_OPTS" "$sudo_prefix"
|
||||
ssh "$ssh_target" "$remote_cmd" <<'REMOTE_SCRIPT'
|
||||
set -euo pipefail
|
||||
repo_dir="$1"; target="$2"; dest_dir="$3"; dest_name="$4"; nix_extra_opts_str="$5"; sudo_pfx="$6"
|
||||
declare -a NIX_OPTS=()
|
||||
[[ -n "$nix_extra_opts_str" ]] && eval "NIX_OPTS=(${nix_extra_opts_str})"
|
||||
cd "$repo_dir"
|
||||
# A single-user Nix install only gets sourced into login shells; this ssh
|
||||
# session is neither, so `nix` wouldn't otherwise be on PATH here even
|
||||
# right after a successful install.
|
||||
. scripts/lib/nix-bootstrap.sh
|
||||
ensure_nix_profile
|
||||
if [[ ! -f "host-keys/${target}_ssh_host_ed25519_key" ]]; then
|
||||
echo "ERROR: host-keys/${target}_ssh_host_ed25519_key not found in ${repo_dir}." >&2
|
||||
echo "Generate the key locally (scripts/secrets/sync-host-keys.sh ${target})" >&2
|
||||
echo "and ensure it was synced here before starting the build." >&2
|
||||
exit 1
|
||||
fi
|
||||
NIXOS_HOST_KEYS_DIR="$(pwd)/host-keys" nix build --impure \
|
||||
--no-use-registries --no-accept-flake-config "${NIX_OPTS[@]}" \
|
||||
".#nixosConfigurations.${target}.config.system.build.tarball" \
|
||||
--out-link "result-${target}"
|
||||
built="$(find "result-${target}/tarball" -maxdepth 1 -type f | head -1)"
|
||||
if [[ -z "$built" ]]; then
|
||||
echo "ERROR: no tarball found under result-${target}/tarball after build." >&2
|
||||
exit 1
|
||||
fi
|
||||
$sudo_pfx mkdir -p "$dest_dir"
|
||||
$sudo_pfx cp "$built" "${dest_dir}/${dest_name}"
|
||||
echo "Built and staged: ${dest_dir}/${dest_name}"
|
||||
REMOTE_SCRIPT
|
||||
local_image="$remote_path"
|
||||
echo "Built on ${node}: ${remote_path}"
|
||||
fi
|
||||
else
|
||||
if [[ "$dry_run" -eq 1 ]]; then
|
||||
echo "[dry-run] would build on ${node}: NIXOS_HOST_KEYS_DIR=\$(pwd)/host-keys nix build --impure --no-use-registries --no-accept-flake-config${nix_opts_display} \\"
|
||||
echo "[dry-run] .#nixosConfigurations.${flake_target}.config.system.build.diskoImagesScript"
|
||||
echo "[dry-run] would run: ${sudo_display}./result-${flake_target} --build-memory 2048"
|
||||
echo "[dry-run] image will be at ${vm_built_raw} (imported from there; no mv to /var/lib/vz/import/)"
|
||||
local_image="<built-image>.raw"
|
||||
else
|
||||
echo "==> Building Disko image for ${flake_target} on ${node}..."
|
||||
# See the LXC branch above for why this is one %q-quoted command
|
||||
# string rather than separate ssh argv elements.
|
||||
# $7 = image_name (hostname, the diskoImagesScript's own output filename).
|
||||
#
|
||||
# NIXOS_HOST_KEYS_DIR + --impure: modules/platforms/proxmox.nix reads
|
||||
# this env var at eval time (like lxc.nix) to embed the clan SSH host
|
||||
# key in environment.etc. nixos-install's own activation then places the
|
||||
# key on the target disk, so sshd-keygen finds it already present and
|
||||
# skips generation. --pre-format-files put the key on the QEMU builder
|
||||
# VM's rootfs (not the target disk), so sshd-keygen regenerated a fresh
|
||||
# key -- one not registered in .sops.yaml -- and sops could never decrypt.
|
||||
printf -v remote_cmd 'bash -s -- %q %q %q %q %q %q %q' \
|
||||
"$remote_repo_dir" "$flake_target" "$remote_dir" "$remote_filename" "$NIX_EXTRA_OPTS" "$sudo_prefix" "$host"
|
||||
ssh "$ssh_target" "$remote_cmd" <<'REMOTE_SCRIPT'
|
||||
set -euo pipefail
|
||||
repo_dir="$1"; target="$2"; dest_dir="$3"; dest_name="$4"; nix_extra_opts_str="$5"; sudo_pfx="$6"; image_name="$7"
|
||||
declare -a NIX_OPTS=()
|
||||
[[ -n "$nix_extra_opts_str" ]] && eval "NIX_OPTS=(${nix_extra_opts_str})"
|
||||
cd "$repo_dir"
|
||||
. scripts/lib/nix-bootstrap.sh
|
||||
ensure_nix_profile
|
||||
if [[ ! -f "host-keys/${target}_ssh_host_ed25519_key" ]]; then
|
||||
echo "ERROR: host-keys/${target}_ssh_host_ed25519_key not found in ${repo_dir}." >&2
|
||||
echo "Generate the key locally (scripts/secrets/sync-host-keys.sh ${target})" >&2
|
||||
echo "and ensure it was synced here before starting the build." >&2
|
||||
exit 1
|
||||
fi
|
||||
# Build diskoImagesScript with NIXOS_HOST_KEYS_DIR so proxmox.nix embeds the
|
||||
# clan SSH key in environment.etc (same as lxc.nix). This causes nixos-install
|
||||
# to place the key on the target disk, so sshd-keygen finds it and skips
|
||||
# generation -- the disk image boots with the registered key, sops decrypts.
|
||||
NIXOS_HOST_KEYS_DIR="$(pwd)/host-keys" nix build --impure \
|
||||
--no-use-registries --no-accept-flake-config "${NIX_OPTS[@]}" \
|
||||
".#nixosConfigurations.${target}.config.system.build.diskoImagesScript" \
|
||||
--out-link "result-${target}"
|
||||
# Remove any stale .raw from a previous failed build so the post-build check
|
||||
# below is unambiguous (diskoImagesScript writes to CWD as ${image_name}.raw).
|
||||
$sudo_pfx rm -f "${image_name}.raw" 2>/dev/null || true
|
||||
$sudo_pfx "./result-${target}" --build-memory 2048
|
||||
if [[ ! -f "${image_name}.raw" ]]; then
|
||||
echo "ERROR: ${image_name}.raw not found in ${repo_dir} after build -- disko/QEMU may have failed." >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "Built image: ${repo_dir}/${image_name}.raw"
|
||||
REMOTE_SCRIPT
|
||||
local_image="$vm_built_raw"
|
||||
echo "Built on ${node}: ${vm_built_raw}"
|
||||
fi
|
||||
fi
|
||||
fi
|
||||
|
||||
# --- upload -- only for an explicit --image; a build above stages its
|
||||
# result directly at ${remote_path} on the node already, and reusing an
|
||||
# image already on the node needs nothing transferred either. ------------
|
||||
echo
|
||||
if [[ -n "$image" ]]; then
|
||||
if [[ "$dry_run" -eq 1 ]]; then
|
||||
echo "[dry-run] would upload: scp ${local_image} ${ssh_target}:${remote_path}"
|
||||
else
|
||||
echo "==> Uploading to ${node}:${remote_path}..."
|
||||
ssh "$ssh_target" "mkdir -p ${remote_dir}"
|
||||
scp "$local_image" "${ssh_target}:${remote_path}"
|
||||
fi
|
||||
fi
|
||||
|
||||
# --- create -----------------------------------------------------------------
|
||||
echo
|
||||
if [[ "$type" == "lxc" ]]; then
|
||||
echo "==> Creating LXC container ${vmid} (${name})..."
|
||||
local_disk_size="${disk_size:-$PROXMOX_DEFAULT_LXC_DISK_GB}"
|
||||
# --memory doesn't touch swap -- it silently stays at Proxmox's own
|
||||
# 512M default otherwise (confirmed live: --memory 2048 left swap at
|
||||
# 512). Default to matching whatever --memory resolved to above.
|
||||
local_swap="${swap:-$memory}"
|
||||
# --unprivileged: read back from modules/platforms/lxc.nix's own
|
||||
# proxmoxLXC.privileged (via flake_target_lxc_privileged) rather than
|
||||
# hardcoded. lxc.nix derives this automatically: any lxc-* host whose
|
||||
# config.fileSystems has an NFS entry gets privileged=true, because the
|
||||
# kernel's NFS client (FS_USERNS_MOUNT not set) rejects NFS mounts from
|
||||
# inside any non-init user namespace -- exactly what an unprivileged
|
||||
# container's UID-mapped root lives in -- with EPERM at the VFS layer,
|
||||
# regardless of AppArmor (see lxc.nix's own comment). The NixOS config
|
||||
# bakes in cgroup/capability/mount expectations matching whichever value
|
||||
# it was built with, so this must stay in sync -- `pct create`'s CLI
|
||||
# default is privileged (unlike the web UI, which defaults the other
|
||||
# way), so leaving it unset would create a privileged container running
|
||||
# a NixOS config that assumes unprivileged, a real mismatch.
|
||||
privileged_eval="$(flake_target_lxc_privileged "$repo_root" "$flake_target")"
|
||||
unprivileged_flag=1
|
||||
[[ "$privileged_eval" == "true" ]] && unprivileged_flag=0
|
||||
#
|
||||
# --features nesting=1,keyctl=1: required for a modern (v247+) systemd
|
||||
# guest to actually boot unprivileged -- confirmed live: without this,
|
||||
# AppArmor denies the nested user namespaces and credential mounts
|
||||
# systemd routinely uses (even plain getty units), and every getty
|
||||
# crash-loops on a denied mount every ~3s (visible as garbage on the
|
||||
# console) while core services like nsncd fail the same way.
|
||||
#
|
||||
# ...,mount=nfs;nfs4: without it AppArmor blanket-denies the `nfs`/
|
||||
# `rpc_pipefs` mount syscalls any NFS client share needs -- confirmed
|
||||
# live on lxc-docker: `mount: /var/lib/nfs/rpc_pipefs: permission
|
||||
# denied`. The value's `;` (Proxmox's own multi-fstype separator for
|
||||
# this one feature, per PVE::LXC's use of PVE::ParseUtils::split_list)
|
||||
# must stay single-quoted here: create_cmd is sent to `remote()`, which
|
||||
# hands the whole string to `ssh` as a single command for the *remote*
|
||||
# shell to parse -- unquoted, that `;` would be read as a remote
|
||||
# command separator and silently truncate this into two commands.
|
||||
create_cmd="${sudo_prefix} pct create ${vmid} ${iso_storage}:vztmpl/${remote_filename} --unprivileged ${unprivileged_flag} --features '${PROXMOX_DEFAULT_LXC_FEATURES}' --rootfs ${storage}:${local_disk_size} --hostname ${name} --cores ${cores} --memory ${memory} --swap ${local_swap} --net0 name=eth0,bridge=${bridge},ip=dhcp"
|
||||
remote "$create_cmd"
|
||||
remote "${sudo_prefix} pct start ${vmid}"
|
||||
else
|
||||
echo "==> Creating VM ${vmid} (${name})..."
|
||||
# pre-enrolled-keys=0 disables OVMF's Secure Boot key pre-enrollment --
|
||||
# required, or systemd-boot (unsigned) can't be trusted by the firmware.
|
||||
# --agent 1: wires up the virtio-serial channel QEMU exposes to the guest.
|
||||
# modules/common/configuration.nix sets services.qemuGuest.enable = true
|
||||
# on every host, so the guest-side qemu-ga daemon is already running --
|
||||
# without this flag Proxmox never creates the channel it listens on, so
|
||||
# `qm guest exec`/`qm agent` and the UI's IP-address display silently
|
||||
# never work for any VM this script creates.
|
||||
remote "${sudo_prefix} qm create ${vmid} --name ${name} --memory ${memory} --cores ${cores} \
|
||||
--net0 virtio,bridge=${bridge} --bios ovmf --machine q35 --scsihw virtio-scsi-pci \
|
||||
--efidisk0 ${storage}:1,efitype=4m,pre-enrolled-keys=0 --agent enabled=1"
|
||||
|
||||
# VMs built on the node: import from the repo dir (where disko/QEMU wrote it).
|
||||
# VMs from --image: import from remote_path (where scp uploaded it).
|
||||
_import_path="${remote_path}"
|
||||
[[ -z "$image" ]] && _import_path="${vm_built_raw}"
|
||||
if [[ "$dry_run" -eq 1 ]]; then
|
||||
echo "[dry-run] ssh ${ssh_target} -- ${sudo_display}qm importdisk ${vmid} ${_import_path} ${storage}"
|
||||
echo "[dry-run] (would parse the resulting disk identifier from that output)"
|
||||
echo "[dry-run] ssh ${ssh_target} -- ${sudo_display}qm set ${vmid} --scsi0 ${storage}:<parsed-disk-id>"
|
||||
else
|
||||
if ! importdisk_output="$(ssh "$ssh_target" "${sudo_prefix} qm importdisk ${vmid} ${_import_path} ${storage}" 2>&1)"; then
|
||||
echo "ERROR: qm importdisk failed:" >&2
|
||||
echo "${importdisk_output}" >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "$importdisk_output"
|
||||
# PVE output format: "unusedN: successfully imported disk '<storage>:<vol>'"
|
||||
# (lowercase "successfully", no "as"; the primary regex targets this form; the
|
||||
# || true inside the substitution prevents set -e from aborting when grep finds
|
||||
# no match -- without it the script would silently exit before reaching the
|
||||
# fallback whenever the PVE format doesn't match).
|
||||
disk_id="$(echo "$importdisk_output" | grep -oP "successfully imported disk '\\K[^']+" || true)"
|
||||
if [[ -z "$disk_id" ]]; then
|
||||
# Fallback for other PVE output variants: read qm config directly.
|
||||
unused_line="$(ssh "$ssh_target" "${sudo_prefix} qm config ${vmid}" | grep '^unused[0-9]*:' | head -1 || true)"
|
||||
if [[ -n "$unused_line" ]]; then
|
||||
disk_id="${unused_line#*: }"
|
||||
echo "Note: disk ID resolved from qm config: ${disk_id}"
|
||||
fi
|
||||
fi
|
||||
if [[ -z "$disk_id" ]]; then
|
||||
echo "ERROR: couldn't parse the imported disk identifier from qm importdisk's output above." >&2
|
||||
echo "The VM shell (${vmid}) and imported disk both exist -- finish attaching it by hand:" >&2
|
||||
echo " ssh ${ssh_target} -- ${sudo_display}qm set ${vmid} --scsi0 ${storage}:<disk-id-from-output-above>" >&2
|
||||
echo " ssh ${ssh_target} -- ${sudo_display}qm set ${vmid} --boot order=scsi0" >&2
|
||||
exit 1
|
||||
fi
|
||||
remote "${sudo_prefix} qm set ${vmid} --scsi0 ${disk_id}"
|
||||
# The disk data is now in ZFS; remove the source raw file (only for images
|
||||
# we built on the node -- --image uploads are the operator's to manage).
|
||||
if [[ -z "$image" ]]; then
|
||||
ssh "$ssh_target" "${sudo_prefix} rm -f '${_import_path}'" 2>/dev/null || \
|
||||
echo "Warning: couldn't remove ${_import_path} from ${node} -- you can delete it manually" >&2
|
||||
fi
|
||||
fi
|
||||
remote "${sudo_prefix} qm set ${vmid} --boot order=scsi0"
|
||||
remote "${sudo_prefix} qm start ${vmid}"
|
||||
fi
|
||||
|
||||
echo
|
||||
if [[ "$dry_run" -eq 1 ]]; then
|
||||
echo "[dry-run] Nothing was built, uploaded, or created."
|
||||
else
|
||||
echo "Done. ${name} (VMID ${vmid}) should be booting on ${node}."
|
||||
fi
|
||||
@@ -1,253 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# recover-hosts.sh — Fix sops/SSH-key/GitHub-token issues on deployed NixOS hosts
|
||||
# and trigger a Switch-nix rebuild on each.
|
||||
#
|
||||
# Run from the repo root on the workstation (nixos@nixos):
|
||||
# bash scripts/recover-hosts.sh [<hostname> ...]
|
||||
#
|
||||
# With no args it discovers and checks every known hostname.
|
||||
# With args it checks only those hostnames:
|
||||
# bash scripts/recover-hosts.sh tor-relay
|
||||
#
|
||||
# Fixes applied automatically (then prompts before rebuilding):
|
||||
# 1. SSH host key drift — live key no longer matches host-keys/<target>_ssh_host_ed25519_key
|
||||
# Fix: scp the registered key back and restore it (needs sudo once per host).
|
||||
# To push new keys proactively (before drift, e.g. right after
|
||||
# sync-host-keys.sh --regenerate-all-keys), use instead:
|
||||
# scripts/secrets/push-host-keys.sh --all
|
||||
# 2. Stale/invalid GitHub access token — the rendered nix-github-token.conf has
|
||||
# a token GitHub rejects (401), blocking any rebuild that fetches disko or
|
||||
# other public GitHub flake inputs.
|
||||
# Fix: empty the rendered file so nix makes unauthenticated requests instead.
|
||||
# Public repos (disko, nixpkgs, etc.) work fine without auth. sops-nix
|
||||
# re-renders the correct new token automatically after the first successful
|
||||
# rebuild.
|
||||
#
|
||||
# Both fixes need one interactive sudo session per host. The script opens a
|
||||
# single ssh -t per broken host so you enter the password once and all steps
|
||||
# run in sequence.
|
||||
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")/.."
|
||||
source scripts/env.sh 2>/dev/null || true
|
||||
|
||||
SSH_OPTS=(-o StrictHostKeyChecking=no -o BatchMode=yes -o ConnectTimeout=5)
|
||||
SSH_USER=nixos
|
||||
|
||||
# Known flake-target → ssh hostname map for all currently-defined hosts.
|
||||
# Add new hosts here as they are deployed.
|
||||
declare -A TARGET_HOST=(
|
||||
[lxc-docker]=docker
|
||||
[lxc-nix-cache]=nix-cache
|
||||
[lxc-pxe-boot]=pxe-boot
|
||||
[lxc-tor-relay]=tor-relay
|
||||
[lxc-minimal]=nix-minimal
|
||||
[proxmox-server]=server
|
||||
[baremetal-gui]=nixos
|
||||
)
|
||||
|
||||
# ── helpers ───────────────────────────────────────────────────────────────────
|
||||
|
||||
info() { echo " [✓] $*"; }
|
||||
warn() { echo " [!] $*"; }
|
||||
step() { echo "==> $*"; }
|
||||
|
||||
ssh_host_age() {
|
||||
ssh-keyscan -t ed25519 "$1" 2>/dev/null \
|
||||
| nix shell nixpkgs#ssh-to-age --command ssh-to-age 2>/dev/null \
|
||||
| head -1 || true
|
||||
}
|
||||
|
||||
registered_age() {
|
||||
local keyfile="host-keys/${1}_ssh_host_ed25519_key.pub"
|
||||
[ -f "$keyfile" ] || return 0
|
||||
nix shell nixpkgs#ssh-to-age --command ssh-to-age < "$keyfile" 2>/dev/null \
|
||||
| head -1 || true
|
||||
}
|
||||
|
||||
github_token_valid() {
|
||||
local host=$1
|
||||
local raw token code
|
||||
raw=$(ssh "${SSH_OPTS[@]}" "$SSH_USER@$host" \
|
||||
"cat /run/secrets/rendered/nix-github-token.conf 2>/dev/null || true")
|
||||
token=$(echo "$raw" | grep -oP '(?<=github\.com=)\S+' || true)
|
||||
if [ -z "$token" ]; then
|
||||
return 0 # no token = unauthenticated, works for public repos
|
||||
fi
|
||||
code=$(curl -s -o /dev/null -w "%{http_code}" \
|
||||
-H "Authorization: token $token" \
|
||||
"https://api.github.com/repos/nix-community/disko" 2>/dev/null || echo 000)
|
||||
[ "$code" = "200" ]
|
||||
}
|
||||
|
||||
# ── discover hosts ────────────────────────────────────────────────────────────
|
||||
|
||||
if [ $# -gt 0 ]; then
|
||||
HOSTNAMES=("$@")
|
||||
else
|
||||
HOSTNAMES=()
|
||||
seen=()
|
||||
for target in "${!TARGET_HOST[@]}"; do
|
||||
h="${TARGET_HOST[$target]}"
|
||||
# deduplicate (e.g. proxmox-server and lxc-server both map to "server")
|
||||
if [[ ! " ${seen[*]:-} " =~ " $h " ]]; then
|
||||
seen+=("$h")
|
||||
if ssh "${SSH_OPTS[@]}" "$SSH_USER@$h" "true" 2>/dev/null; then
|
||||
HOSTNAMES+=("$h")
|
||||
fi
|
||||
fi
|
||||
done
|
||||
fi
|
||||
|
||||
if [ ${#HOSTNAMES[@]} -eq 0 ]; then
|
||||
echo "No reachable hosts found. Pass hostnames explicitly or check SSH."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo ""
|
||||
echo "Hosts to check: ${HOSTNAMES[*]}"
|
||||
echo ""
|
||||
|
||||
# ── check phase ───────────────────────────────────────────────────────────────
|
||||
|
||||
NEEDS_FIX=()
|
||||
|
||||
for host in "${HOSTNAMES[@]}"; do
|
||||
step "$host"
|
||||
|
||||
if ! ssh "${SSH_OPTS[@]}" "$SSH_USER@$host" "true" 2>/dev/null; then
|
||||
warn "SSH unreachable — clearing stale known_hosts entry"
|
||||
ssh-keygen -R "$host" 2>/dev/null || true
|
||||
continue
|
||||
fi
|
||||
|
||||
flake_target=$(ssh "${SSH_OPTS[@]}" "$SSH_USER@$host" \
|
||||
"cat /etc/flake-target 2>/dev/null || true")
|
||||
echo " flake-target: ${flake_target:-unknown}"
|
||||
|
||||
host_broken=false
|
||||
|
||||
# SSH host key
|
||||
if [ -n "$flake_target" ] && [ -f "host-keys/${flake_target}_ssh_host_ed25519_key.pub" ]; then
|
||||
live=$(ssh_host_age "$host")
|
||||
want=$(registered_age "$flake_target")
|
||||
if [ "$live" = "$want" ]; then
|
||||
info "SSH host key OK"
|
||||
else
|
||||
warn "SSH host key MISMATCH (live ≠ host-keys/) -- use push-host-keys.sh proactively next time"
|
||||
echo " live: $live"
|
||||
echo " registered: $want"
|
||||
host_broken=true
|
||||
fi
|
||||
else
|
||||
echo " [~] No host-keys/ entry for ${flake_target:-unknown} — skipping key check"
|
||||
fi
|
||||
|
||||
# GitHub token
|
||||
if github_token_valid "$host"; then
|
||||
info "GitHub token OK"
|
||||
else
|
||||
warn "GitHub token invalid (rebuild will fail with 401)"
|
||||
host_broken=true
|
||||
fi
|
||||
|
||||
# sops-nix result
|
||||
sops_result=$(ssh "${SSH_OPTS[@]}" "$SSH_USER@$host" \
|
||||
"systemctl show sops-nix --property=Result --value 2>/dev/null || echo unknown")
|
||||
if [ "$sops_result" = "success" ]; then
|
||||
info "sops-nix: success"
|
||||
else
|
||||
warn "sops-nix: $sops_result"
|
||||
fi
|
||||
|
||||
$host_broken && NEEDS_FIX+=("$host")
|
||||
echo ""
|
||||
done
|
||||
|
||||
# ── fix phase ─────────────────────────────────────────────────────────────────
|
||||
|
||||
if [ ${#NEEDS_FIX[@]} -eq 0 ]; then
|
||||
echo "All hosts healthy — nothing to fix."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
echo "Hosts needing fixes: ${NEEDS_FIX[*]}"
|
||||
echo ""
|
||||
echo "Each fix requires one sudo session per host. You will be prompted for"
|
||||
echo "the nixos sudo password once per host; all steps run in that session."
|
||||
echo ""
|
||||
read -r -p "Proceed with fixes + Switch-nix on each broken host? [y/N] " confirm
|
||||
[[ "$confirm" =~ ^[Yy]$ ]] || { echo "Aborted."; exit 0; }
|
||||
echo ""
|
||||
|
||||
for host in "${NEEDS_FIX[@]}"; do
|
||||
step "Fixing $host"
|
||||
flake_target=$(ssh "${SSH_OPTS[@]}" "$SSH_USER@$host" \
|
||||
"cat /etc/flake-target 2>/dev/null || true")
|
||||
|
||||
fix_script=""
|
||||
|
||||
# Fix 1: restore SSH host key
|
||||
live=$(ssh_host_age "$host")
|
||||
want=$(registered_age "${flake_target:-}")
|
||||
if [ -n "$want" ] && [ "$live" != "$want" ]; then
|
||||
echo " Uploading registered SSH host key (private + public)..."
|
||||
scp -o StrictHostKeyChecking=no \
|
||||
"host-keys/${flake_target}_ssh_host_ed25519_key" \
|
||||
"$SSH_USER@$host:/tmp/recover_ed25519_key"
|
||||
scp -o StrictHostKeyChecking=no \
|
||||
"host-keys/${flake_target}_ssh_host_ed25519_key.pub" \
|
||||
"$SSH_USER@$host:/tmp/recover_ed25519_key.pub"
|
||||
fix_script+='
|
||||
echo "[fix] Restoring SSH host key..."
|
||||
install -m 0600 /tmp/recover_ed25519_key /etc/ssh/ssh_host_ed25519_key
|
||||
install -m 0644 /tmp/recover_ed25519_key.pub /etc/ssh/ssh_host_ed25519_key.pub
|
||||
rm -f /tmp/recover_ed25519_key /tmp/recover_ed25519_key.pub
|
||||
echo " Done."
|
||||
'
|
||||
ssh-keygen -R "$host" 2>/dev/null || true
|
||||
fi
|
||||
|
||||
# Fix 2: clear invalid GitHub token
|
||||
if ! github_token_valid "$host"; then
|
||||
fix_script+='
|
||||
echo "[fix] Clearing stale GitHub token (nix will use unauthenticated access)..."
|
||||
echo "" > /run/secrets/rendered/nix-github-token.conf
|
||||
systemctl restart nix-daemon 2>/dev/null || true
|
||||
echo " Done."
|
||||
'
|
||||
fi
|
||||
|
||||
# Fix 3: rebuild
|
||||
fix_script+='
|
||||
echo "[fix] Running nixos-rebuild switch..."
|
||||
nixos-rebuild switch \
|
||||
--no-write-lock-file \
|
||||
--refresh \
|
||||
--flake "git+https://gitea.lan.ddnsgeek.com/beatzaplenty/nixos.git#$(cat /etc/flake-target)"
|
||||
echo "[fix] Rebuild complete."
|
||||
'
|
||||
|
||||
echo " Opening SSH session (enter sudo password when prompted)..."
|
||||
if ssh -t -o StrictHostKeyChecking=no "$SSH_USER@$host" \
|
||||
"sudo bash -s" <<< "$fix_script"; then
|
||||
echo ""
|
||||
info "$host fixed and rebuilt"
|
||||
else
|
||||
rc=$?
|
||||
echo ""
|
||||
warn "$host: rebuild exited with code $rc (may still have succeeded — check sops-nix below)"
|
||||
fi
|
||||
|
||||
# Verify: re-check sops-nix result post-rebuild
|
||||
sops_result_after=$(ssh "${SSH_OPTS[@]}" "$SSH_USER@$host" \
|
||||
"systemctl show sops-nix --property=Result --value 2>/dev/null || echo unknown" 2>/dev/null || echo "ssh-failed")
|
||||
if [ "$sops_result_after" = "success" ]; then
|
||||
info "$host sops-nix: success post-rebuild"
|
||||
else
|
||||
warn "$host sops-nix: $sops_result_after post-rebuild (may need another pass)"
|
||||
fi
|
||||
echo ""
|
||||
done
|
||||
|
||||
echo "Recovery complete."
|
||||
@@ -1,150 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Backs up the local sops age key (the private key that decrypts
|
||||
# secrets/*.yaml -- normally the one trusted as &admin) to an arbitrary
|
||||
# destination path, e.g. a USB drive or other offline storage, so it can
|
||||
# later be restored and handed to rotate-admin-key.sh if this machine's
|
||||
# copy is ever lost, or to run either script from a different machine.
|
||||
#
|
||||
# Usage:
|
||||
# scripts/secrets/backup-admin-key.sh <dest-path> [--key-file <path>] [--force] [--dry-run]
|
||||
#
|
||||
# Source key resolution matches sops/age's own default order:
|
||||
# $SOPS_AGE_KEY (inline identity text) if set, else
|
||||
# --key-file if given, else
|
||||
# $SOPS_AGE_KEY_FILE if set, else
|
||||
# ${XDG_CONFIG_HOME:-$HOME/.config}/sops/age/keys.txt
|
||||
set -euo pipefail
|
||||
|
||||
repo_root="$(cd "$(dirname "$0")/../.." && pwd)"
|
||||
sops_yaml="${repo_root}/.sops.yaml"
|
||||
|
||||
# shellcheck source=../env.sh
|
||||
source "${repo_root}/scripts/env.sh"
|
||||
# shellcheck source=../lib/sops-age.sh
|
||||
source "${repo_root}/scripts/lib/sops-age.sh"
|
||||
|
||||
# Pin cwd for the same reason rotate-admin-key.sh does: age/sops calls
|
||||
# below should never depend on wherever the caller's shell happened to be.
|
||||
cd "$repo_root"
|
||||
|
||||
usage() {
|
||||
cat <<EOF
|
||||
Usage: $0 <dest-path> [--key-file <path>] [--force] [--dry-run]
|
||||
|
||||
<dest-path> Where to write the backup. Parent directories are
|
||||
created as needed. Written with 0600 permissions.
|
||||
--key-file <path> Read the key from here instead of the default
|
||||
sops/age resolution (\$SOPS_AGE_KEY_FILE, then
|
||||
\${XDG_CONFIG_HOME:-\$HOME/.config}/sops/age/keys.txt).
|
||||
Ignored if \$SOPS_AGE_KEY is set (that always wins,
|
||||
same precedence sops/age itself uses).
|
||||
--force Overwrite <dest-path> if it already exists.
|
||||
--dry-run Print what would happen; write nothing.
|
||||
EOF
|
||||
}
|
||||
|
||||
dry_run=0
|
||||
force=0
|
||||
key_file="$DEFAULT_SOPS_AGE_KEY_FILE"
|
||||
args=()
|
||||
|
||||
while [[ $# -gt 0 ]]; do
|
||||
case "$1" in
|
||||
--dry-run)
|
||||
dry_run=1
|
||||
shift
|
||||
;;
|
||||
--force)
|
||||
force=1
|
||||
shift
|
||||
;;
|
||||
--key-file)
|
||||
key_file="${2:?--key-file requires a path}"
|
||||
shift 2
|
||||
;;
|
||||
-h | --help)
|
||||
usage
|
||||
exit 0
|
||||
;;
|
||||
--*)
|
||||
echo "Unknown option: $1" >&2
|
||||
usage >&2
|
||||
exit 1
|
||||
;;
|
||||
*)
|
||||
args+=("$1")
|
||||
shift
|
||||
;;
|
||||
esac
|
||||
done
|
||||
|
||||
if [[ "${#args[@]}" -ne 1 ]]; then
|
||||
usage >&2
|
||||
exit 1
|
||||
fi
|
||||
dest="${args[0]}"
|
||||
|
||||
nix_extra_opts
|
||||
|
||||
if [[ -n "${SOPS_AGE_KEY:-}" ]]; then
|
||||
echo "==> Source: \$SOPS_AGE_KEY (inline identity from the environment)."
|
||||
src_content="$SOPS_AGE_KEY"
|
||||
else
|
||||
[[ -s "$key_file" ]] || {
|
||||
echo "ERROR: no key found. \$SOPS_AGE_KEY is unset and ${key_file} doesn't exist or is empty." >&2
|
||||
exit 1
|
||||
}
|
||||
echo "==> Source: ${key_file}"
|
||||
src_content="$(cat "$key_file")"
|
||||
fi
|
||||
|
||||
# Round-trip through a private scratch file (rather than trusting the
|
||||
# source string as-is) so age-keygen -y validates it's a real identity
|
||||
# before anything is written to <dest-path>.
|
||||
scratch="$(mktemp)"
|
||||
trap 'rm -f "$scratch"' EXIT
|
||||
( umask 077; printf '%s\n' "$src_content" > "$scratch" )
|
||||
|
||||
src_pub="$(age_pubkey_from_identity_file "$scratch")" || {
|
||||
echo "ERROR: source doesn't look like a valid age identity (age-keygen -y failed)." >&2
|
||||
exit 1
|
||||
}
|
||||
echo " public key: ${src_pub}"
|
||||
|
||||
current_admin_pub="$(sops_yaml_admin_pubkey "$sops_yaml")"
|
||||
if [[ -n "$current_admin_pub" && "$current_admin_pub" != "$src_pub" ]]; then
|
||||
echo "NOTE: this key does not match .sops.yaml's current &admin entry (${current_admin_pub})."
|
||||
echo " Backing it up anyway -- this script doesn't require it to be the admin key."
|
||||
fi
|
||||
|
||||
if [[ -e "$dest" && "$force" -ne 1 ]]; then
|
||||
echo "ERROR: ${dest} already exists. Pass --force to overwrite." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if [[ "$dry_run" -eq 1 ]]; then
|
||||
echo
|
||||
echo "[dry-run] would write $(wc -c <"$scratch" | tr -d ' ') bytes to ${dest} (mode 0600)"
|
||||
[[ -e "$dest" ]] && echo "[dry-run] would overwrite existing file (--force given)"
|
||||
echo "[dry-run] Nothing was written. Re-run without --dry-run to apply this."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
mkdir -p "$(dirname "$dest")"
|
||||
install -m 600 "$scratch" "$dest"
|
||||
|
||||
dest_pub="$(age_pubkey_from_identity_file "$dest")"
|
||||
if [[ "$dest_pub" != "$src_pub" ]]; then
|
||||
echo "ERROR: ${dest} was written but its public key doesn't match the source -- investigate before relying on this backup." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
cat <<EOF
|
||||
|
||||
Done. Backed up to: ${dest}
|
||||
public key: ${dest_pub}
|
||||
|
||||
This is a private key -- store it somewhere offline/secure, not in this
|
||||
repo or anywhere it'd get committed. Restore it with:
|
||||
scripts/secrets/rotate-admin-key.sh ${dest}
|
||||
EOF
|
||||
@@ -1,318 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Pushes newly-generated SSH host keys from host-keys/ to already-running
|
||||
# NixOS hosts, so they can decrypt sops secrets after a nixos-rebuild
|
||||
# following scripts/secrets/sync-host-keys.sh --regenerate-all-keys.
|
||||
#
|
||||
# Before pushing any key, verifies that .sops.yaml and secrets/*.yaml are
|
||||
# committed and pushed to the remote -- hosts rebuild from the remote Gitea
|
||||
# flake, so recipient changes must land there before any rebuild, not just
|
||||
# before the key push.
|
||||
#
|
||||
# push-host-keys.sh --all [--dry-run] [--skip-git-check]
|
||||
# push-host-keys.sh <target> [--dry-run] [--skip-git-check]
|
||||
#
|
||||
# --all Push to every reachable managed host. Default when no
|
||||
# target is given.
|
||||
# <target> Push to one flake target only (e.g. lxc-server).
|
||||
# --dry-run Print what would be done; write nothing.
|
||||
# --skip-git-check Skip the commit/push check. Use only when the remote
|
||||
# already has the current .sops.yaml/secrets/*.yaml.
|
||||
#
|
||||
# SSH: connects as SSH_USER@<hostname> (default: nixos, the user with the
|
||||
# admin authorized key), then installs files via sudo -S (reads the sudo
|
||||
# password from stdin). The password is prompted once at startup and reused
|
||||
# for every host -- no PTY or terminal required on the remote side.
|
||||
# Hosts are reached at their bare hostname (relies on LAN DNS/mDNS).
|
||||
set -euo pipefail
|
||||
|
||||
repo_root="$(cd "$(dirname "$0")/../.." && pwd)"
|
||||
keydir="${repo_root}/host-keys"
|
||||
|
||||
# shellcheck source=../env.sh
|
||||
source "${repo_root}/scripts/env.sh"
|
||||
# shellcheck source=../lib/nix-eval.sh
|
||||
source "${repo_root}/scripts/lib/nix-eval.sh"
|
||||
|
||||
: "${SSH_USER:=nixos}"
|
||||
SSH_OPTS=(-o StrictHostKeyChecking=no -o BatchMode=yes -o ConnectTimeout=5)
|
||||
|
||||
dry_run=0
|
||||
skip_git_check=0
|
||||
sudo_password=""
|
||||
|
||||
usage() {
|
||||
cat <<EOF
|
||||
Usage: $0 [--all | <target>] [--dry-run] [--skip-git-check]
|
||||
|
||||
--all Push to every reachable managed host. Default when no
|
||||
target is given.
|
||||
<target> Push to one flake target only (e.g. lxc-server).
|
||||
--dry-run Print what would be done; write nothing.
|
||||
--skip-git-check Skip the check that .sops.yaml/secrets/*.yaml are
|
||||
committed and pushed to the remote repo.
|
||||
|
||||
Environment:
|
||||
SSH_USER SSH username (default: nixos).
|
||||
SUDO_PASS Sudo password (skips the interactive prompt; useful
|
||||
when calling from another script).
|
||||
EOF
|
||||
}
|
||||
|
||||
# Prompt for the sudo password once; store it for all _do_push calls.
|
||||
# Accepts SUDO_PASS from the environment to allow non-interactive callers.
|
||||
prompt_sudo_password() {
|
||||
[[ "$dry_run" -eq 1 ]] && return
|
||||
if [[ -n "${SUDO_PASS:-}" ]]; then
|
||||
sudo_password="$SUDO_PASS"
|
||||
return
|
||||
fi
|
||||
# read exits non-zero when stdin is not a terminal (e.g. CI, background
|
||||
# agents). Catch that and give a clear message rather than a silent exit.
|
||||
if ! read -r -s -p "sudo password for ${SSH_USER} on remote hosts: " sudo_password; then
|
||||
echo >&2
|
||||
echo "ERROR: stdin is not a terminal -- cannot prompt for sudo password." >&2
|
||||
echo " Set SUDO_PASS=<password> in the environment and re-run." >&2
|
||||
exit 1
|
||||
fi
|
||||
echo >&2
|
||||
}
|
||||
|
||||
locally_managed_hosts() {
|
||||
for f in "${keydir}"/*_ssh_host_ed25519_key.pub; do
|
||||
[[ -e "$f" ]] || continue
|
||||
basename "$f" _ssh_host_ed25519_key.pub
|
||||
done
|
||||
}
|
||||
|
||||
# --- git state check/fix --------------------------------------------------
|
||||
# Hosts rebuild from the remote Gitea flake:
|
||||
# nixos-rebuild switch --flake "git+https://<gitea>/nixos.git#<target>"
|
||||
# so .sops.yaml (updated recipients) and secrets/*.yaml (re-encrypted DEKs)
|
||||
# must be committed and pushed before any rebuild can succeed. This check
|
||||
# catches the common case where --regenerate-all-keys was just run but the
|
||||
# resulting diff hasn't been committed/pushed yet.
|
||||
ensure_remote_current() {
|
||||
[[ "$skip_git_check" -eq 1 ]] && return
|
||||
|
||||
cd "$repo_root"
|
||||
|
||||
local dirty_unstaged dirty_staged
|
||||
dirty_unstaged="$(git diff --name-only -- .sops.yaml secrets/ 2>/dev/null || true)"
|
||||
dirty_staged="$(git diff --cached --name-only -- .sops.yaml secrets/ 2>/dev/null || true)"
|
||||
|
||||
if [[ -n "$dirty_unstaged" || -n "$dirty_staged" ]]; then
|
||||
echo "Uncommitted changes in sops-managed files:"
|
||||
[[ -n "$dirty_unstaged" ]] && sed 's/^/ (unstaged) /' <<<"$dirty_unstaged"
|
||||
[[ -n "$dirty_staged" ]] && sed 's/^/ (staged) /' <<<"$dirty_staged"
|
||||
echo
|
||||
if [[ "$dry_run" -eq 1 ]]; then
|
||||
echo "[dry-run] would prompt to commit .sops.yaml/secrets/ before continuing."
|
||||
else
|
||||
read -rp "Commit .sops.yaml + secrets/ now? [y/N]: " ans
|
||||
if [[ "$ans" =~ ^[Yy]$ ]]; then
|
||||
git add -- .sops.yaml secrets/
|
||||
git commit -m "secrets: update recipients and re-encrypt for host key changes"
|
||||
echo "Committed."
|
||||
else
|
||||
echo "Continuing with uncommitted changes -- the remote won't have the"
|
||||
echo "updated recipients until you commit and push."
|
||||
fi
|
||||
fi
|
||||
echo
|
||||
fi
|
||||
|
||||
# Check if we're ahead of the remote tracking branch
|
||||
local ahead
|
||||
ahead="$(git rev-list --count '@{upstream}..HEAD' 2>/dev/null || echo "")"
|
||||
if [[ -z "$ahead" ]]; then
|
||||
echo "NOTE: no remote tracking branch found -- skipping push check."
|
||||
echo " Ensure the remote has the current .sops.yaml/secrets/ before"
|
||||
echo " triggering nixos-rebuild on any host."
|
||||
echo
|
||||
return
|
||||
fi
|
||||
|
||||
if [[ "$ahead" -gt 0 ]]; then
|
||||
echo "Local branch is ${ahead} commit(s) ahead of remote."
|
||||
if [[ "$dry_run" -eq 1 ]]; then
|
||||
echo "[dry-run] would prompt to push before continuing."
|
||||
else
|
||||
read -rp "Push to remote now? [y/N]: " ans
|
||||
if [[ "$ans" =~ ^[Yy]$ ]]; then
|
||||
git push
|
||||
echo "Pushed."
|
||||
else
|
||||
echo "Continuing without pushing -- remember to push before running"
|
||||
echo "nixos-rebuild on any of these hosts."
|
||||
fi
|
||||
fi
|
||||
echo
|
||||
fi
|
||||
}
|
||||
|
||||
# --- key installation (shared) -------------------------------------------
|
||||
_do_push() {
|
||||
local hostname="$1" target="$2"
|
||||
local keyfile="${keydir}/${target}_ssh_host_ed25519_key"
|
||||
local pubfile="${keyfile}.pub"
|
||||
|
||||
if [[ "$dry_run" -eq 1 ]]; then
|
||||
echo " [dry-run] would scp host-keys/${target}_ssh_host_ed25519_key{,.pub} to /tmp/"
|
||||
echo " [dry-run] would: sudo -S install -m 0600/0644 to /etc/ssh/ and rm /tmp copies"
|
||||
return
|
||||
fi
|
||||
|
||||
# Upload to /tmp (writable as nixos, no privilege needed)
|
||||
scp -o StrictHostKeyChecking=no \
|
||||
"$keyfile" "${SSH_USER}@${hostname}:/tmp/push_ed25519_key"
|
||||
scp -o StrictHostKeyChecking=no \
|
||||
"$pubfile" "${SSH_USER}@${hostname}:/tmp/push_ed25519_key.pub"
|
||||
|
||||
# Install via sudo -S: the password is piped via herestring so no PTY is
|
||||
# needed on either side. -p '' suppresses sudo's own prompt string.
|
||||
ssh -o StrictHostKeyChecking=no "${SSH_USER}@${hostname}" \
|
||||
"sudo -S -p '' bash -c '
|
||||
install -m 0600 /tmp/push_ed25519_key /etc/ssh/ssh_host_ed25519_key
|
||||
install -m 0644 /tmp/push_ed25519_key.pub /etc/ssh/ssh_host_ed25519_key.pub
|
||||
rm -f /tmp/push_ed25519_key /tmp/push_ed25519_key.pub
|
||||
echo \" [ok] host key installed\"
|
||||
'" <<< "$sudo_password"
|
||||
|
||||
# Drop the stale known_hosts entry for this host (public key just changed)
|
||||
ssh-keygen -R "$hostname" 2>/dev/null || true
|
||||
|
||||
echo " Done. Run nixos-rebuild switch on ${hostname} to activate."
|
||||
}
|
||||
|
||||
# --- single named target --------------------------------------------------
|
||||
push_target() {
|
||||
local target="$1"
|
||||
local keyfile="${keydir}/${target}_ssh_host_ed25519_key"
|
||||
|
||||
if [[ ! -f "$keyfile" ]]; then
|
||||
echo "ERROR: host-keys/${target}_ssh_host_ed25519_key not found." >&2
|
||||
echo " This target may not be locally managed (e.g. &${target} was" >&2
|
||||
echo " registered from the host's real SSH key, not generated here)." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
local hostname
|
||||
hostname="$(flake_target_hostname "$repo_root" "$target")"
|
||||
if [[ -z "$hostname" ]]; then
|
||||
echo "ERROR: cannot resolve hostname for '${target}' from the flake." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "==> ${target} (→ ${hostname})"
|
||||
|
||||
if ! ssh "${SSH_OPTS[@]}" "${SSH_USER}@${hostname}" true 2>/dev/null; then
|
||||
echo " SKIP: ${SSH_USER}@${hostname} unreachable."
|
||||
return
|
||||
fi
|
||||
|
||||
# Sanity-check that /etc/flake-target on the host agrees
|
||||
local live_target
|
||||
live_target="$(ssh "${SSH_OPTS[@]}" "${SSH_USER}@${hostname}" \
|
||||
"cat /etc/flake-target 2>/dev/null || true")"
|
||||
if [[ -n "$live_target" && "$live_target" != "$target" ]]; then
|
||||
echo " WARN: host reports /etc/flake-target='${live_target}', not '${target}'."
|
||||
echo " Pushing the key you specified (${target}) anyway."
|
||||
fi
|
||||
|
||||
_do_push "$hostname" "$target"
|
||||
}
|
||||
|
||||
# --- all managed hosts ----------------------------------------------------
|
||||
# For each unique hostname derived from managed targets, SSHes in and reads
|
||||
# /etc/flake-target to determine which key to push -- handles the case where
|
||||
# multiple targets share a hostname (e.g. lxc-server and proxmox-server both
|
||||
# resolve to "server"; only one is actually running).
|
||||
push_all() {
|
||||
mapfile -t managed < <(locally_managed_hosts)
|
||||
if [[ "${#managed[@]}" -eq 0 ]]; then
|
||||
echo "No managed keys in host-keys/ -- nothing to push."
|
||||
return
|
||||
fi
|
||||
|
||||
echo "Pushing to all reachable managed hosts..."
|
||||
echo
|
||||
|
||||
declare -A seen_hostnames=()
|
||||
local t hostname
|
||||
for t in "${managed[@]}"; do
|
||||
hostname="$(flake_target_hostname "$repo_root" "$t" 2>/dev/null || true)"
|
||||
[[ -z "$hostname" ]] && continue
|
||||
[[ -n "${seen_hostnames[$hostname]+x}" ]] && continue
|
||||
seen_hostnames["$hostname"]=1
|
||||
|
||||
echo "==> checking ${hostname}"
|
||||
|
||||
if ! ssh "${SSH_OPTS[@]}" "${SSH_USER}@${hostname}" true 2>/dev/null; then
|
||||
echo " SKIP: ${SSH_USER}@${hostname} unreachable."
|
||||
continue
|
||||
fi
|
||||
|
||||
# Ask the host which flake target it actually is
|
||||
local live_target
|
||||
live_target="$(ssh "${SSH_OPTS[@]}" "${SSH_USER}@${hostname}" \
|
||||
"cat /etc/flake-target 2>/dev/null || true")"
|
||||
|
||||
if [[ -z "$live_target" ]]; then
|
||||
echo " SKIP: no /etc/flake-target on host -- can't determine which key to push."
|
||||
continue
|
||||
fi
|
||||
|
||||
local live_keyfile="${keydir}/${live_target}_ssh_host_ed25519_key"
|
||||
if [[ ! -f "$live_keyfile" ]]; then
|
||||
echo " SKIP: host is '${live_target}' but no host-keys/${live_target}_... (hand-registered key, not managed here)."
|
||||
continue
|
||||
fi
|
||||
|
||||
echo " target: ${live_target}"
|
||||
_do_push "$hostname" "$live_target"
|
||||
done
|
||||
}
|
||||
|
||||
# --- main -----------------------------------------------------------------
|
||||
mode="all"
|
||||
target_arg=""
|
||||
extra_args=()
|
||||
|
||||
for arg in "$@"; do
|
||||
case "$arg" in
|
||||
--dry-run) dry_run=1 ;;
|
||||
--skip-git-check) skip_git_check=1 ;;
|
||||
--all) mode="all" ;;
|
||||
-h|--help) usage; exit 0 ;;
|
||||
--*) echo "Unknown option: $arg" >&2; usage >&2; exit 1 ;;
|
||||
*) extra_args+=("$arg") ;;
|
||||
esac
|
||||
done
|
||||
|
||||
if [[ "${#extra_args[@]}" -gt 1 ]]; then
|
||||
echo "ERROR: specify at most one target (or --all)." >&2
|
||||
usage >&2; exit 1
|
||||
elif [[ "${#extra_args[@]}" -eq 1 ]]; then
|
||||
mode="single"
|
||||
target_arg="${extra_args[0]}"
|
||||
fi
|
||||
|
||||
[[ "$dry_run" -eq 1 ]] && { echo "[dry-run] no changes will be made"; echo; }
|
||||
|
||||
nix_extra_opts
|
||||
ensure_remote_current
|
||||
prompt_sudo_password
|
||||
|
||||
if [[ "$mode" == "single" ]]; then
|
||||
push_target "$target_arg"
|
||||
else
|
||||
push_all
|
||||
fi
|
||||
|
||||
echo
|
||||
if [[ "$dry_run" -eq 1 ]]; then
|
||||
echo "[dry-run] Nothing was changed. Re-run without --dry-run to apply."
|
||||
else
|
||||
echo "Key push complete. For each updated host, run nixos-rebuild switch to"
|
||||
echo "apply the config and let sops-nix decrypt secrets with the new key."
|
||||
fi
|
||||
@@ -1,193 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Rotates the &admin sops age key: decrypts with a backed-up copy of the
|
||||
# key CURRENTLY trusted as &admin, replaces .sops.yaml's &admin entry with
|
||||
# a new key already present in this environment, and re-encrypts every
|
||||
# secrets/*.yaml for the new recipient set. After this runs, the old key
|
||||
# can no longer decrypt anything -- this is a real, one-way handoff of
|
||||
# trust, not a preview.
|
||||
#
|
||||
# This is the automation for the manual steps create-proxmox-resource.sh /
|
||||
# sync-host-keys.sh print when they bootstrap a brand-new, not-yet-trusted
|
||||
# age key on a machine that's never had admin access before:
|
||||
#
|
||||
# scripts/secrets/rotate-admin-key.sh /path/to/backed-up/admin/keys.txt
|
||||
#
|
||||
# The backup key's *public* key must match .sops.yaml's current &admin
|
||||
# entry -- this script verifies that by deriving it, it doesn't just trust
|
||||
# the filename or take it on faith. The new key defaults to wherever sops
|
||||
# itself would already look ($SOPS_AGE_KEY_FILE, then the XDG default), so
|
||||
# the common case is just pointing this at the restored backup.
|
||||
set -euo pipefail
|
||||
|
||||
repo_root="$(cd "$(dirname "$0")/../.." && pwd)"
|
||||
sops_yaml="${repo_root}/.sops.yaml"
|
||||
|
||||
# shellcheck source=../env.sh
|
||||
source "${repo_root}/scripts/env.sh"
|
||||
# shellcheck source=../lib/sops-age.sh
|
||||
source "${repo_root}/scripts/lib/sops-age.sh"
|
||||
|
||||
# sops resolves .sops.yaml by walking up from the process's cwd, not from
|
||||
# the target file's own path -- if this script were invoked from somewhere
|
||||
# other than the repo root (or from inside another checkout/worktree that
|
||||
# happens to have its own .sops.yaml), `sops updatekeys` would silently
|
||||
# re-encrypt against the WRONG config's recipient list instead of this
|
||||
# repo's. Pin cwd here so every sops/age call below is unambiguous
|
||||
# regardless of where the caller's shell started out.
|
||||
cd "$repo_root"
|
||||
|
||||
usage() {
|
||||
cat <<EOF
|
||||
Usage: $0 <path-to-backed-up-admin-key> [--new-key-file <path>] [--dry-run]
|
||||
|
||||
<path-to-backed-up-admin-key> age identity file for the key CURRENTLY
|
||||
trusted as &admin. Only ever read -- never
|
||||
copied or modified.
|
||||
--new-key-file <path> age identity file for the key to promote
|
||||
to &admin. Defaults to \$SOPS_AGE_KEY_FILE,
|
||||
then
|
||||
\${XDG_CONFIG_HOME:-\$HOME/.config}/sops/age/keys.txt
|
||||
(sops/age's own default resolution order).
|
||||
--dry-run Print what would change; touches nothing
|
||||
(.sops.yaml untouched, no sops updatekeys
|
||||
calls).
|
||||
EOF
|
||||
}
|
||||
|
||||
dry_run=0
|
||||
new_key_file="$DEFAULT_SOPS_AGE_KEY_FILE"
|
||||
args=()
|
||||
|
||||
while [[ $# -gt 0 ]]; do
|
||||
case "$1" in
|
||||
--dry-run)
|
||||
dry_run=1
|
||||
shift
|
||||
;;
|
||||
--new-key-file)
|
||||
new_key_file="${2:?--new-key-file requires a path}"
|
||||
shift 2
|
||||
;;
|
||||
-h | --help)
|
||||
usage
|
||||
exit 0
|
||||
;;
|
||||
--*)
|
||||
echo "Unknown option: $1" >&2
|
||||
usage >&2
|
||||
exit 1
|
||||
;;
|
||||
*)
|
||||
args+=("$1")
|
||||
shift
|
||||
;;
|
||||
esac
|
||||
done
|
||||
|
||||
if [[ "${#args[@]}" -ne 1 ]]; then
|
||||
usage >&2
|
||||
exit 1
|
||||
fi
|
||||
backup_key="${args[0]}"
|
||||
|
||||
[[ -s "$backup_key" ]] || { echo "ERROR: backup key file not found or empty: ${backup_key}" >&2; exit 1; }
|
||||
[[ -s "$new_key_file" ]] || { echo "ERROR: new key file not found or empty: ${new_key_file}" >&2; exit 1; }
|
||||
|
||||
nix_extra_opts
|
||||
|
||||
echo "==> Deriving public keys..."
|
||||
old_pub="$(age_pubkey_from_identity_file "$backup_key")"
|
||||
new_pub="$(age_pubkey_from_identity_file "$new_key_file")"
|
||||
echo " backup (old admin) key: ${old_pub}"
|
||||
echo " new admin key: ${new_pub}"
|
||||
|
||||
if [[ "$old_pub" == "$new_pub" ]]; then
|
||||
echo "ERROR: backup key and new key are identical -- nothing to rotate." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
current_admin_pub="$(sops_yaml_admin_pubkey "$sops_yaml")"
|
||||
if [[ -z "$current_admin_pub" ]]; then
|
||||
echo "ERROR: couldn't find a '&admin age1...' line in ${sops_yaml}." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if [[ "$current_admin_pub" != "$old_pub" ]]; then
|
||||
echo "ERROR: ${backup_key} doesn't match the current &admin key in .sops.yaml." >&2
|
||||
echo " .sops.yaml &admin: ${current_admin_pub}" >&2
|
||||
echo " backup key pubkey: ${old_pub}" >&2
|
||||
echo "Wrong backup file, or .sops.yaml has already moved on -- not touching anything." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
mapfile -t secrets_files < <(find "${repo_root}/secrets" -maxdepth 1 -name '*.yaml' | sort)
|
||||
if [[ "${#secrets_files[@]}" -eq 0 ]]; then
|
||||
echo "ERROR: no secrets/*.yaml files found under ${repo_root}/secrets." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# sops_can_decrypt <key-file> <secrets-file>: used both to confirm the
|
||||
# backup key still works before touching anything, and again after
|
||||
# rotation to confirm the new key does too.
|
||||
sops_can_decrypt() {
|
||||
local key_file="$1" secrets_file="$2"
|
||||
SOPS_AGE_KEY_FILE="$key_file" nix-shell "${NIX_OPTS[@]}" -p sops --run \
|
||||
"sops -d '${secrets_file}'" >/dev/null
|
||||
}
|
||||
|
||||
echo "==> Confirming the backup key can actually decrypt..."
|
||||
if ! sops_can_decrypt "$backup_key" "${secrets_files[0]}"; then
|
||||
echo "ERROR: backup key failed to decrypt $(basename "${secrets_files[0]}") -- aborting." >&2
|
||||
exit 1
|
||||
fi
|
||||
echo " OK: decrypted $(basename "${secrets_files[0]}")"
|
||||
|
||||
if [[ "$dry_run" -eq 1 ]]; then
|
||||
echo
|
||||
echo "[dry-run] would replace .sops.yaml's &admin line:"
|
||||
echo "[dry-run] - ${current_admin_pub}"
|
||||
echo "[dry-run] + ${new_pub}"
|
||||
echo "[dry-run] would then re-encrypt (sops updatekeys --yes) for the new recipient set:"
|
||||
for f in "${secrets_files[@]}"; do
|
||||
echo "[dry-run] secrets/$(basename "$f")"
|
||||
done
|
||||
echo
|
||||
echo "[dry-run] Nothing was changed. Re-run without --dry-run to apply this."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
echo "==> Rotating .sops.yaml's &admin key..."
|
||||
sed -i "s|^ - &admin age1[a-z0-9]*| - \&admin ${new_pub}|" "$sops_yaml"
|
||||
grep -qF "$new_pub" "$sops_yaml" || {
|
||||
echo "ERROR: sed edit didn't take -- .sops.yaml left unchanged, check it by hand." >&2
|
||||
exit 1
|
||||
}
|
||||
echo " Updated."
|
||||
|
||||
echo "==> Re-encrypting secrets/*.yaml for the new recipient set..."
|
||||
for f in "${secrets_files[@]}"; do
|
||||
echo "==> $(basename "$f")"
|
||||
sops_updatekeys "$f" "$backup_key"
|
||||
done
|
||||
|
||||
echo "==> Verifying the new key can decrypt everything..."
|
||||
for f in "${secrets_files[@]}"; do
|
||||
if ! sops_can_decrypt "$new_key_file" "$f"; then
|
||||
echo "ERROR: new key failed to decrypt $(basename "$f") after rotation -- investigate before committing." >&2
|
||||
exit 1
|
||||
fi
|
||||
echo " OK: $(basename "$f")"
|
||||
done
|
||||
|
||||
cat <<EOF
|
||||
|
||||
Done. .sops.yaml's &admin key is now:
|
||||
${new_pub}
|
||||
|
||||
The old key (${old_pub}) can no longer decrypt any secrets/*.yaml
|
||||
re-encrypted above.
|
||||
|
||||
Review the diff, then commit:
|
||||
git add .sops.yaml secrets/*.yaml
|
||||
git commit -m "Rotate sops admin age key"
|
||||
EOF
|
||||
@@ -1,110 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Detects and fixes drift between the ed25519 SSH host key nix-cache is
|
||||
# actually serving right now and vars.nixCacheHostKey (variables.nix) --
|
||||
# the value modules/nix-cache/remote-builder-client.nix bakes into every
|
||||
# client's declarative programs.ssh.knownHosts, and
|
||||
# scripts/proxmox/configure-nix-cache-client.sh hardcodes as its own
|
||||
# default for non-NixOS clients.
|
||||
#
|
||||
# This value has no automatic source of truth: nix-cache's host key is
|
||||
# generated once (first boot / container recreate) and never touches this
|
||||
# repo again unless someone remembers to update it by hand afterwards. It
|
||||
# drifted silently once already -- confirmed live: variables.nix recorded
|
||||
# a key that no longer matched what nix-cache actually presented, which
|
||||
# would fail every real client's SSH host-key verification for
|
||||
# distributed builds without ever producing an obvious error pointing
|
||||
# back here (a client just sees "Host key verification failed" against
|
||||
# *some* key, with no hint that the trusted value itself was stale).
|
||||
#
|
||||
# codex-maintenance.sh runs this in --check mode on every invocation so
|
||||
# that drift surfaces as a warning instead of a future debugging session.
|
||||
#
|
||||
# Usage:
|
||||
# scripts/secrets/sync-nix-cache-host-key.sh [--check] [--dry-run] [--host <name>]
|
||||
#
|
||||
# --check Only report drift (exit 1 if found, 2 if nix-cache is
|
||||
# unreachable); never writes. For CI/maintenance use.
|
||||
# --dry-run Show what would change; never writes.
|
||||
# --host Override the hostname to scan (default: variables.nix's
|
||||
# nixCacheHost / env.sh's NIX_CACHE_HOST).
|
||||
set -euo pipefail
|
||||
|
||||
repo_root="$(cd "$(dirname "$0")/../.." && pwd)"
|
||||
variables_nix="${repo_root}/variables.nix"
|
||||
client_script="${repo_root}/scripts/proxmox/configure-nix-cache-client.sh"
|
||||
|
||||
# shellcheck source=../env.sh
|
||||
source "${repo_root}/scripts/env.sh"
|
||||
|
||||
check_only=0
|
||||
dry_run=0
|
||||
host="${NIX_CACHE_HOST}"
|
||||
|
||||
while [[ $# -gt 0 ]]; do
|
||||
case "$1" in
|
||||
--check) check_only=1; shift ;;
|
||||
--dry-run) dry_run=1; shift ;;
|
||||
--host)
|
||||
host="${2:?--host requires a hostname}"
|
||||
shift 2
|
||||
;;
|
||||
-h|--help)
|
||||
sed -n '2,23p' "$0"
|
||||
exit 0
|
||||
;;
|
||||
*)
|
||||
echo "ERROR: unknown argument: $1" >&2
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
done
|
||||
|
||||
current_value="$(grep -oE 'nixCacheHostKey = "[^"]+"' "$variables_nix" | sed -E 's/nixCacheHostKey = "(.*)"/\1/')"
|
||||
if [[ -z "$current_value" ]]; then
|
||||
echo "ERROR: couldn't find nixCacheHostKey in $variables_nix" >&2
|
||||
exit 1
|
||||
fi
|
||||
current_type_blob="$(awk '{print $1, $2}' <<<"$current_value")"
|
||||
current_label="$(awk '{print $3}' <<<"$current_value")"
|
||||
|
||||
echo "Scanning ${host} for its current ed25519 SSH host key..."
|
||||
nix_extra_opts
|
||||
scanned="$(nix-shell "${NIX_OPTS[@]}" -p openssh --run "ssh-keyscan -t ed25519 -T 5 '${host}'" 2>/dev/null | grep -v '^#' | head -1 || true)"
|
||||
if [[ -z "$scanned" ]]; then
|
||||
echo "ERROR: couldn't reach ${host} (or got no ed25519 host key back) via ssh-keyscan." >&2
|
||||
exit 2
|
||||
fi
|
||||
scanned_type_blob="$(awk '{print $2, $3}' <<<"$scanned")"
|
||||
|
||||
if [[ "$current_type_blob" == "$scanned_type_blob" ]]; then
|
||||
echo "Up to date: ${host}'s host key matches variables.nix's nixCacheHostKey."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
echo "DRIFT DETECTED:"
|
||||
echo " variables.nix has: $current_type_blob"
|
||||
echo " ${host} is now: $scanned_type_blob"
|
||||
|
||||
if [[ "$check_only" -eq 1 ]]; then
|
||||
echo
|
||||
echo "Run 'scripts/secrets/sync-nix-cache-host-key.sh' (no flags) to fix." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
new_value="${scanned_type_blob} ${current_label}"
|
||||
|
||||
if [[ "$dry_run" -eq 1 ]]; then
|
||||
echo "(--dry-run: would update variables.nix and ${client_script##*/} to:)"
|
||||
echo " $new_value"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
sed -i "s|nixCacheHostKey = \"[^\"]*\"|nixCacheHostKey = \"${new_value}\"|" "$variables_nix"
|
||||
sed -i "s|NIX_CACHE_HOST_KEY:=[^}]*}|NIX_CACHE_HOST_KEY:=${new_value}}|" "$client_script"
|
||||
|
||||
echo "Updated variables.nix and ${client_script##*/} to:"
|
||||
echo " $new_value"
|
||||
echo
|
||||
echo "This only takes effect on already-deployed NixOS clients after their"
|
||||
echo "next rebuild (programs.ssh.knownHosts is declarative). Review with"
|
||||
echo "'git diff', then run 'bash scripts/codex-maintenance.sh' before committing."
|
||||
@@ -11,35 +11,26 @@
|
||||
# sync-host-keys.sh --regenerate-all-keys Remove and freshly regenerate
|
||||
# every locally-managed key.
|
||||
#
|
||||
# "Generate/register" is idempotent and additive only: an existing clan
|
||||
# var is never overwritten, and .sops.yaml only ever gains an anchor/alias
|
||||
# it doesn't already have -- safe to re-run any time, e.g. right after
|
||||
# adding a new host to flake.nix.
|
||||
# "Generate/register" is idempotent and additive only: an existing
|
||||
# host-keys/ file is never touched, and .sops.yaml only ever gains an
|
||||
# anchor/alias it doesn't already have -- safe to re-run any time, e.g.
|
||||
# right after adding a new host to flake.nix.
|
||||
#
|
||||
# --remove and --regenerate-all-keys only ever operate on anchors that
|
||||
# have a corresponding clan var (vars/per-machine/<name>/openssh/) or
|
||||
# host-keys/ file. Anchors without either (&admin) are never listed,
|
||||
# removed, or regenerated -- this tooling only ever touches keys it itself
|
||||
# manages.
|
||||
# --remove and --regenerate-all-keys only ever operate on anchors that have
|
||||
# a corresponding host-keys/<name>_ssh_host_ed25519_key file. Anchors
|
||||
# without one (&admin, and any anchor for an already-deployed host whose
|
||||
# real /etc/ssh key was registered by hand, e.g. &docker/&server/&nix-cache
|
||||
# today) are never listed, removed, or regenerated -- this tooling only
|
||||
# ever touches keys it itself manages.
|
||||
set -euo pipefail
|
||||
|
||||
repo_root="$(cd "$(dirname "$0")/../.." && pwd)"
|
||||
repo_root="$(cd "$(dirname "$0")/.." && pwd)"
|
||||
sops_yaml="${repo_root}/.sops.yaml"
|
||||
keydir="${repo_root}/host-keys"
|
||||
editor="${repo_root}/scripts/lib/sync-host-keys-edit-sops.py"
|
||||
|
||||
# shellcheck source=../env.sh
|
||||
# shellcheck source=env.sh
|
||||
source "${repo_root}/scripts/env.sh"
|
||||
# shellcheck source=../lib/nix-eval.sh
|
||||
source "${repo_root}/scripts/lib/nix-eval.sh"
|
||||
# shellcheck source=../lib/ssh-host-keys.sh
|
||||
source "${repo_root}/scripts/lib/ssh-host-keys.sh"
|
||||
# shellcheck source=../lib/sops-age.sh
|
||||
source "${repo_root}/scripts/lib/sops-age.sh"
|
||||
# shellcheck source=../lib/confirm.sh
|
||||
source "${repo_root}/scripts/lib/confirm.sh"
|
||||
# shellcheck source=../lib/clan-vars.sh
|
||||
source "${repo_root}/scripts/lib/clan-vars.sh"
|
||||
|
||||
mkdir -p "$keydir"
|
||||
|
||||
@@ -55,14 +46,13 @@ Usage: $0 --all [--dry-run]
|
||||
<flake-target> Same, for just one target (e.g. lxc-server).
|
||||
Reports if it already has one.
|
||||
--remove Interactively pick one locally-managed key to
|
||||
remove from .sops.yaml and vars/per-machine/
|
||||
(or host-keys/ for legacy keys).
|
||||
remove from .sops.yaml and host-keys/.
|
||||
--regenerate-all-keys Remove every locally-managed key and generate
|
||||
fresh clan-var replacements for every current
|
||||
flake target. Destructive -- requires typed
|
||||
fresh replacements for every current flake
|
||||
target. Destructive -- requires typed
|
||||
confirmation.
|
||||
--dry-run Combine with any of the above: print what would
|
||||
change (clan vars, .sops.yaml anchors and
|
||||
change (host-keys/ files, .sops.yaml anchors and
|
||||
key_groups, which secrets/*.yaml would be
|
||||
re-encrypted) without touching anything. No keys
|
||||
generated, no files written, no sops calls,
|
||||
@@ -84,11 +74,7 @@ ensure_admin_decrypt_key() {
|
||||
return
|
||||
fi
|
||||
|
||||
local key_file="$DEFAULT_SOPS_AGE_KEY_FILE"
|
||||
# Expand a leading ~ that survived variable substitution without tilde
|
||||
# expansion (happens when SOPS_AGE_KEY_FILE or XDG_CONFIG_HOME is set with
|
||||
# a literal ~ in the caller's environment).
|
||||
key_file="${key_file/#~\//$HOME/}"
|
||||
local key_file="${SOPS_AGE_KEY_FILE:-${XDG_CONFIG_HOME:-$HOME/.config}/sops/age/keys.txt}"
|
||||
|
||||
if [[ -s "$key_file" ]]; then
|
||||
echo "Found existing sops age key at ${key_file}."
|
||||
@@ -97,46 +83,54 @@ ensure_admin_decrypt_key() {
|
||||
|
||||
if [[ "$dry_run" -eq 1 ]]; then
|
||||
echo "[dry-run] No sops age decryption key found (checked \$SOPS_AGE_KEY, \$SOPS_AGE_KEY_FILE, ${key_file})."
|
||||
echo "[dry-run] Continuing dry run without one -- any 'would re-encrypt' output below"
|
||||
echo "[dry-run] couldn't actually run for real until a key is present."
|
||||
echo "[dry-run] Would generate a new one here -- continuing the dry run without one; any"
|
||||
echo "[dry-run] 'would re-encrypt' output below couldn't actually run for real yet."
|
||||
return
|
||||
fi
|
||||
|
||||
cat >&2 <<EOF
|
||||
No sops age decryption key found (checked \$SOPS_AGE_KEY, \$SOPS_AGE_KEY_FILE, ${key_file}).
|
||||
echo "No sops age decryption key found (checked \$SOPS_AGE_KEY, \$SOPS_AGE_KEY_FILE, ${key_file})."
|
||||
echo "Generating a new one at ${key_file}..."
|
||||
mkdir -p "$(dirname "$key_file")"
|
||||
nix-shell "${NIX_OPTS[@]}" -p age --run "age-keygen -o '${key_file}'" 2>&1 | grep -v "^Public key:" || true
|
||||
local new_pub
|
||||
new_pub="$(nix-shell "${NIX_OPTS[@]}" -p age --run "age-keygen -y '${key_file}'")"
|
||||
|
||||
Place your admin age private key at ${key_file}, or set SOPS_AGE_KEY (inline
|
||||
key) or SOPS_AGE_KEY_FILE (path to a different key file) and re-run.
|
||||
cat <<EOF
|
||||
|
||||
If the key is truly missing (not just mislocated), this is a manual recovery
|
||||
situation -- generating a brand-new admin key won't help, since it cannot
|
||||
decrypt anything already encrypted for the old one. Each secrets/*.yaml is
|
||||
also encrypted for its respective host key(s), so a running deployed host can
|
||||
still decrypt what it needs -- but the admin key is required for re-encryption
|
||||
(e.g. adding new recipients via sops updatekeys).
|
||||
A brand-new age key was just generated -- it cannot decrypt anything that
|
||||
already exists in secrets/*.yaml, since nothing was ever encrypted for it.
|
||||
That trust can't be bootstrapped automatically (nobody can decrypt a file
|
||||
for a recipient that didn't exist when it was last encrypted).
|
||||
|
||||
To actually use this key:
|
||||
1. Have someone who currently CAN decrypt replace the &admin entry in
|
||||
.sops.yaml with this public key:
|
||||
${new_pub}
|
||||
2. They re-encrypt every secrets/*.yaml:
|
||||
sops updatekeys --yes secrets/common.yaml
|
||||
sops updatekeys --yes secrets/nix-cache.yaml
|
||||
sops updatekeys --yes secrets/server.yaml
|
||||
3. Re-run this script.
|
||||
|
||||
Exiting without making any other changes.
|
||||
EOF
|
||||
exit 1
|
||||
}
|
||||
|
||||
discover_targets() {
|
||||
nix eval --json --no-use-registries --no-accept-flake-config \
|
||||
"${repo_root}#nixosConfigurations" --apply builtins.attrNames \
|
||||
| jq -r '.[] | select(. != "installer")'
|
||||
# installer is the one nixosConfigurations target that doesn't import
|
||||
# sops-nix at all (see CLAUDE.md's "Security Notes" -- hardcoded login
|
||||
# password instead) -- config.sops.secrets doesn't exist for it.
|
||||
list_flake_targets "$repo_root" | grep -v '^installer$'
|
||||
}
|
||||
|
||||
locally_managed_hosts() {
|
||||
{
|
||||
for f in "$keydir"/*_ssh_host_ed25519_key.pub; do
|
||||
[[ -e "$f" ]] || continue
|
||||
basename "$f" _ssh_host_ed25519_key.pub
|
||||
done
|
||||
local d
|
||||
for d in "${repo_root}/vars/per-machine"/*/openssh/ssh_host_ed25519_key/secret; do
|
||||
[[ -f "$d" ]] || continue
|
||||
basename "$(dirname "$(dirname "$(dirname "$d")")")"
|
||||
done
|
||||
} | sort -u
|
||||
for f in "$keydir"/*_ssh_host_ed25519_key.pub; do
|
||||
[[ -e "$f" ]] || continue
|
||||
basename "$f" _ssh_host_ed25519_key.pub
|
||||
done
|
||||
}
|
||||
|
||||
add_keys_json="[]"
|
||||
@@ -146,15 +140,13 @@ dry_run=0
|
||||
queue_host_sync() {
|
||||
local host="$1"
|
||||
local keyfile="${keydir}/${host}_ssh_host_ed25519_key"
|
||||
local has_local_key=0 has_clan_key=0 has_anchor=0
|
||||
local has_local_key=0 has_anchor=0
|
||||
[[ -f "$keyfile" ]] && has_local_key=1
|
||||
clan_ssh_key_exists "$host" "$repo_root" && has_clan_key=1
|
||||
grep -qE "^ - &${host} age1" "$sops_yaml" && has_anchor=1
|
||||
|
||||
if [[ "$has_local_key" -eq 0 && "$has_clan_key" -eq 0 && "$has_anchor" -eq 1 ]]; then
|
||||
if [[ "$has_local_key" -eq 0 && "$has_anchor" -eq 1 ]]; then
|
||||
echo "SKIP ${host}: .sops.yaml already has an &${host} anchor, but"
|
||||
echo " neither host-keys/${host}_ssh_host_ed25519_key nor"
|
||||
echo " vars/per-machine/${host}/openssh/ exist locally."
|
||||
echo " host-keys/${host}_ssh_host_ed25519_key is missing locally."
|
||||
echo " Not generating a replacement -- it wouldn't match whatever's"
|
||||
echo " already registered (and possibly deployed). Remove the"
|
||||
echo " &${host} line from .sops.yaml first if you really want a"
|
||||
@@ -162,28 +154,23 @@ queue_host_sync() {
|
||||
return 1
|
||||
fi
|
||||
|
||||
if [[ "$has_local_key" -eq 0 && "$has_clan_key" -eq 0 ]]; then
|
||||
if [[ "$has_local_key" -eq 0 ]]; then
|
||||
if [[ "$dry_run" -eq 1 ]]; then
|
||||
echo "[dry-run] ${host}: would generate host key via clan vars"
|
||||
echo "[dry-run] ${host}: would generate host key"
|
||||
else
|
||||
echo "==> ${host}: generating host key via clan vars"
|
||||
clan_generate_ssh_key "$host" "$repo_root"
|
||||
has_clan_key=1
|
||||
echo "==> ${host}: generating host key"
|
||||
nix-shell "${NIX_OPTS[@]}" -p openssh --run "ssh-keygen -t ed25519 -N '' -C '${host}' -f '${keyfile}'" >/dev/null
|
||||
fi
|
||||
elif [[ "$has_clan_key" -eq 1 ]]; then
|
||||
echo "==> ${host}: clan-managed SSH host key already present"
|
||||
else
|
||||
echo "==> ${host}: host key already present (host-keys/)"
|
||||
echo "==> ${host}: host key already present"
|
||||
fi
|
||||
|
||||
if [[ "$has_anchor" -eq 0 ]]; then
|
||||
local age_pub
|
||||
if [[ "$dry_run" -eq 1 ]]; then
|
||||
age_pub="dry-run-placeholder-not-a-real-key"
|
||||
elif [[ "$has_clan_key" -eq 1 ]]; then
|
||||
age_pub="$(ssh_pubkey_to_age "$(clan_ssh_pubkey_path "$host" "$repo_root")")"
|
||||
else
|
||||
age_pub="$(ssh_pubkey_to_age "${keyfile}.pub")"
|
||||
age_pub="$(nix-shell "${NIX_OPTS[@]}" -p ssh-to-age --run "ssh-to-age -i '${keyfile}.pub'")"
|
||||
fi
|
||||
add_keys_json="$(jq --arg host "$host" --arg key "$age_pub" \
|
||||
'. + [{host: $host, age_key: $key}]' <<<"$add_keys_json")"
|
||||
@@ -250,7 +237,7 @@ apply_edit_plan() {
|
||||
while IFS= read -r basename; do
|
||||
[[ -z "$basename" ]] && continue
|
||||
echo "==> secrets/${basename}"
|
||||
sops_updatekeys "${repo_root}/secrets/${basename}"
|
||||
nix-shell "${NIX_OPTS[@]}" -p sops --run "sops updatekeys --yes '${repo_root}/secrets/${basename}'"
|
||||
done <<<"$changed"
|
||||
fi
|
||||
fi
|
||||
@@ -306,7 +293,7 @@ cmd_remove() {
|
||||
local hosts
|
||||
mapfile -t hosts < <(locally_managed_hosts)
|
||||
if [[ "${#hosts[@]}" -eq 0 ]]; then
|
||||
echo "No locally-managed keys found (checked host-keys/ and vars/per-machine/) -- nothing to remove."
|
||||
echo "No locally-managed keys in host-keys/ -- nothing to remove."
|
||||
return
|
||||
fi
|
||||
|
||||
@@ -315,9 +302,7 @@ cmd_remove() {
|
||||
for host in "${hosts[@]}"; do
|
||||
local registered="not registered in .sops.yaml"
|
||||
grep -qE "^ - &${host} age1" "$sops_yaml" && registered="registered in .sops.yaml"
|
||||
local where="host-keys/"
|
||||
clan_ssh_key_exists "$host" "$repo_root" && where="clan-vars"
|
||||
printf ' %d) %s [%s, %s]\n' "$i" "$host" "$where" "$registered"
|
||||
printf ' %d) %s (%s)\n' "$i" "$host" "$registered"
|
||||
i=$((i + 1))
|
||||
done
|
||||
|
||||
@@ -334,7 +319,7 @@ cmd_remove() {
|
||||
local target="${hosts[$((choice - 1))]}"
|
||||
|
||||
if [[ "$dry_run" -ne 1 ]]; then
|
||||
read -rp "Really remove '${target}'? Its key files will be deleted and it will lose access to every secrets file it can currently decrypt. (y/N): " confirm
|
||||
read -rp "Really remove '${target}'? Its host-keys/ files will be deleted and it will lose access to every secrets file it can currently decrypt. (y/N): " confirm
|
||||
if [[ ! "$confirm" =~ ^[Yy]$ ]]; then
|
||||
echo "Cancelled."
|
||||
return
|
||||
@@ -347,13 +332,11 @@ cmd_remove() {
|
||||
apply_edit_plan "$plan"
|
||||
|
||||
if [[ "$dry_run" -eq 1 ]]; then
|
||||
echo "[dry-run] would delete host-keys/${target}_ssh_host_ed25519_key(.pub) if present."
|
||||
echo "[dry-run] would delete vars/per-machine/${target}/openssh/ if present."
|
||||
echo "[dry-run] would delete host-keys/${target}_ssh_host_ed25519_key(.pub)."
|
||||
echo "[dry-run] Nothing was changed. Re-run without --dry-run to apply this."
|
||||
else
|
||||
rm -f "${keydir}/${target}_ssh_host_ed25519_key" "${keydir}/${target}_ssh_host_ed25519_key.pub"
|
||||
rm -rf "${repo_root}/vars/per-machine/${target}/openssh"
|
||||
echo "Removed key for ${target} (host-keys/ and/or vars/per-machine/ as applicable)."
|
||||
echo "Removed host-keys/${target}_ssh_host_ed25519_key(.pub)."
|
||||
echo
|
||||
echo "Review the diff, then commit and push."
|
||||
fi
|
||||
@@ -363,21 +346,19 @@ cmd_regenerate_all() {
|
||||
local hosts
|
||||
mapfile -t hosts < <(locally_managed_hosts)
|
||||
if [[ "${#hosts[@]}" -eq 0 ]]; then
|
||||
echo "No locally-managed keys found (checked host-keys/ and vars/per-machine/) -- nothing to regenerate."
|
||||
echo "No locally-managed keys in host-keys/ -- nothing to regenerate."
|
||||
return
|
||||
fi
|
||||
|
||||
echo "This will remove and freshly regenerate ALL locally-managed keys:"
|
||||
printf ' %s\n' "${hosts[@]}"
|
||||
echo
|
||||
echo "After regenerating, each host needs its new key before it can decrypt secrets:"
|
||||
echo " • Already running: push the key before rebuilding:"
|
||||
echo " scripts/secrets/push-host-keys.sh --all"
|
||||
echo " • Not yet deployed: rebuild the install image with the new keys baked in"
|
||||
echo " (see docs/auto-installer.md)."
|
||||
echo "Every host above will need its new key baked into a rebuilt install"
|
||||
echo "image/tarball before it can decrypt secrets again."
|
||||
|
||||
if [[ "$dry_run" -ne 1 ]]; then
|
||||
if ! confirm_typed "REGENERATE" "Type REGENERATE to confirm: "; then
|
||||
read -rp "Type REGENERATE to confirm: " confirm
|
||||
if [[ "$confirm" != "REGENERATE" ]]; then
|
||||
echo "Cancelled."
|
||||
return
|
||||
fi
|
||||
@@ -392,8 +373,8 @@ cmd_regenerate_all() {
|
||||
apply_edit_plan "$plan"
|
||||
|
||||
if [[ "$dry_run" -eq 1 ]]; then
|
||||
echo "[dry-run] would delete ${#hosts[@]} key pair(s) from host-keys/ and/or vars/per-machine/."
|
||||
echo "[dry-run] would then generate fresh clan vars replacements for the same hosts"
|
||||
echo "[dry-run] would delete ${#hosts[@]} host-keys/ file pair(s)."
|
||||
echo "[dry-run] would then generate fresh replacements for the same hosts"
|
||||
echo "[dry-run] (not simulated further here -- run without --dry-run, or"
|
||||
echo "[dry-run] preview a specific target with: $0 <target> --dry-run)."
|
||||
echo
|
||||
@@ -405,23 +386,12 @@ cmd_regenerate_all() {
|
||||
local host
|
||||
for host in "${hosts[@]}"; do
|
||||
rm -f "${keydir}/${host}_ssh_host_ed25519_key" "${keydir}/${host}_ssh_host_ed25519_key.pub"
|
||||
rm -rf "${repo_root}/vars/per-machine/${host}/openssh"
|
||||
done
|
||||
echo "Removed ${#hosts[@]} key pair(s)."
|
||||
echo "Removed ${#hosts[@]} host-keys/ file pair(s)."
|
||||
|
||||
echo
|
||||
echo "Regenerating fresh keys for every current flake target..."
|
||||
cmd_all
|
||||
|
||||
echo
|
||||
echo "Next steps:"
|
||||
echo " 1. Commit and push .sops.yaml + secrets/ so the remote flake is current."
|
||||
echo " 2. Push the new host key to each already-running managed host:"
|
||||
echo " scripts/secrets/push-host-keys.sh --all"
|
||||
echo " (this also prompts to commit/push if step 1 wasn't done yet)"
|
||||
echo " 3. Run nixos-rebuild switch on each updated host."
|
||||
echo " 4. For hosts not yet deployed, rebuild the install image (see"
|
||||
echo " docs/auto-installer.md)."
|
||||
}
|
||||
|
||||
main() {
|
||||
+52
-205
@@ -1,234 +1,81 @@
|
||||
root-hashedPassword: ENC[AES256_GCM,data:Kp0nOZI7vDoLhJHiOJBwJn0rQZ5yhnwapGnAcA+qh8vlDETtFs/iQdetF/2ZxmANf62SviTNd+Ag0q5JIF1996x7onZGXqxgSMCuVzZLBdUlsO5IR0BslWWz47khYGTe4WkUg4NB1itBfQ==,iv:5Sra5vJ79V8hxQT3g9qJ+dOj2W2sumIhqpitqnHjJdk=,tag:3Igu0+8GeUZHqS3fKUVwog==,type:str]
|
||||
nixos-hashedPassword: ENC[AES256_GCM,data:pT7tVRN6X4a+DNUgB7fIUUE3CbnetkjxmoSL1PxSU+ktsFU+fB0mEvJjA1uujsGH5Rcztg7YM815+M0Z67ILmHaXbza5DtFacrqhi4/b277xly0SHRX4yOvBwQh6mJG1jn/0O/wvUUIYdw==,iv:bp2nfhC8nFbk6o5iWDAugvbzu7J/a1xayFnBEtkhNpE=,tag:HqWgkIpSrSM/K9OK2WO+VQ==,type:str]
|
||||
nix-github-token: ENC[AES256_GCM,data:k1vYz7SqVhzpWa6jTL6NUD8lKOCpHCgTm+HT4IcnbzbSTUZP/bJUYw==,iv:UqAULZnr/4+VcioUDfTwvOSuwM8K9JgGhiApvYQPyoc=,tag:1LKHXhWAO/AHPDIZFBb04A==,type:str]
|
||||
nix-github-token: ENC[AES256_GCM,data:OfNRGJg16Ede6EilWUetCs9za+xk5/Lsa3SpVajsqz8PMdA1xQNeCWdX7ZAMdijHClpBhU6ETFGsXvt41O9aORS951uijeGSW7/NH35/bnPISrKdYeBx/+xEiqwH,iv:QGU3v7xOy89uzRTCb1U9ICyJ8XYIpXrUsDt12aL3g2Y=,tag:Bde2wcWNv8H4WLxSEUAodg==,type:str]
|
||||
sops:
|
||||
age:
|
||||
- enc: |
|
||||
-----BEGIN AGE ENCRYPTED FILE-----
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBJR3NKa2kxZGRHd3RWdWRR
|
||||
WDkreDcxZzE1RVZ5NFFUN1pLUktzVHZSUVd3CmJkMzVsRVVraGZWVEZLWkQydVJI
|
||||
NGF5Ui9ranRZU1lqS2RKZzRmWGJVYmMKLS0tIGdIRlRMOFBDMnNnT3hZdjcvUVA3
|
||||
cUxXa0tjWWk3K0ZTNUx2OU9MRHpobzQKMZLQH+z8o27s1bAXyJI8HD8jHnU5JaZV
|
||||
WLHwwptGYz7pYyCkWc25IkA0nR3KR4lfHT2o5eGn2BBivpI5o/QQMw==
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBpWjNSdEdZbUUzamswa08w
|
||||
aCtzSHB0bFVZMnYxTkpuM1psdVYzWW55SzMwCkliMWVOUlBqRG5wOGZjQVg4MkFz
|
||||
NEVkMXdkTjhWRlZmVGlzZElid2pUMXMKLS0tIGtHUmRCNXNhVmloUHYzQnE5YlBS
|
||||
YnVSQjJlT3JnQ1RNMm9xV2xKOGRZUDAKc4VTl9NEI9Rv8+4J3JTeHTt2h8Dr2IJv
|
||||
tfvoNJQM/w6RAJWNTkaDmzZa9OnUW+grDlBQKlDuAnr6fZmuNTH2hQ==
|
||||
-----END AGE ENCRYPTED FILE-----
|
||||
recipient: age1njap586hc0q43kr03g6c8eqhdsmk8zcafkl3f83xwlc2gqhlmfgs4tmwad
|
||||
recipient: age10nd382a9klsn2mrs60emdtsxe43pht3a0m9p29phfrhy0wfyt3vsq9r667
|
||||
- enc: |
|
||||
-----BEGIN AGE ENCRYPTED FILE-----
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBpOVVzNDZvcE1Kb1Zua2pF
|
||||
cjllRWZkTDZUb0JtVW84Z01OOHF0Q3JsZW1RCmswV0lxUW02NEZzY3lWeHNXQ1Ez
|
||||
K0VILzFhemZjTTFLbEpVZVdJY21HTTAKLS0tIHVqZ2E5aU1NVk10bEJ5UXkvMVFK
|
||||
dXdSWEx2OXBSZG9aM0VGNVQzTjVRUkUKYgZjGG7I0ea9I+gG4Ah1VSkiONNAHdDQ
|
||||
5zG9LnKQZ5faY4tWsX7JHFWDX2Y83mXbHV6jMCutAfQG2zt/5psTMw==
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBOR25UN1c0aE5SYWphbU0y
|
||||
SHRTZ1B0WC9NU3Z5VHpzTXpLSUxHTDI1ZFE0CnFpYUN1eGZQejJMblZPd1ROeUth
|
||||
dklZYVVNa1ZNZ1d4dW9vMCsvQWp1RkEKLS0tIERxakx5L0JrQitib1EyNDRMbDQ5
|
||||
Z3hDWUFEazdxczVhaHJYK3VZeEJSSDgKkw9T4ZuT+VHIF4WopqRHt8vW30kOysJ3
|
||||
vOq6EZ3Fqkgmoxm69Zp2gFnuE9GZIBy3VPQVLU2k6dZGJ3IvmLYeBA==
|
||||
-----END AGE ENCRYPTED FILE-----
|
||||
recipient: age19m0m7vdfg86yqy8l5mmle5jdd0unrn3f55t232w8h5ey42cqw34sfpt32n
|
||||
recipient: age19gfn2yedg76dmztm4hncr7vf3r3c9j0qpt4rap7y7gersjk4m3ks2lhd0e
|
||||
- enc: |
|
||||
-----BEGIN AGE ENCRYPTED FILE-----
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSAxS3VuaHlaVGJObVNPb3Rh
|
||||
UEdBMk9meDQvWi85NVRHR3dXSlhkcFZVa1RBClZDR1MxOUdwVjhzSjFaWVFWelVX
|
||||
cUgzMWNrUzl3L1R6elc1ZTRGamxDU2MKLS0tIDFwUUJFQ0hIUWNTVmFwdGt0czZi
|
||||
V3VGSndCT0tQb2Nzd2w2TWxBbEFBeUEK8kQQVqISb3h0snOtqM0w/mhWEpjIxlhk
|
||||
N84QzgxhD40tvutfO+57HsfOIqjKi4Yhdmbd8+iW9GzGnrbfrDuVGw==
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSA0THJzMFBTTCtDMWRmZ25M
|
||||
dU55OVhBb0trWWRUNlArTnEzRjhiYngvRENNCjF4d0M5NlYyQW50TTdMRXpuUjRr
|
||||
M1NwV05JOHV6T2cxT2FheVpuZ0w2T0kKLS0tIFBxdlVpVEoxOUpSWjk1ejRsK1NM
|
||||
V1UwTU1scG91L2FIemtwSW5JbFlmeG8K/1WIlaIidy3x3ptoRpS/DG88064LQ6Mq
|
||||
GbfB0jfq5PILDQMMuZu5oIBY31SxwnhZ02Ns7gA67kgNIRSCmk9WyQ==
|
||||
-----END AGE ENCRYPTED FILE-----
|
||||
recipient: age1rrxqea6q6pn39sw8y5te63h2py8jgjl9v0jyper86w3ggtn67upqg3ah39
|
||||
recipient: age1ll6hj5ggruetgjwjfnplpn5xtq35uhlcdflksx3xmnjm6s3uad9sz70jkf
|
||||
- enc: |
|
||||
-----BEGIN AGE ENCRYPTED FILE-----
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBoYXBpZlpjT2oxMURTcW9t
|
||||
NTdNWkMyT0d0SVBCSFBqVC9vNkxYUUk5Wnd3CmlyNUhIa0FmR0lwSFEwK2QzRjlO
|
||||
ajNKMitzZGlDaXpRS1pNNDhhaFdPYjgKLS0tIFVtSjNXSWdlQTVwOEUzcjdEVExu
|
||||
aklINzU2QUFsZmxPaUI5aFR2Z3V0bXMK/rDuNYW2r022tHjuk4KqIEOxqvjQsBIg
|
||||
D2lG4ih5h3idm++gnhCSlP/9tKSaVHoq+BpunsmFObrrWzZRhKI6jw==
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBmRURQVGw1a1E1MkIzV2pa
|
||||
RG5HSkM0c0huUGVWcnFZOGlacGlGOHFMTVFrCm5OS09HRFc3TGVUYmtzTStxL0Q5
|
||||
N2hRMEcxUE9MTmZXL0wvME5EZXF1Z00KLS0tIHdHMVVHcTZzMmdXU0s4QlVqSS9Y
|
||||
ZUVmcWhPaURIUFJGR0V4bUZwKzM1bm8KlvGMNEClbLlfvJqNQHhd0dI4ihShLChF
|
||||
GI/fydgrBruw3Otv6KLZu3CBC7iNcKlvZxz+YGD2qbicmyQ5hAhDSQ==
|
||||
-----END AGE ENCRYPTED FILE-----
|
||||
recipient: age1adur9g330gua4l6ndk8cqjg35qc8yxwgme6wrl2hpylcc7vxm38q05ejuy
|
||||
recipient: age120le4a5l8dh3lyfgvmj3d9ksmej6ajs5mer5y7r0vfg3x9fn69dqf8xgzu
|
||||
- enc: |
|
||||
-----BEGIN AGE ENCRYPTED FILE-----
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSArbkhEaEhoUk1SUHNjVm5C
|
||||
Q2svcmpselJ1eU5qMVJKV2JWV1J0eUpSWlV3CnhIWDF2NXFMM0VMbDQyeHdkNDJV
|
||||
aForNXJ0UFd0aXFGZzZhekNTdUd2U3cKLS0tIFljclNTa2I5RHJTUU1yMm5XakVo
|
||||
amlHVmZzSDZRTndFSEpjQlVOUTVhRVEKz4Tx0PceUCxw8MsREB2HRPDyC/lIskhs
|
||||
/HWmseMlnsoEMu8E1OIO0TpWY3qowHrdFFr4njpxMJUTXusGvRK8Qw==
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSA5bWpFenBlQldna3RhSFpr
|
||||
bEJBaHJDMzM3OHlqTmozcWU0VDM1bTFUWFZzCnZtNHZjZ1U1RzNkUlBHZFozWXdt
|
||||
VVkxQjMvTDJtbFZnclpkUEd4TEVmNTAKLS0tIFlhV2ZSSzJLRVNoMmFyVktDOElR
|
||||
YUxqZUFoY1ZWeGlldGplMjVQa1A5aUUKWelY6yO7Mr6dRvj4MVMbq/Z9JgrAnahz
|
||||
BDhHqObzJrOCtfDCTWiYuP+0yvIFWItMWhGSMw9MwwivvwnrEa+ZuQ==
|
||||
-----END AGE ENCRYPTED FILE-----
|
||||
recipient: age17e89ty6p0fw24daanen57wg8uald9s025t3wwxsw269svwpmgvrshfvfvt
|
||||
recipient: age1qz9d4ka4xgexujyd247s7lp737sulp5fhxl5d65fj2ykvc4j4edqrsdks8
|
||||
- enc: |
|
||||
-----BEGIN AGE ENCRYPTED FILE-----
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBXbm1jWW1DZDVNdnVZOE5E
|
||||
bTRLTzRhckVMNjV5VkMrcllLZk9NeVRiT0NrCkh3RDRicXFTQTNqSmw4VEQ0QlNt
|
||||
cGJNRlFTSjU5YWFXWmM1MjBjVGZSeVkKLS0tIHFPOERUTGVOVSs5U0VMalAwUXcx
|
||||
QTYvK2RScDErVENFdUhpYUVsejBYR0UKCNBN3n4q2X1YLturxpjDv4IlnLXoPtOM
|
||||
DqXMNEgGubagOBhfVFrOjpFxonP7JThSfzEJT/pmL+S7bCvgyosYXA==
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBiK2hKb2VaVnoyaEVYUHZZ
|
||||
OWVUSGtONGs5dEljTjJQdlhEcjNjdjViT2pVCkpaOHVZMlpXOFRveVlMeXZqWmoz
|
||||
ZHJRQTR2dmJQSEozeTRGMEdUdFlmZ2MKLS0tIERWS1RVdW1jQytBZzlkb3puNjhH
|
||||
ZjdlZmtzNXVOQ25DeCthUzhRRm1MT2cKaxc7zGm57iJFSeYc2IPqF4Eaxa44nR37
|
||||
pWZw+erG4F9AAZ2F047q+oLKe0B8FLSF54IbcXdQhitgGNR7B2HVeA==
|
||||
-----END AGE ENCRYPTED FILE-----
|
||||
recipient: age1hrx8qj02fj2ea6d4g9vqhyj9hl7fppkjqfdx2l37py3h6pdkr95s8n8rvs
|
||||
recipient: age120whqj96g26lsgy4udvgsn8dc9lumh8jeu3a564fx79rjr5lxffqmrljuu
|
||||
- enc: |
|
||||
-----BEGIN AGE ENCRYPTED FILE-----
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBXdDRCMlVyaTdPWFQ5VVhh
|
||||
UWMzUTBkZVYrSDRQWUpWLytDeHVmcVh0L0g0CjVIZE1jZHA0N3RtSmZKVThrcmt5
|
||||
N3VrenFGTS9LUDdZRHBiNDhORE4wa2sKLS0tIGdieDQxQ0lWZnVMZUZzSExKWEJU
|
||||
azl2Q01YdHZacDRSaElpZ0h3SG9xcGsK4s30qRC2eXbuKqPUHfRUJrE8FCMdz5EQ
|
||||
25UhdrspVBadt0G92hHV06Uwm/KKnG4Mi5crLKIMI+HAF+5Uxrh/xA==
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSA1ekhUTG5VOFErL3pFeWZM
|
||||
RG9NVnN6NFl3bzlFeTQzdHFtZmhwem04alQ0ClpJRStObERMZ0w2V0NhR1FSeW96
|
||||
M1d2V2NjUkUrLzN2ZVNSbGY4bll5WmsKLS0tIC94dVFQcXJ6d3pLU0VHNEFGR0ls
|
||||
a1Q2UmNuSjVMNG5XZGZKV1VmNHFPQXcKQJrZGw/9fPnXeFZ4omrkEgrzwplhwvRW
|
||||
i0FXuepoU353sR7enyL34qPoOdm05ivowuPKNzkq8D4i5AF6vGv+YA==
|
||||
-----END AGE ENCRYPTED FILE-----
|
||||
recipient: age1e7l8dusgmgfzd2cxrrzwepzjxt69hzqj4epee0cs27u6yg4kxcuqm34ncx
|
||||
recipient: age164px2a8e48ptsf9ngtan38aa6jls4jdl26mzrgzf6sn3vcvt49hqjrgr8w
|
||||
- enc: |
|
||||
-----BEGIN AGE ENCRYPTED FILE-----
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBwL2xrRnJyOWxRT01YbHI0
|
||||
OG8vU2xQaEJySkZkQnp6ck90dmkzQlV1RER3CmtGc3duTGRxZXhlbDlHd2hGUDdH
|
||||
QmxYTjBVbkVWRVlIcEhpaml2MlVzc0kKLS0tIFRWTnZGQm94VDhJK0J6SS9oa2FB
|
||||
aGpJUGFydUhvQlpFaURWSm9VS1AzM28KWvU5knwB/ViUrSxbP0zPR4iUE4LXxYi3
|
||||
qraWrbv3jXZwu4Sgv0H+/k6St/Xo5RU+nIqSOCKEy/kpACwk4BhMXg==
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBoZ2lHM2RBQ3lUK216dkcr
|
||||
Y3RrcXR3QTdFVjJRWlhpQWVFQ1cyZWNabkRFClAwRi9PSHF5ZWFSUzJuRXF3bU1R
|
||||
b0U5TFRZaFdmR1NMS3RRT3E3M2hUdE0KLS0tIEtrUy8ydkNyVHBiOVR6WEdjV2VN
|
||||
NVJHUVgwRkhxcmlwcFkrRlFwTEF6YVkKzXyJk0UnmUsvb+NzNVcf/gf7OEEt3P/K
|
||||
OIGxDrGfs/zNQgeKXNbQlQ4p4jOaybG8aCmX+A4qTk6/I8yY8LTJWg==
|
||||
-----END AGE ENCRYPTED FILE-----
|
||||
recipient: age1jcx3yajjhghn8qh8za3yeu8nxykzlg3p4nrv03vnfvzl0mzayg2qmg940e
|
||||
- enc: |
|
||||
-----BEGIN AGE ENCRYPTED FILE-----
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSB3K2tIY25TZCtReTQ2Zkc5
|
||||
d2w2aGZqYXNBSmZ2VVc5WWtZNDU4YlJpNm5VCkpGUkNOdGVSeXU1WG1YV2NQYjhu
|
||||
c2NKd0czSm5FUWxmNTdHVGVVWGlUMWcKLS0tIDJmRy9vbGJDNGxMVDM3b25TRVVI
|
||||
blpZTkZsTzNrdmFwUGQzR2RIOVJINlkKwa06bE6TmYtXPx0daHadbwL9/u1iYuGm
|
||||
n62kEwTKdyfncODl895qqjWkiA3JnbwQxzsaVDtEGfauzUph9DPY2g==
|
||||
-----END AGE ENCRYPTED FILE-----
|
||||
recipient: age1sweerhrga9yf8x6sv0apz4ed4g48rnlcq34rpv20t0rcelwgpgeqwvndzz
|
||||
- enc: |
|
||||
-----BEGIN AGE ENCRYPTED FILE-----
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSA1ZlVxYkZ4UFJmRVBHeExi
|
||||
OWZ0akNwc0dMenlHQ2dUSDlERGxSNzZtdFJJCnRYK2cwNHJRK0xtY0NnYlh1a1pG
|
||||
aTk0WURSYVdUY1ZTSUhBNVpZZUFyaUUKLS0tIHZKTFZmS2ZsNlZ6UjZUWmhNK0Q4
|
||||
RC9TREh3ZWZRTzc0WmFkQm5qVlZJcFkKUngIUXCVV1CKpUKfAlal1KjoJwk83mq5
|
||||
1MvU9mjx8Oq/suUlW4axFfAO99mUc9yCwb8TCMUZDhQp802v5FjBcw==
|
||||
-----END AGE ENCRYPTED FILE-----
|
||||
recipient: age1f7usptjx9rv4rxauasve200gxtdt9jkqhhdqstlf20wvlm7u75rsjfw50m
|
||||
- enc: |
|
||||
-----BEGIN AGE ENCRYPTED FILE-----
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBRM0hrc0I3UCtBOXJWUm1H
|
||||
MW5QWHAwV3lETFp2TGtPWVZHZ2s0ZHJmQ2w4CnBlQmlTVmEvRTlLMjVXMkRJeWIw
|
||||
QUR3aXpPUndocCt4aWUybndxK3RUN00KLS0tIDJrNkY0d1Jqby9pa0tXWVVuTER5
|
||||
YTlVNklTakFlaGQxdjJqRks5cUZoK3MKkV199oB7lPBzNYd30nSY0J6Cd09ViBA8
|
||||
uZWlmMGKlveBQ98qOt/vDTouRDxedASz0ZGg7+jxUXkUKbujkCYYXw==
|
||||
-----END AGE ENCRYPTED FILE-----
|
||||
recipient: age17jqc66x9yeshfgd9v78mj483r4zzarqdtuxtrkxe4x5mw679gphshd94th
|
||||
- enc: |
|
||||
-----BEGIN AGE ENCRYPTED FILE-----
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBwVmlDOHh6dUFaR1YxOHJM
|
||||
N09LdWxscStGSTYvMm1QaWRkZVVod1JTTlJnCkp0Z1pSa3lVU2JqUnU0ZnAzWWEv
|
||||
bHc3S2ZjK3E5Z1BRL2pYWklvSFoxSnMKLS0tIGxTd29MK0dRWFFnaGVwa2NVQXM2
|
||||
TjRQdHRZNDFTTWNWR2FvZXhQMVZORncKdAcSOB50bNQsaTGtAqnvjYyFpcjdMGzO
|
||||
NvwI5ZquoT/B37Xsg4NQ7++/ZkemzkuX5VYoXLVAQ9WJwQxewoMFKg==
|
||||
-----END AGE ENCRYPTED FILE-----
|
||||
recipient: age1px0h5l9zp2dww0m8fncrc82kfdmzplsfv2ltat7sna28xpg09pqqcl3s2k
|
||||
- enc: |
|
||||
-----BEGIN AGE ENCRYPTED FILE-----
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSA3V0M5bnBzamdGemU4Y0g0
|
||||
TEdZdzQvUVpYSkozZlRsMWM4MStmN2lxckYwCms0ZTF3cTB5OTRzWWE0a1FXWkhG
|
||||
SGptVCtiOHptV3V0aVNQUUJ4NkE0dnMKLS0tIEEvK0lwa1pnc1YwWXhkYmh3cy94
|
||||
S2E0SUk4RDdSWktQTjZoTkNUallLTVEKNqTmhQo74Q00ZhrSrD+/JB3TwxL9OQdC
|
||||
XSONu6XXzPzfhllNIwe3BnmRVJM/1ru8rvllk98Cvle0k4K5+xBJ/g==
|
||||
-----END AGE ENCRYPTED FILE-----
|
||||
recipient: age1ufg390ydrmma849t9xfkxxl5xvdkk6mngnlzhmy7mvuaje8sgcmsmnq6l7
|
||||
- enc: |
|
||||
-----BEGIN AGE ENCRYPTED FILE-----
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBKYnFtM29UUzVRYWpTN0w2
|
||||
MDVmQk83THEwR3liN0hmemxBZlF3eDVZV0RRCjRFcjZKQk8yaXViWm0vTUdrSlM3
|
||||
UDNnemoyMG0wbnpzK0FRYXNiRTRnaW8KLS0tIHd1QVYvWWl2QlcxVGZzbFBLbjlu
|
||||
djV4dVpnQzF6LzFiWTREaTYrMHAreUkK82l9Njt3B7HJBLtdLjNSnQ4TDTvBgG9/
|
||||
iUEn3fd9OmvfSGcbp4hhwB7q0UJqb7AP47ctwDOEg/FYeF7eSBvVgg==
|
||||
-----END AGE ENCRYPTED FILE-----
|
||||
recipient: age16j42pdc5dr6wnj7xayhkqdj2rny9u68fcqejs50hqq42scssh4gsnrrnlt
|
||||
- enc: |
|
||||
-----BEGIN AGE ENCRYPTED FILE-----
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBoZnVZekVieVVRT2ViWUpN
|
||||
bFk3ckJUU1RHRUJrbkNkTUZkbVEzRGd1VWtRCk03MFEweENBOWtOS3gvM3ZpNTF2
|
||||
WG1UWTM3OFp3RERNNC9iV05aOTZnWmMKLS0tIFZqZ0NhZVhxODMyWENhVlI3SUR0
|
||||
b2doaGp1VHBOUVFacG51Rk02d2o1MWMKnNYktA42DgSWWGuy0XTOtUN9ry49WOBd
|
||||
NaGq27ke16XhQ0ZC+r2a1g7YAFaL+qWJxpaH+0ojhUjti3x9O3pWqw==
|
||||
-----END AGE ENCRYPTED FILE-----
|
||||
recipient: age1nruncs4l0ufk7yuc4des8p99c0alfndl0lhsws8tycl5pplfp56s30af5f
|
||||
- enc: |
|
||||
-----BEGIN AGE ENCRYPTED FILE-----
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBRTkQ0cm9MSTMzZjF5T0Vh
|
||||
TWlOdVJzZ1JvQWdLTzZsYWZ6djJhTjk2dnpFCmxUZkhrOW5OTWtrS2FLOUdsWm9F
|
||||
WVB2VzhTdGwxNzZIZ2lSRmRielAzZkUKLS0tIEd1RlIwQXhocmJkVWtvVGMvTlVB
|
||||
WWtyVjUrWjYrWDFmc3VtdzJUaEFPeGMKvWONIbN6B9Ims3f0l/lfUTU116TFPXew
|
||||
K6kAvyySbh9Z9JK8Uyp31WLSbqVy56eGr2Su8Hj6rAGIoJdo0v5NoQ==
|
||||
-----END AGE ENCRYPTED FILE-----
|
||||
recipient: age1k7d2du5mejsmv5rzavm4xwgpthqvcfsehduquv28nzs53zppa3kqngfxq2
|
||||
- enc: |
|
||||
-----BEGIN AGE ENCRYPTED FILE-----
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBlSWtxbHJsQllITit4NVZv
|
||||
b2Z3YzJZTHFOTGU3VzkwWEFZUjdWTC96UjNVCi83SkYyaFdoTVN3eFhLQnA0bDVa
|
||||
OExZRFl1M28yV0VTUC9hZUJJR3dYTWMKLS0tIGNuc00rZDI2dHFqWEc2R1ZVb1E5
|
||||
Vm45YUpvWlMvMXUxdDB0NHpncWpIMjgKcMIZK2ww/VPuuBXlL8klSTm8ySME2njC
|
||||
vbhCzg9VIqIV3q2i8WEbWFmo7uyiyF89Z52xf/3uUc/TUEqjVVDnZQ==
|
||||
-----END AGE ENCRYPTED FILE-----
|
||||
recipient: age16kqfmvz4e23hmdlqresnyw69ej604s320mmd49h4hm3fhqchtgyqrws0k2
|
||||
- enc: |
|
||||
-----BEGIN AGE ENCRYPTED FILE-----
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBZRkRvN2dBZ1ErZmNhSktm
|
||||
LzNLdzJLR3RhR2VqS0xCdlVSVUJtRnpLTkVNCjFmZlp3U05Vb2FwbDBBdmhjVUJt
|
||||
Uk01dDVzS1lWeEZueDlmcUw4MlFFTHMKLS0tIGJ1bENFSFZPa0tNNFZsblFOM05x
|
||||
THAyY3draFB5eHpydk94Um1qU1BZb0UKUVszMlSqUPsEj3vVseGksI+SEEjEblNo
|
||||
zK5fwkYH0aDubsotkVcJjoXc2ZArHyygIMImLNpRBJYIScDGzzTBRA==
|
||||
-----END AGE ENCRYPTED FILE-----
|
||||
recipient: age1arhf2q45zw6wf2uevju4savp575x3m2tfvved5zzq3ay92ynua9s3cm92c
|
||||
- enc: |
|
||||
-----BEGIN AGE ENCRYPTED FILE-----
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBic1JIWi9xKzVkVUtqdHBx
|
||||
OVU1djJDbDcxYTFCSEQ0MTVRaS8zYm5QWFhRCjVNNFE0RUtPTlE5U3pRTlh2eVNy
|
||||
OUU0cnpLNlluemdkVVNjWGxhV1NTRUEKLS0tIDNkWmhoSTNkWExWek4zdjduaFBB
|
||||
UEJjYnZwYVN3L3p5Vk9sRGVQM3dld0kKsFjQKYOClBjWBaU3kCEP5yYGBphUfgOO
|
||||
E1epPU6UI3asa2AC/svfFBZI3u/7EpxDH9jaSSpPec4QTwkheiVVrw==
|
||||
-----END AGE ENCRYPTED FILE-----
|
||||
recipient: age19mn8zrxl8zpps9yvrh4euquvygpp4fp8queg7xc6qhtnl4ng8c9qx02qwn
|
||||
- enc: |
|
||||
-----BEGIN AGE ENCRYPTED FILE-----
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBEb0VlcGQrWnZHWlM4YjhG
|
||||
NlN6c2I3UnpSdkhYTXg4NlM5aG82K3YzUGwwCkdsM0JLUm03RTk0cm5SSitzR1ls
|
||||
bVBLZ3kzaElDaTRHR1Q3bjdXYm9BMk0KLS0tIEpHNE9MTDJuMDEzV002YTVpT2J5
|
||||
Rm9nem5FSnA3M1NzNi91NnBtRi9XMmMKaGB7uE6HCE2cYHfI1VscO8meeq6O5JPi
|
||||
7bTansMUKCPKhDTBYlSnxxjlDE3DRB8ZdxW7LiG+8iVI7HUbsjTejQ==
|
||||
-----END AGE ENCRYPTED FILE-----
|
||||
recipient: age1jlltcv5jcnm40z5k0q6hv053k2rqpqvemtuecdwn527uw8uqz4es3x7m68
|
||||
- enc: |
|
||||
-----BEGIN AGE ENCRYPTED FILE-----
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBlSnYrVVFObDRmUE5UN0xY
|
||||
K0ZIMzNtUWM5RE8xMmlhTm5pcHNEemVwYVZNCnVLWGRObkxPcWdaOGVMcExqRlRw
|
||||
VngrNGJFWXBHZGh3YXVBQmtSZXBsTzgKLS0tIFovVG43V0FBbmRJYW1qRGJCRUZp
|
||||
RENmSmFuMmhFV0xHUHB2N2xWbHVvSWcKJXaXCfz3Zh4SRgTJkMuPabOq3laRorIo
|
||||
F4R4bJDkao5L5QWDH1BWSi97kTTLz3QgmsEGPHm/SpODM7Llx7EPMQ==
|
||||
-----END AGE ENCRYPTED FILE-----
|
||||
recipient: age1ug787sgt6st6k82fgkrug2lzltw4qsukrrqqs3w27ewwqj8rg4hsxcmylz
|
||||
- enc: |
|
||||
-----BEGIN AGE ENCRYPTED FILE-----
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSAyR3VNVisxbnoyQXZGaTJV
|
||||
Y0lMemtYK1F1QjAyWFVqSm9VRzhhbHRCRFRJCnZINVpiS0ZCRXUrVEJhcVlSZk9H
|
||||
TnhaeDU4b0loQjdOazJVbkpEbDZ4K3cKLS0tIGFKRjd0Rkh3RTk2WG9uQUVpWU1K
|
||||
c3IxVmU0Tnc5dnBBWUZhYUgxV0NaaE0K+00bh1AiHdTL3gsA98fvFI2/IDsnWiiK
|
||||
3tOBK4UB5OE5PRupqVXlpEK3ZqKXju/MQbu0/KdaVwHxlL+WXdjAGA==
|
||||
-----END AGE ENCRYPTED FILE-----
|
||||
recipient: age1529taqdwr6t0w7cvzmty0d5y5593wffl0krt48j6uc4u39k56g2qf6ywtp
|
||||
- enc: |
|
||||
-----BEGIN AGE ENCRYPTED FILE-----
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBjSHFHbmU0WGg2a1poMzRF
|
||||
Ry85ZWZCYVdVWkVoNnRWWTZXWWNqQURtendJCndHcmtIaFNPRGMrblJndElMKy9Q
|
||||
NHlaeDVOS2k1OSt1bWs0RlU3a0dJTFEKLS0tIFRrMGJWYThKbVQ5SzdjLzlQbk9a
|
||||
b0xlZGFvTVJGRjZHQk5XM3VObFRnaUkKMavcISNlQh+5yHpA1M5JIkQEF2qasnXH
|
||||
Pd/JWKhnvk3Lyd45ZBJEqFV2wOknJF7v4Z0jdbo4WiUaLh5shqOp2A==
|
||||
-----END AGE ENCRYPTED FILE-----
|
||||
recipient: age1zhfyuzlq40reuqlr34gf77852nhs3t6mqfzrqmas8z6sxk7tcfhsungrm0
|
||||
- enc: |
|
||||
-----BEGIN AGE ENCRYPTED FILE-----
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSA3ckNVU2lUQS94RkRGeTFo
|
||||
aWNJRTdpcWxUOWZEOHF5ODIrRkZ0MkpnRndBClREbDNNUW9RZWFSRHdON3pOT3VZ
|
||||
bUlFZnFGTnJpSzhFVjFLMVhKZTNDWUkKLS0tIG44Ui9YL2RPYzJLck1UVVBiRTRJ
|
||||
MWxURWJlVEFBQ3RSZzZFYWJvaW1FVW8KX8/o/4LP+Wp/qOfF4wt7cTt0O+kAZbWR
|
||||
Xu+dthboRmZBV21zzrEvzKGRFBj2T3EMyGDqqPcfX/vuhbVEJJgZyg==
|
||||
-----END AGE ENCRYPTED FILE-----
|
||||
recipient: age1nxlnrevqs2msdatze562vz6ym4pgydt2zndye83lwqauy6flggtqrevyw5
|
||||
- enc: |
|
||||
-----BEGIN AGE ENCRYPTED FILE-----
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBYck0zNDJaMmhLZUxxVUxR
|
||||
OGJqdTBHMFpzb2tQV3NsUVhPc1JTeGxCWWkwCmtWeWhqRXJFNVRoVVBTcjJtWlM2
|
||||
UmQ1OXh3SnJVQWQ2NnloVjRCdUZnaDQKLS0tIGZITFdwcjJFMU5iRHhvRllKaTA0
|
||||
dW1xN2ZhUmZ3VlNzWFprYWNOVG90aVUKH8Gd6gkbDIZydnb1mL0tpujmucLhbPbm
|
||||
sUm8KZpNUfWpP+t7SK5M91HFaAZiqGJOARb31oszwqLGcNDliV40MQ==
|
||||
-----END AGE ENCRYPTED FILE-----
|
||||
recipient: age1scfc8p53q5aq2a87tcmsazmj8sfeft0s8kxg0et4nm3ucxyp3c3s6te82y
|
||||
lastmodified: "2026-07-23T21:15:41Z"
|
||||
mac: ENC[AES256_GCM,data:qFhnPra6IE3wyKQ4WKweON0S0YtD5I0adGZVfA0m6BVilN6bX5oC/1j5NK2oHrsz920hSl0SOF8LrpqOrUyGjSRkPsN4kq8qr9bJcrX4URiktP0oRden5LLt6hf+ZRP7WmRXFqixPkPHJnZIoAvkNnTFce7cDq5NEAHkKUEKG7k=,iv:nyblUDGeu3TUfFivYylOn3C/HITj99qiPI2+mh8AGh4=,tag:FrtRzSCylC4wlIoqZdfx7w==,type:str]
|
||||
recipient: age10at8862478urh0eeuwh8hzln6ck78jgwtztgxatwqlzwagg77y5snm4xzg
|
||||
lastmodified: "2026-07-19T02:30:40Z"
|
||||
mac: ENC[AES256_GCM,data:UiL3VMDF6rq4Nr87KspcDx434q3tfNXeb5pwH2O+4ssNQ6xzcYDdzXBnhAY3zLBsqPMKrvHBd4Ot/gEMcq3FMIVe7Q6p9yWKpep66KZ/yWEhAlwIVhD79Oj8VS+1CHKjf25zpRdhZorp04oeFQQd9VfjJB4EE/Q1aVbwTGlpIic=,iv:i/0conaFgFia+wzNTdUL6tlSTw35HTK3Ap1Sr5RGHf8=,tag:ULbz5FllShA/JjlSRdxA0g==,type:str]
|
||||
unencrypted_suffix: _unencrypted
|
||||
version: 3.13.2
|
||||
version: 3.13.1
|
||||
|
||||
@@ -1,26 +0,0 @@
|
||||
{
|
||||
"data": "ENC[AES256_GCM,data://9IEHIVCfHjyMNao5sCu2zNlZ/CaW+JyxVpGpD/vab2qvURunCUY7eMfSOyvOx/2WPXnWWlkoVJPCR1ec/yUg09EaSMfxrvqlu4UJI3Sxvu9xNuDszMwMMD/sCulJDiMWFNLp8qaYt7UGzexp4+GlGiqWDxk3ZEu/iLmSApzBrpciTF0lfehT4qblDovo9QXG2KDWFhCt2SwEKmHJ61Yl3pVAQnPLyTWaNhwWD/mYL76mIiDVKq7DJlvi9MBxzi3aYh/ttuHgRCvnCL0C9oUvOyT2cljwra0LXuOrum7FhPIXXboA7WTEbTHJm1LB5yLfxDrnWCrGxQzFQAwNixVHIcjuvjjvA/LifTvbj5FYM8x7an+DwXPfPhruFrDew1paAWsEGNNYXSmAlju2QLI/OwLHFFLuDyLnBKQ6RLtZbCEAOUKf2h4iZj7nH86eb/aGU=,iv:rj76+MJBCpiYyx2Ogut5UxJj3Gn+bygvu2mtuY5hFLA=,tag:yQmTvWVu8ZLug/7fo6RnwA==,type:str]",
|
||||
"sops": {
|
||||
"age": [
|
||||
{
|
||||
"enc": "-----BEGIN AGE ENCRYPTED FILE-----\nYWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBqcURBS2tIMGl1aWxLMHFQ\nS3ZGVTE0cnYzTmpnQ0daL1F3QnlpemRCMmtjClJiZzIrc2syN29sRXVoaXlPRUNF\nakw4RndmbEduTFg5ZVRKUTVwRWpGRjgKLS0tIHZ5ZFlyODlGWTJoNGpXR3RhY0ZH\nOGpPdDZQVmt6UWZXbHkxQTBoeW1JencKbsfH1V1lUj8mmHyLNj36VaRDgaBojcDU\ndoQWmSEXxjticJqdadbVKb3UABpvzAZxASCy81sa3wH0gT+7zLaNyQ==\n-----END AGE ENCRYPTED FILE-----\n",
|
||||
"recipient": "age1njap586hc0q43kr03g6c8eqhdsmk8zcafkl3f83xwlc2gqhlmfgs4tmwad"
|
||||
},
|
||||
{
|
||||
"enc": "-----BEGIN AGE ENCRYPTED FILE-----\nYWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBrbFdaMmRjcHgyV3Y0eWZL\neG1uMEtpQVUxQmNTU0Z3aC95Zi9uQlgvMEhRCmJiVlRTR2NocGlnbVU1UmlLVVA2\neERiMXpiQlVPNVhRU09XVGxrY0Ixa2sKLS0tIFRuOHdoelJxOXNRRHBJNnlhSE9j\nUUZHZmQzOVFwRmhFV3pvdnpRMTMraGcKRHBuSUpbHaEzH2tuSBE5MsLJDCuH3vUx\nO0jnDldCWkCw7Wvr/tQAkaDI8axZcYVDUkCEk+xqAdxDozLCPhEO6g==\n-----END AGE ENCRYPTED FILE-----\n",
|
||||
"recipient": "age17e89ty6p0fw24daanen57wg8uald9s025t3wwxsw269svwpmgvrshfvfvt"
|
||||
},
|
||||
{
|
||||
"enc": "-----BEGIN AGE ENCRYPTED FILE-----\nYWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBZMW5taXh2QkJkT0JjaVpq\nQlBjVFpFYUhXTHhOT3prbFcvUFB0YnRGUzMwCjhHT01lY3dPWDE2dDZWZnJza1hN\nS1AvZWIvdjZoYmRjWlliK1hiOEdrT00KLS0tIDljWWY0NlQveS9VR2hOSUNyTUtH\nK2gvbEVibmZSdktPemdEQ2p5U0Y4d2MKKitoTi2vbxJ41IoWMlj4vO91Ahpj0hHP\nrKCPyx7ws/IUMmKGyvDLpZ3pYqHD9jl5pLfB05Hh4Emhv4lA1qtwwQ==\n-----END AGE ENCRYPTED FILE-----\n",
|
||||
"recipient": "age17jqc66x9yeshfgd9v78mj483r4zzarqdtuxtrkxe4x5mw679gphshd94th"
|
||||
},
|
||||
{
|
||||
"enc": "-----BEGIN AGE ENCRYPTED FILE-----\nYWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSA3K2dVV1BHbDk2Z3FhNmdV\nZEN2Um94K0pma1k4b1V5ZUJxQ1hWU1grZjJNCi9iRVVqOVpEMmtBUi80NThlYldY\nTUtZOVErT2VSWk43NndpVzBlL1lQaTgKLS0tIEpCM3Q4NjM3N1lpUE56N04rTXN0\nZkZ6TkV1TU9OV3dpc25RNWRpVnprM00KItUzKBdShakOfX8Sr+k906nsvYPl8QLb\nge//1GA+ukGsaS9rcChOY89vFdm61JDmj1jXSJ0CN4wLMW9/eblZRw==\n-----END AGE ENCRYPTED FILE-----\n",
|
||||
"recipient": "age1arhf2q45zw6wf2uevju4savp575x3m2tfvved5zzq3ay92ynua9s3cm92c"
|
||||
}
|
||||
],
|
||||
"lastmodified": "2026-07-28T01:44:28Z",
|
||||
"mac": "ENC[AES256_GCM,data:efFXIqbauOURR7lrVpK7kqRIPgYjgWfenBYoX7mUrQ/thFc+ApcAx58Fi1zg4biwIs7NarJAgDqAxZi+yKb2Sll8H/wlsEjWDU4iQlLJdIQw7wey9eDW8pgBLd+6E7FXrgn1Vh49dS5GXlC6clAYk1lCiB4NvfzPDy5Ktpl+Chs=,iv:+zCqj5I1MLJfRRiIrqgocYB49PZnlV45PUTH4gYlfrQ=,tag:WxY1sF4kFOTEzbSFcfJFbQ==,type:str]",
|
||||
"version": "3.13.2"
|
||||
}
|
||||
}
|
||||
@@ -1,52 +0,0 @@
|
||||
wifi-password: ENC[AES256_GCM,data:SZQPtU6PYHbf9o83wq3KTupx,iv:FxO68Pn/+N58r/OPLfkAMYPFpP8TYxszMniFd/01E38=,tag:jwxaY6zDEcO5r9OWSfvUyw==,type:str]
|
||||
sops:
|
||||
age:
|
||||
- enc: |
|
||||
-----BEGIN AGE ENCRYPTED FILE-----
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSAzTFFlZHdGUzk0b1Zva1U5
|
||||
V1UwREEwS05icWxYNEdKTE8rQ3lJMjM5YXdzCkw4OE4xWVUzVWZMaXg5OFo1UG8z
|
||||
WkNwK20yb09rV2VVSENwNWUvTmhJNk0KLS0tIGVPWUhFS2RDcTNjY2JLaWxvcTZt
|
||||
Z2RscURQdDVCMUdiVng2YWRMSlEvVVEKmyd3re6AaKn4gBjoT0x3e/zJznvJFYKn
|
||||
ugKu3EsUX+gbailPmY1ss9+MVtpJFGZa2FiM0x1wSMKm6UJH0aPhVA==
|
||||
-----END AGE ENCRYPTED FILE-----
|
||||
recipient: age1njap586hc0q43kr03g6c8eqhdsmk8zcafkl3f83xwlc2gqhlmfgs4tmwad
|
||||
- enc: |
|
||||
-----BEGIN AGE ENCRYPTED FILE-----
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBtK2tud3pWSmd2aC95ZkV4
|
||||
RkVvZXNrK09ZVmZsYWhqZ3JlaXBCd1NRWGhzCk5lelhSN2N6N2VhWEVXUjhLTjZH
|
||||
V2VlMm5XdUtuK2d0MjIyRi9xeTh2ZXcKLS0tIDh5Rk1UT0dWblBkUXYzU3YzVGkw
|
||||
OFk0bkJ5RXZpWE4rK05QUHlLQktYSmsK/HsVEIhBEIo3qqVWdUJEWnHZiKB3uHVH
|
||||
R+nJGuXa2B/oUoxEwMP2YBHwwjLLiJCTYy+aQtiPdTrVq0YJ0HmC7w==
|
||||
-----END AGE ENCRYPTED FILE-----
|
||||
recipient: age1rrxqea6q6pn39sw8y5te63h2py8jgjl9v0jyper86w3ggtn67upqg3ah39
|
||||
- enc: |
|
||||
-----BEGIN AGE ENCRYPTED FILE-----
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBpVVVxWE1nUDBLTVFUaGJJ
|
||||
Vm9EUldwRGcvbXhuT3JPd3N0WXo2S0gwRnh3CktabkdaSndaZUNSeUJGRzFKcjlH
|
||||
b0x0SFdEQ0VqNWdQcEkxb0drVVJNcVUKLS0tICtyVGRqUmFtRHV1SlpXSVd3N2VN
|
||||
elQrVFdhWTJ1R2pJbjdYa2w4NmRmc2sKJHqLYdNQJcna71KNhGF80iS1hIYG1U1w
|
||||
I2kihepJsrmYr76ld9k+u1ZfnuIuJ1ozYsZothE+dr4pV0k8s6wkpg==
|
||||
-----END AGE ENCRYPTED FILE-----
|
||||
recipient: age1adur9g330gua4l6ndk8cqjg35qc8yxwgme6wrl2hpylcc7vxm38q05ejuy
|
||||
- enc: |
|
||||
-----BEGIN AGE ENCRYPTED FILE-----
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBSWXJxUEk5Mkp5eTZXamFR
|
||||
SXZ4L1JTSGNaQjFiVXFGaEdvWkV2ZzBlVzJJCmd4NmUzZEdCbi9vdWdkVGZRL2Qv
|
||||
cnVXa05xd2gzaXh1SnlMZmtueGpZMHMKLS0tIFowdnhFVGFheURQU1V2M0ZuM1Y0
|
||||
NTJFWXBEYUxmOU9ROTkxWGhYUmlqMHMK3pexvc16BLKjh2meqtNm3M1zyLQ3eEsz
|
||||
7C5WkdcSpCkW1lPDGtW7pEdAIL15StD4x7ut4MkSk0BjG1S+RpDbzA==
|
||||
-----END AGE ENCRYPTED FILE-----
|
||||
recipient: age1hrx8qj02fj2ea6d4g9vqhyj9hl7fppkjqfdx2l37py3h6pdkr95s8n8rvs
|
||||
- enc: |
|
||||
-----BEGIN AGE ENCRYPTED FILE-----
|
||||
YWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBkR0o0SFh1L2xjZ1RjSC9K
|
||||
ZnFyb1lJbnlMbENYN0VjSVQyU1F6RDBrL2hjCnRwTUp6Y1hJWldVeFdQZSttZ1Vw
|
||||
L0dCS0ZROFArb0ppVzB5WmV4bWI5alEKLS0tIDh4VDV2TFhhaUp4L09jYm52UCsr
|
||||
NGh5a3VMY2ZMZVBQbmRHeWsrQnZVWDQKR1UeSZ/EzZEXMqyjB1I2SHELv8Ha/tmI
|
||||
kJKs2WT1RtDhAiTrbty3f4oVrXWSYKZr40kNiP/RLbUcH1s65ys/tQ==
|
||||
-----END AGE ENCRYPTED FILE-----
|
||||
recipient: age19mn8zrxl8zpps9yvrh4euquvygpp4fp8queg7xc6qhtnl4ng8c9qx02qwn
|
||||
lastmodified: "2026-07-22T01:15:21Z"
|
||||
mac: ENC[AES256_GCM,data:dC/oIqMUHkOh3AocOwP7Gc6XGH3L+nTqJfhFNts1DNbRXsopNIxVBtIz2pEhwnWSQrqPisDLmPHFBwRpGVn01u8w8IU1FKbAKC0J2nJXF8ozpInbjzDOmehqPWZG7yaKoq8cwAnp5XOk+IVO4l6tPxLxkExU5fT2ALuMq+sgOko=,iv:jaVyArpf6zMCFa6J9X1aQMGrmFq+W2CPZdWO6vVW68c=,tag:S+qF8/FkgHc4uW0e4ICmSQ==,type:str]
|
||||
unencrypted_suffix: _unencrypted
|
||||
version: 3.13.2
|
||||
@@ -1,22 +0,0 @@
|
||||
{
|
||||
"data": "ENC[AES256_GCM,data:olqZm/3yxtVnlDeYzAWnXjUP9pgkruFEpeS6qONHSfdrj0lJOch44zn1pbG6Jg5QYcksny0rHki9PJEZSbRIyrZ2mSKfTJing3n9M5ODm9xPMbNYn8cpVB4AzZmCSAH7mN4p/98+5hJ1qzopGwCgKTcOOJ9DMUl9AStXORY=,iv:R2K3OehIfyKZXRFH3Y9KYnvEp6CjHPT4Ki2P1YrC6+I=,tag:1FM/E0n9teARAZ1HZM4b3A==,type:str]",
|
||||
"sops": {
|
||||
"age": [
|
||||
{
|
||||
"enc": "-----BEGIN AGE ENCRYPTED FILE-----\nYWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSA3RFRoRW1rRklxbXYybUtu\nL1lEOXJBMDNCN1paU1k5N0hQRTJYQ1dQcTBVCk9OYnloOFZad2toOHo0UXFYSzNR\nZDUwV00vKzVnckMwSE9MMnE4RUozZTgKLS0tIFByL3Z3SzJFeEVNZjhuRDNTMzFo\nNU5tZ2tQU1ZORE5qNm1JUEJrTGdzUHMKypYeJz6BeUUY4aPKazB1nxncOA3DGkal\nfemLr9uwCJw+D3xfXzrIKrI0w3OH7bu2LmWqrNDSz3Bwa5VDuHoLqQ==\n-----END AGE ENCRYPTED FILE-----\n",
|
||||
"recipient": "age1njap586hc0q43kr03g6c8eqhdsmk8zcafkl3f83xwlc2gqhlmfgs4tmwad"
|
||||
},
|
||||
{
|
||||
"enc": "-----BEGIN AGE ENCRYPTED FILE-----\nYWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBqTzhtOHM1Q1lWRXdadHB6\nZ1VrMXFsQWZEbDBMMnQvYnZSN090VnpyQnd3CkdJaHBsbzQ4SERmU0dBS2RZMUpi\nZVYrZmtSNXhiaDNTZkNMVEZud3ZzT3cKLS0tIFhETER5eWJRK05ValFhenRnSGg1\nc1prdWVTVmMyWWt2UkNXWVRLcWRHcm8KjVn1faGmsWiFzcNg4PnxZQfeQONFKz/i\nyJGJc7w2KSZ3La+44jfitMziJQ3AFRUlAhGDX25kWZJ7PtjmFvYwuQ==\n-----END AGE ENCRYPTED FILE-----\n",
|
||||
"recipient": "age1nxlnrevqs2msdatze562vz6ym4pgydt2zndye83lwqauy6flggtqrevyw5"
|
||||
},
|
||||
{
|
||||
"enc": "-----BEGIN AGE ENCRYPTED FILE-----\nYWdlLWVuY3J5cHRpb24ub3JnL3YxCi0+IFgyNTUxOSBkVXFQazlvUEx2NzVUNlF1\nN3FQNTdFL1FtSDhCdmNQclRMdG1vRVhvcEhzClF4Ykh1ZS9JekFzOHkvOVphejVy\nd2s0a0IrZ1d4V3diOGNvMlBHUVlwTmMKLS0tIEFhTE13Q2F1di9ESXdJVDhJUVl4\na1kzNDBLcHhlRGg2Y3hSSmdEc2k5MlEKpU2sIv5Pq0LaKkvqA/fRkiB9PcvsSjvq\no4iEFzGNDi63QzXqftWnzengEy/6nWGNA31fzScQftESesGlKfYYlA==\n-----END AGE ENCRYPTED FILE-----\n",
|
||||
"recipient": "age1scfc8p53q5aq2a87tcmsazmj8sfeft0s8kxg0et4nm3ucxyp3c3s6te82y"
|
||||
}
|
||||
],
|
||||
"lastmodified": "2026-07-28T06:46:00Z",
|
||||
"mac": "ENC[AES256_GCM,data:xJHyh4JuY+qnKljS3mtfN4Cy0pGPUszhU4ud2L8JxTFGUHTXhZzwQ3s1vM88efLTwM7yDK7oBMBUH2xligAzglDnqkE1T0l7R9k+hyldm4nYqOK6Y8GyPt/aald6oLYGCx3J7Ap0Oqq/crU+s4NiY9enf+R+08jliFyrPBHuRSY=,iv:FGtURIfSrf1tRGqpPfn40GqybhNYFBcvR7n1YkcAMGY=,tag:ir0TU7YuxpPuew+nyC9/eg==,type:str]",
|
||||
"version": "3.13.3"
|
||||
}
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user