#!/usr/bin/env bash # Deploy a NixOS host from this flake. ALL arguments are mandatory (no defaults). # # ./deploy kexec headless kexec into a RAM installer, for a # read-only-root box (ZimaOS) where # nixos-anywhere can't ssh-copy-id. Ships our # SSH login key. Then run `install`. # is only used to look up the vault item. # ./deploy kexec-local [--yes] kexec THIS machine into the RAM installer, # no ssh/second machine involved. Run as root, # locally, on the box you're installing onto. # Disks are untouched; console drops for # ~1-2 min then comes back as the installer. # Prompts for confirmation (--yes skips it), # because run on the wrong terminal this # kexecs your laptop. TMPDIR (default # /var/tmp) must be exec-capable and hold # ~3x the tarball. # Then run `install localhost`. # ./deploy install first install. Wipes the OS disk. Ships the # host's sops key. =localhost/127.0.0.1 # skips nixos-anywhere/ssh and runs disko + # nixos-install directly against /mnt — but # ONLY once actually inside a live installer # (hostname nixos-installer, from kexec, or # homelab-installer, from installer-iso). # Run from the REAL running OS instead (e.g. # a box where kexec-local doesn't work), # it builds installer-iso, stages its # kernel/initrd on the ESP + the iso file on # a non-OS-disk partition, sets a systemd-boot # one-shot entry with homelab.install= # on its kernel cmdline, and reboots — a real # ACPI reboot, not a kexec jump. The booted # installer's homelab-auto-install.service # reads that cmdline param and re-runs this # exact command itself once its repo checkout # (homelab-checkout.service) succeeds, finishing # the install unattended. See CLAUDE.md. # ./deploy switch rebuild + activate on a running host. # ./deploy boot stage for next boot, don't activate now. # ./deploy test activate without adding a boot entry. # ./deploy image build an SD-card image (e.g. rpi mercury). # ./deploy flash build SD image, write to , and (if # ~/.config/homelab//age.txt exists) # drop the sops key on its boot partition. # # = a nixosConfigurations name (e.g. jupiter, vps). Its pre-generated # SSH host key must be at ~/.config/homelab//ssh_host_ed25519_key. # # Runs from a non-NixOS host too (nixos-rebuild / nixos-anywhere via `nix run`). # # Password prompts are auto-filled from the "HomeLab" Proton Pass vault when # `pass-cli` is installed and logged in; otherwise every command prompts exactly # as before. Both items are keyed by , never by : the address is # incidental (DHCP, a new box, localhost) while the config name is the stable # identity of the machine being built. # darman@ darman's sudo password (switch/boot/test) # root@ root's ssh password (kexec/install) # Override with HOMELAB_PASS_ITEM / HOMELAB_PASS_ROOT_ITEM / HOMELAB_PASS_VAULT. set -euo pipefail shopt -s nullglob # Captured before anything shifts/parses $@, so require_root() below can # re-exec the ORIGINAL invocation under sudo — inside a function, "$@"/"$1" # refer to the function's own args (empty here), not the script's, so this # has to be a global array instead of relying on positional-parameter scoping. SCRIPT_ARGS=("$@") # Locate the repo root (flake dir) regardless of where this script lives on disk. SCRIPT_PATH="$(realpath "$0")" # absolute — "$0" itself may be relative, # and require_root() re-execs after cd "$REPO" SCRIPT_DIR="$(dirname "$SCRIPT_PATH")" REPO="$(git -C "$SCRIPT_DIR" rev-parse --show-toplevel 2>/dev/null || dirname "$SCRIPT_DIR")" cd "$REPO" export PATH="/nix/var/nix/profiles/default/bin:$PATH" die() { echo "error: $*" >&2; exit 1; } need() { command -v "$1" >/dev/null 2>&1 || die "missing required tool: $1"; } # Self-elevate instead of dying: re-exec this exact invocation under sudo. # -E preserves the environment (HOMELAB_* overrides, Proton Pass vault vars) # across the re-exec. A no-op once already root. require_root() { [ "$(id -u)" = 0 ] && return 0 echo ">> $1 needs root — re-executing under sudo" >&2 exec sudo -E -- "$SCRIPT_PATH" "${SCRIPT_ARGS[@]}" } # Exactly one path matching a glob, or die. `ls glob | head -1` silently yields # an empty string when nothing matches (head exits 0, so set -e never fires) and # the failure only surfaces later as a confusing tar/dd error. one_match() { local what="$1"; shift local f=("$@") # caller expands the glob (nullglob is on) [ "${#f[@]}" -gt 0 ] || die "no $what found — did the build actually produce one?" printf '%s\n' "${f[0]}" } # Sets tb / cpio / bbox — the kexec tarball plus the static cpio+gzip that # kexec-run.sh needs on PATH to rebuild its initrd. # # HOMELAB_KEXEC_TARBALL (with _CPIO / _GZIP) short-circuits the build and uses a # prebuilt installer instead. That lets the VM test in flake.nix drive this # script offline, and lets you re-kexec a box without rebuilding ~500MB. kexec_artifacts() { if [ -n "${HOMELAB_KEXEC_TARBALL:-}" ]; then tb="$HOMELAB_KEXEC_TARBALL" [ -f "$tb" ] || die "HOMELAB_KEXEC_TARBALL=$tb is not a file" cpio="${HOMELAB_KEXEC_CPIO:-$(command -v cpio || true)}" bbox="${HOMELAB_KEXEC_GZIP:-$(command -v gzip || true)}" { [ -n "$cpio" ] && [ -n "$bbox" ]; } \ || die "set HOMELAB_KEXEC_CPIO / HOMELAB_KEXEC_GZIP, or put cpio+gzip on PATH" echo ">> using prebuilt kexec installer: $tb" else need nix echo ">> building kexec installer + static tools" nix build .#nixosConfigurations.kexec.config.system.build.kexecInstallerTarball \ -o result-kexec tb="$(one_match 'kexec tarball' result-kexec/*.tar.gz)" cpio="$(nix build --no-link --print-out-paths nixpkgs#pkgsStatic.cpio)/bin/cpio" bbox="$(nix build --no-link --print-out-paths nixpkgs#pkgsStatic.busybox)/bin/busybox" fi } # True inside one of the throwaway live-installer environments this repo # produces (kexec's nixos-installer, or installer-iso's homelab-installer) — # i.e. `install localhost` should wipe/install right here. False on # any real running OS, where the same command instead means "prepare and # reboot into an installer for THIS box" (see local_install_prepare_and_reboot). is_live_installer() { case "$(uname -n)" in nixos-installer | homelab-installer) return 0 ;; *) return 1 ;; esac } # `install localhost` run on a REAL running OS (not already inside a # live installer): builds installer-iso, stages its kernel/initrd + iso file # locally, points a systemd-boot one-shot entry at them with # homelab.install= on the kernel cmdline, and reboots — a real ACPI # reboot through firmware POST, deliberately NOT a kexec jump (see terra's # kexec-local gotcha in CLAUDE.md). The booted installer's # homelab-auto-install.service reads that cmdline param and re-runs this exact # `install localhost` command itself (now genuinely inside the # installer) once homelab-checkout.service has fetched the repo, finishing the # job unattended. local_install_prepare_and_reboot() { local config="$1" require_root "preparing a local reinstall" [ -d /sys/firmware/efi ] || die "not booted UEFI — the one-shot boot entry needs systemd-boot" need bootctl need nix need lsblk need findmnt # No default/auto-picked location — the wrong disk here is destroyed # mid-install (see the OS-disk check below), so this always asks rather # than guessing. HOMELAB_INSTALLER_STAGE_DIR skips the prompt for scripted # use, but is otherwise just as explicit a choice as typing it in. local stagedir="${HOMELAB_INSTALLER_STAGE_DIR:-}" if [ -z "$stagedir" ]; then echo ">> currently mounted filesystems:" lsblk -o NAME,SIZE,FSTYPE,MOUNTPOINT read -rp ">> path to stage the installer iso on (must NOT be on the OS disk being wiped): " stagedir fi [ -n "$stagedir" ] || die "no staging path given" [ -d "$stagedir" ] \ || die "staging dir $stagedir doesn't exist — needs to be an existing partition that is NOT the OS disk being wiped" # Refuse if the staging partition turns out to live on the same disk # disko is about to wipe — the iso file (and the running installer # loopback-mounted from it) would be destroyed mid-install. local osdisk osdisk_real stage_src stage_pkname stage_disk_real osdisk="$(nix eval --raw ".#nixosConfigurations.$config.config.disko.devices.disk" \ --apply 'd: (builtins.head (builtins.attrValues d)).device' 2>/dev/null)" \ || die "couldn't read the OS disk device from hosts/$config/disk-config.nix" osdisk_real="$(readlink -f "$osdisk")" stage_src="$(findmnt -no SOURCE --target "$stagedir")" \ || die "$stagedir doesn't resolve to a mounted filesystem" stage_pkname="$(lsblk -no PKNAME "$stage_src" 2>/dev/null || true)" if [ -n "$stage_pkname" ]; then stage_disk_real="$(readlink -f "/dev/$stage_pkname")" [ "$stage_disk_real" = "$osdisk_real" ] \ && die "$stagedir is on the OS disk ($osdisk) that install would wipe — re-run and pick a different disk" fi echo ">> building installer-iso (kernel + initrd + iso image)" local kernel initrd isodir iso mnt_point iso_relpath kernel="$(nix build --no-link --print-out-paths .#nixosConfigurations.installer-iso.config.system.build.kernel)/bzImage" initrd="$(nix build --no-link --print-out-paths .#nixosConfigurations.installer-iso.config.system.build.initialRamdisk)/initrd" isodir="$(nix build --no-link --print-out-paths .#nixosConfigurations.installer-iso.config.system.build.isoImage)" iso="$(one_match 'installer iso' "$isodir"/iso/*.iso)" echo ">> staging kernel/initrd on the ESP, iso image on $stagedir" install -Dm644 "$kernel" /boot/homelab-installer/bzImage install -Dm644 "$initrd" /boot/homelab-installer/initrd install -Dm644 "$iso" "$stagedir/homelab-installer.iso" # findiso= is a path relative to whatever partition the initrd finds it on # (it mounts every blkid-visible partition looking for it) — not necessarily # relative to `/`, if $stagedir is a subdirectory of a bigger filesystem # rather than a mountpoint itself. mnt_point="$(findmnt -no TARGET --target "$stagedir")" iso_relpath="${stagedir#"$mnt_point"}/homelab-installer.iso" cat >/boot/loader/entries/homelab-installer.conf <> one-shot boot into the installer, then rebooting — it will finish this install itself" bootctl set-oneshot homelab-installer.conf systemctl reboot } # Flakes only see git-tracked files: an untracked hosts// is silently # invisible to `nix build`/`nixos-install`, which then fails obscurely or builds # a stale config. Check before doing anything destructive. require_tracked() { local config="$1" cfgfile="hosts/$1/configuration.nix" [ -e "$cfgfile" ] || die "no $cfgfile in the repo" # No .git at all (e.g. a tarball export of the repo, no working tree) means # there's nothing that CAN be untracked — nothing to check. Only skip on a # MISSING .git, not on any other git failure. git -C "$REPO" rev-parse --is-inside-work-tree >/dev/null 2>&1 || return 0 git -C "$REPO" ls-files --error-unmatch "$cfgfile" >/dev/null 2>&1 \ || die "$cfgfile is untracked — 'git add hosts/$config' first (flakes ignore untracked files)" } # The password field of a Proton Pass item ("--field password" prints the bare # value, one line), or empty if pass-cli is missing / logged out / has no such # item — every caller then falls back to the normal interactive prompt. # # Resolve the title to an item id among ACTIVE items first, because `item view # --item-title` has no state filter: Proton Pass keeps deleted items in the # trash, and if a trashed item shares the title, view can match THAT one and # return an empty password with exit 0. Empty is indistinguishable from "no such # item", so the only symptom is a silent fall back to the interactive prompt # even though the vault clearly holds the entry. (Hit for real on darman@neptun, # which had an Active and a Trashed copy.) proton_pass_password() { local title="$1" vault="${HOMELAB_PASS_VAULT:-HomeLab}" id pw command -v pass-cli >/dev/null 2>&1 || return 0 # Lines look like: - [ITEM_ID]: the title (state=Active) id="$(pass-cli item list --vault-name "$vault" --filter-state active \ --output human 2>/dev/null \ | awk -v t="$title" ' { i = index($0, "]: "); if (i == 0) next id = substr($0, 4, i - 4) rest = substr($0, i + 3) sub(/ \(state=[^)]*\)$/, "", rest) if (rest == t) { print id; exit } }' || true)" if [ -n "$id" ]; then pw="$(pass-cli item view --vault-name "$vault" --item-id "$id" \ --field password --output human 2>/dev/null | head -1 || true)" # Resolved an active item but its password field is blank: worth saying so, # otherwise this looks identical to having no vault entry at all. [ -n "$pw" ] || echo ">> vault item '$title' resolved but its password field is empty" >&2 else # No active match — fall back to the title lookup so an older pass-cli # without --filter-state still works exactly as it used to. pw="$(pass-cli item view --vault-name "$vault" --item-title "$title" \ --field password --output human 2>/dev/null | head -1 || true)" fi printf '%s' "$pw" } # Path to an sshpass binary (system one, else built from nixpkgs). Empty if # neither is available. sshpass_bin() { command -v sshpass 2>/dev/null && return 0 nix build --no-link --print-out-paths nixpkgs#sshpass 2>/dev/null \ | sed 's|$|/bin/sshpass|' } cmd="${1:-}"; [ -n "$cmd" ] || die "usage: ./deploy ..." case "$cmd" in kexec) config="${2:-}"; host="${3:-}" { [ -n "$config" ] && [ -n "$host" ]; } || die "usage: ./deploy kexec " need ssh; need scp # kexec/run rebuilds an initrd with `cpio` + `gzip` from PATH — ZimaOS lacks # both. Ship static ones: GNU cpio (reliable -o -H newc), busybox as gzip. kexec_artifacts # One password prompt: multiplex scp + ssh over a shared control connection. cm="/tmp/homelab-cm-%r@%h:%p" o=(-o ControlMaster=auto -o "ControlPath=$cm" -o ControlPersist=300 \ -o StrictHostKeyChecking=accept-new) # Root's password from Proton Pass, fed to ssh/scp via sshpass -e. Only the # first (master) connection authenticates; the rest ride the control socket. # # SSHPASS is exported here rather than passed as `env SSHPASS=... sshpass`. # Both end up equally safe at rest: `env` execs its target immediately, so # the assignment is only in argv for the sub-millisecond before exec, after # which /proc/PID/cmdline reads plain `sshpass -e`. Exporting just closes # that race window and drops a process. Either way the secret lives in the # child's environ, which is readable by the owner and root only. sp=() root_item="${HOMELAB_PASS_ROOT_ITEM:-root@$config}" root_pw="$(proton_pass_password "$root_item" || true)" if [ -n "$root_pw" ]; then sshpass="$(sshpass_bin || true)" if [ -n "$sshpass" ]; then export SSHPASS="$root_pw" sp=("$sshpass" -e) # sshpass drives the password prompt; don't let a key/agent short-circuit # into an interactive one for a host that only accepts passwords. o+=(-o PreferredAuthentications=password -o PubkeyAuthentication=no) else echo ">> sshpass unavailable — falling back to the interactive prompt" >&2 fi fi unset root_pw if [ "${#sp[@]}" -gt 0 ]; then echo ">> connecting to root@$host (password from Proton Pass: $root_item)" else echo ">> connecting to root@$host (enter the root password once)" fi "${sp[@]+"${sp[@]}"}" ssh "${o[@]}" "root@$host" 'mkdir -p /tmp/bin' scp "${o[@]}" "$cpio" "root@$host:/tmp/bin/cpio" scp "${o[@]}" "$bbox" "root@$host:/tmp/bin/gzip" # busybox as gzip (argv0) echo ">> streaming installer + kexec-ing. SSH drops as the box jumps into the" echo " RAM installer. Disks are untouched." rc=0 ssh "${o[@]}" "root@$host" \ 'chmod +x /tmp/bin/*; mkdir -p /tmp/k && tar -C /tmp/k -xzf - && PATH=/tmp/bin:$PATH /tmp/k/kexec/run' \ < "$tb" || rc=$? # A dropped connection IS the success case here, so this can't be fatal — # but a full /tmp, a tar error or a missing static tool looks identical from # the outside, so at least surface the code instead of swallowing it. [ "$rc" -eq 0 ] || echo ">> ssh exited $rc — expected if the box jumped; suspect this if install can't connect" >&2 ssh "${o[@]}" -O exit "root@$host" 2>/dev/null || true # close control socket unset SSHPASS # NB: no ssh-keygen -R here on purpose. kexec-run.sh copies /etc/ssh/ssh_host_* # into the appended initrd and restore-remote-access.nix installs them back # into the installer's /etc/ssh, so the host key SURVIVES the jump. Clearing # known_hosts would just throw away the TOFU record for no reason. echo ">> box is kexec-ing. Wait ~1-2 min for the installer + network, then:" echo " ./deploy install $config $host" ;; kexec-local) # No ssh, no second machine: build the same RAM installer as `kexec`, but # run it directly on this box (you're sitting at it). The current shell # drops when the kernel switches, same as any reboot — that's expected, # not a failure. Disks are untouched; only the running kernel changes. # # This is a one-way trip on the machine you are typing at, so every check # that can fail is done BEFORE the point of no return, and nothing that the # jump depends on is cleaned up behind it (see the trap discussion below). require_root "kexec-local" assume_yes="" [ "${2:-}" = "--yes" ] && assume_yes=1 need tar; need install; need mktemp; need find; need sync; need nohup # --- preflight: reasons the jump would fail, checked while it's still safe -- # CONFIG_KEXEC. Without it kexec --load fails and you've built ~1GB for # nothing; with lockdown it fails at load time too (Secure Boot blocks the # kexec_load syscall unless the image is signed). [ -e /sys/kernel/kexec_loaded ] \ || die "this kernel has no kexec support (CONFIG_KEXEC) — boot install media instead" if [ -r /sys/kernel/security/lockdown ] \ && ! grep -q '\[none\]' /sys/kernel/security/lockdown 2>/dev/null; then die "kernel lockdown is active — kexec_load is blocked. Disable Secure Boot, or boot install media." fi kexec_artifacts # Staging lives on /var/tmp, not /tmp: kexec-run.sh APPENDS a fresh cpio to # kexec/initrd in place and runs binaries out of this directory, so it needs # real space and exec permission. A tmpfs /tmp is both size-capped (ENOSPC # mid-append = a half-written initrd) and frequently noexec. stage="$(TMPDIR="${TMPDIR:-/var/tmp}" mktemp -d)" trap 'rm -rf "$stage"' EXIT # noexec would only surface as a cryptic "Permission denied" on kexec/run. printf '#!/bin/sh\nexit 0\n' > "$stage/.execmove" chmod +x "$stage/.execmove" "$stage/.execmove" 2>/dev/null \ || die "$stage is mounted noexec — set TMPDIR to an exec-capable filesystem" rm -f "$stage/.execmove" # Need room for the tarball, its extracted contents, and the appended cpio. tb_sz="$(stat -Lc %s "$tb")" avail="$(df -B1 --output=avail "$stage" | tail -1 | tr -d ' ')" [ "$avail" -ge $(( tb_sz * 3 )) ] \ || die "only $(( avail / 1024 / 1024 ))MB free on $stage, need ~$(( tb_sz * 3 / 1024 / 1024 ))MB — set TMPDIR elsewhere" install -Dm755 "$cpio" "$stage/bin/cpio" install -Dm755 "$bbox" "$stage/bin/gzip" # busybox as gzip (argv0) mkdir -p "$stage/k" && tar -C "$stage/k" -xzf "$tb" # Fail loudly here rather than with a bare "not found" from kexec/run. for f in run kexec bzImage initrd ip; do [ -f "$stage/k/kexec/$f" ] || die "kexec tarball is missing kexec/$f — bad build?" done # The installer root is a RAM disk: the initrd has to fit in memory with # room to unpack. Refuse rather than OOM halfway into the new kernel, where # there is no way back except the power button. img_sz="$(( $(stat -Lc %s "$stage/k/kexec/initrd") + $(stat -Lc %s "$stage/k/kexec/bzImage") ))" mem_kb="$(awk '/^MemTotal:/{print $2}' /proc/meminfo)" [ "$(( mem_kb * 1024 ))" -ge "$(( img_sz * 3 ))" ] \ || die "only $(( mem_kb / 1024 ))MB RAM for a $(( img_sz / 1024 / 1024 ))MB RAM-disk installer — too tight, boot install media instead" # --- confirmation: this script normally runs on the LAPTOP ------------------ # A stray `kexec-local` here would jump the machine you develop on, not the # target. Name it explicitly so a wrong-terminal mistake is visible. echo ">> about to kexec THIS machine:" echo " hostname: $(uname -n)" echo " kernel: $(uname -r)" echo " root: $(findmnt -no SOURCE / 2>/dev/null || echo '?')" echo " Disks are untouched; only the running kernel changes. Console drops" echo " for ~1-2 min, then comes back as the installer." if [ -z "$assume_yes" ]; then read -rp ">> type 'yes' to kexec $(uname -n) into the RAM installer: " ok [ "$ok" = yes ] || die "aborted" fi # kexec -e jumps straight to the new kernel: no unmount, no journal flush, # no systemd shutdown. Anything still in page cache is lost and the OS disk # is left dirty. Cheap insurance, since the box may yet be rebooted back. sync echo ">> loading the new kernel" if ! PATH="$stage/bin:$PATH" "$stage/k/kexec/run"; then rm -rf "$stage" die "kexec/run failed — machine untouched, still on the old kernel" fi # kexec --load succeeded iff the kernel now reports a loaded image. If this # is 0 the background `kexec -e` below will do nothing and we'd hang forever # waiting for a jump that cannot happen. [ "$(cat /sys/kernel/kexec_loaded 2>/dev/null || echo 0)" = 1 ] \ || { rm -rf "$stage"; die "kexec reported success but no image is loaded — aborting"; } # THE trap MUST GO NOW. kexec-run.sh backgrounds `nohup sh -c "sleep 6 && # $SCRIPT_DIR/kexec -e"` and returns immediately, so the binary that # performs the jump still has to exist ~6s after this script would normally # exit. Letting the EXIT trap rm -rf "$stage" deletes it out from under that # sleeping shell and the machine silently never jumps. trap - EXIT sync echo ">> kernel loaded; jumping in ~6s. (staging left at $stage on purpose —" echo " the backgrounded kexec still needs it; it's gone after the jump.)" # Outlive the sleep 6 so the jump happens while this script is still alive. sleep 60 die "still here after 60s — the jump did not happen. Check dmesg; the old system is intact." ;; install) config="${2:-}"; host="${3:-}" { [ -n "$config" ] && [ -n "$host" ]; } || die "usage: ./deploy install " hostkey="$HOME/.config/homelab/$config/ssh_host_ed25519_key" [ -f "$hostkey" ] || die "missing host key: $hostkey" [ -d "./hosts/$config" ] || die "no ./hosts/$config directory in the repo" require_tracked "$config" if [ "$host" = "localhost" ] || [ "$host" = "127.0.0.1" ]; then if ! is_live_installer; then # Not already inside a live installer: build one, stage it, one-shot # boot into it, and let it finish this exact command itself. See # local_install_prepare_and_reboot above and CLAUDE.md. local_install_prepare_and_reboot "$config" exit 0 fi # Local install: no ssh, no nixos-anywhere. Run after `kexec-local` (or # from a live ISO) so /mnt is free to wipe — this IS the box, no second # machine in the loop, so skip straight to disko + nixos-install. require_root "local install" [ -f "./hosts/$config/disk-config.nix" ] || die "no ./hosts/$config/disk-config.nix" echo ">> disko .#$config onto this box's OS disk (WILL be wiped)" nix run github:nix-community/disko -- \ --mode disko "./hosts/$config/disk-config.nix" echo ">> installing sops host key so it can decrypt on boot #1" install -Dm600 "$hostkey" /mnt/etc/ssh/ssh_host_ed25519_key install -Dm644 "$hostkey.pub" /mnt/etc/ssh/ssh_host_ed25519_key.pub echo ">> nixos-install .#$config into /mnt" nixos-install --root /mnt --flake ".#$config" else # Stage the pre-generated SSH host key so sops can decrypt on boot #1. stage="$(mktemp -d)" trap 'rm -rf "$stage"' EXIT install -Dm600 "$hostkey" "$stage/etc/ssh/ssh_host_ed25519_key" install -Dm644 "$hostkey.pub" "$stage/etc/ssh/ssh_host_ed25519_key.pub" echo ">> nixos-anywhere .#$config onto root@$host (OS disk WILL be wiped)" anywhere=(--flake ".#$config" --extra-files "$stage" --generate-hardware-config nixos-generate-config "./hosts/$config/hardware-configuration.nix" --target-host "root@$host") # nixos-anywhere's --env-password reads root's ssh password from $SSHPASS # (it ships its own sshpass), so a vault hit skips the ssh-copy-id prompt. # Exported rather than `env SSHPASS=...` for consistency with `kexec`; # see the note there — it's a marginal win, not a leak fix. root_item="${HOMELAB_PASS_ROOT_ITEM:-root@$config}" root_pw="$(proton_pass_password "$root_item" || true)" if [ -n "$root_pw" ]; then echo ">> root ssh password from Proton Pass ($root_item)" export SSHPASS="$root_pw" unset root_pw nix run github:nix-community/nixos-anywhere -- \ --env-password "${anywhere[@]}" unset SSHPASS else nix run github:nix-community/nixos-anywhere -- "${anywhere[@]}" fi fi ;; switch|boot|test) config="${2:-}"; host="${3:-}" { [ -n "$config" ] && [ -n "$host" ]; } || die "usage: ./deploy $cmd " require_tracked "$config" echo ">> nixos-rebuild $cmd .#$config on darman@$host" # --ask-sudo-password, not the deprecated --use-remote-sudo: common.nix sets # security.sudo.wheelNeedsPassword = true, and --use-remote-sudo only # prefixes with sudo without ever prompting. Asks for darman's password # (the darman_password hash in each host's sops file). rebuild=(nix run nixpkgs#nixos-rebuild -- "$cmd" --flake ".#$config" --target-host "darman@$host" --ask-sudo-password) item="${HOMELAB_PASS_ITEM:-darman@$config}" pw="$(proton_pass_password "$item" || true)" if [ -n "$pw" ] && command -v setsid >/dev/null 2>&1; then # nixos-rebuild prompts with getpass(), which reads /dev/tty and ignores a # piped stdin. setsid drops the controlling terminal, so getpass falls back # to stdin and takes the vault password (it warns about echo — harmless, # nothing is echoed since the password never reaches the terminal). # # Caveat of dropping the tty: EVERY prompt in the subtree now reads this # stdin, not just the sudo one. Feed the line a few times so a retry or a # second sudo ask doesn't hit EOF and hang. Anything else that prompts # (an ssh key passphrase, a host-key confirmation) will still fail — fix # those out of band rather than by feeding more lines here. echo ">> sudo password from Proton Pass ($item)" printf '%s\n%s\n%s\n' "$pw" "$pw" "$pw" | setsid -w "${rebuild[@]}" else "${rebuild[@]}" fi unset pw ;; image|flash) if [ "$cmd" = flash ]; then config="${2:-}"; dev="${3:-}" { [ -n "$config" ] && [ -n "$dev" ]; } \ || die "usage: ./deploy flash (e.g. /dev/sdX)" # Validate the device and the tools BEFORE building, so a bad `flash` # fails fast instead of dying halfway through writing the card. [ -b "$dev" ] || die "$dev is not a block device" need zstdcat; need dd; need lsblk else config="${2:-}"; [ -n "$config" ] || die "usage: ./deploy image " fi require_tracked "$config" echo ">> building SD image for .#$config (aarch64 needs qemu binfmt)" nix build ".#nixosConfigurations.$config.config.system.build.sdImage" -o result-sd img="$(one_match 'SD image' result-sd/sd-image/*.img.zst)" echo ">> image: $img" [ "$cmd" = image ] && exit 0 echo ">> TARGET DEVICE — everything on it will be ERASED:" lsblk -o NAME,SIZE,MODEL,TRAN,MOUNTPOINTS "$dev" read -rp ">> type 'yes' to write $config to $dev: " ok [ "$ok" = yes ] || die "aborted" zstdcat "$img" | sudo dd of="$dev" bs=4M status=progress oflag=sync sync # If this config has a dedicated sops age key, drop it on the ROOT ext4 # partition at /var/lib/sops-nix/age.txt so sops decrypts on first boot. # (The Pi's vfat partition isn't mounted at runtime, so the key can't live # there.) Key stays off-repo, out of the nix store, and out of the image. keyfile="$HOME/.config/homelab/$config/age.txt" if [ -f "$keyfile" ]; then echo ">> installing sops age key onto the root partition" sudo partprobe "$dev" 2>/dev/null || sudo blockdev --rereadpt "$dev" 2>/dev/null || true sudo udevadm settle 2>/dev/null || true # Largest ext4 partition = the NixOS root. -P (KEY="value" pairs) instead # of a columnar listing: an empty FSTYPE collapses under whitespace # splitting and shifts every later field. rootpart="$(lsblk -bPo PATH,FSTYPE,SIZE "$dev" \ | sed -n 's/^PATH="\([^"]*\)" FSTYPE="ext4" SIZE="\([0-9]*\)"$/\2 \1/p' \ | sort -rn | head -1 | cut -d' ' -f2)" [ -n "$rootpart" ] || die "no ext4 root partition found on $dev — place $keyfile at /var/lib/sops-nix/age.txt manually" mnt="$(mktemp -d)" sudo mount "$rootpart" "$mnt" sudo install -Dm600 "$keyfile" "$mnt/var/lib/sops-nix/age.txt" sudo sync sudo umount "$mnt"; rmdir "$mnt" echo ">> age key installed (/var/lib/sops-nix/age.txt)" fi echo ">> done — insert the card into the Pi and boot." ;; *) die "unknown command '$cmd' (kexec|kexec-local|install|switch|boot|test|image|flash)" ;; esac