#!/usr/bin/env bash
# Deploy a NixOS host from this flake. ALL arguments are mandatory (no defaults).
#
#   ./deploy kexec   <config> <host>  headless kexec into a RAM installer, for a
#                                     read-only-root box (ZimaOS) where
#                                     nixos-anywhere can't ssh-copy-id. Ships our
#                                     SSH login key. Then run `install`. <config>
#                                     is only used to look up the vault item.
#   ./deploy kexec-local [--yes]      kexec THIS machine into the RAM installer,
#                                     no ssh/second machine involved. Run as root,
#                                     locally, on the box you're installing onto.
#                                     Disks are untouched; console drops for
#                                     ~1-2 min then comes back as the installer.
#                                     Prompts for confirmation (--yes skips it),
#                                     because run on the wrong terminal this
#                                     kexecs your laptop. TMPDIR (default
#                                     /var/tmp) must be exec-capable and hold
#                                     ~3x the tarball.
#                                     Then run `install <config> localhost`.
#   ./deploy install <config> <host> [--yes]
#                                     first install. Wipes the OS disk. Ships the
#                                     host's sops key. <host>=localhost/127.0.0.1
#                                     skips nixos-anywhere/ssh and runs disko +
#                                     nixos-install directly against /mnt — but
#                                     ONLY once actually inside a live installer
#                                     (hostname nixos-installer, from kexec, or
#                                     homelab-installer, from installer-iso).
#                                     Run from the REAL running OS instead (e.g.
#                                     a box where kexec-local doesn't work),
#                                     it builds installer-iso, stages its
#                                     kernel/initrd + the host key on the boot
#                                     partition and the iso file on a non-OS-disk
#                                     partition, sets a systemd-boot one-shot
#                                     entry with homelab.install=<config> +
#                                     homelab.keypart=<PARTUUID> on its kernel
#                                     cmdline, and reboots — a real ACPI reboot,
#                                     not a kexec jump. The booted installer's
#                                     homelab-auto-install.service reads those
#                                     cmdline params, picks the host key back up
#                                     and re-runs this exact command itself once
#                                     its repo checkout (homelab-checkout.service)
#                                     succeeds, finishing the install unattended.
#                                     It confirms before rebooting; --yes skips
#                                     that (it is what the ISO passes itself).
#                                     See CLAUDE.md.
#   ./deploy switch  <config> <host>  rebuild + activate on a running host.
#   ./deploy boot    <config> <host>  stage for next boot, don't activate now.
#   ./deploy test    <config> <host>  activate without adding a boot entry.
#   ./deploy image   <config>         build an SD-card image (e.g. rpi mercury).
#   ./deploy flash   <config> <dev>   build SD image, write to <dev>, and (if
#                                     ~/.config/homelab/<config>/age.txt exists)
#                                     drop the sops key on its boot partition.
#
# <config> = a nixosConfigurations name (e.g. jupiter, vps). Its pre-generated
# SSH host key must be at ~/.config/homelab/<config>/ssh_host_ed25519_key.
#
# Runs from a non-NixOS host too (nixos-rebuild / nixos-anywhere via `nix run`).
#
# Password prompts are auto-filled from the "HomeLab" Proton Pass vault when
# `pass-cli` is installed and logged in; otherwise every command prompts exactly
# as before. Both items are keyed by <config>, never by <host>: the address is
# incidental (DHCP, a new box, localhost) while the config name is the stable
# identity of the machine being built.
#   darman@<config>   darman's sudo password  (switch/boot/test)
#   root@<config>     root's ssh password     (kexec/install)
# Override with HOMELAB_PASS_ITEM / HOMELAB_PASS_ROOT_ITEM / HOMELAB_PASS_VAULT.
set -euo pipefail
shopt -s nullglob

# Captured before anything shifts/parses $@, so require_root() below can
# re-exec the ORIGINAL invocation under sudo — inside a function, "$@"/"$1"
# refer to the function's own args (empty here), not the script's, so this
# has to be a global array instead of relying on positional-parameter scoping.
SCRIPT_ARGS=("$@")

# Locate the repo root (flake dir) regardless of where this script lives on disk.
SCRIPT_PATH="$(realpath "$0")"           # absolute — "$0" itself may be relative,
                                          # and require_root() re-execs after cd "$REPO"
SCRIPT_DIR="$(dirname "$SCRIPT_PATH")"
REPO="$(git -C "$SCRIPT_DIR" rev-parse --show-toplevel 2>/dev/null || dirname "$SCRIPT_DIR")"
cd "$REPO"
export PATH="/nix/var/nix/profiles/default/bin:$PATH"

# Off-repo material keyed by <config>: pre-generated SSH host keys (install)
# and per-config sops age keys (flash).
#
# Resolved defensively rather than as a bare $HOME, because this script also
# runs from installer-iso's homelab-auto-install.service, and systemd does not
# set $HOME for a system service without User= (systemd.exec(5):
# SetLoginEnvironment= "defaults to true if User=, DynamicUser= or PAMName= are
# set, false otherwise"). Under `set -u` that aborted the whole unattended run
# with an "unbound variable" that read like a bug in this script.
KEYDIR="${HOMELAB_KEY_DIR:-${HOME:-/root}/.config/homelab}"

die() { echo "error: $*" >&2; exit 1; }

need() { command -v "$1" >/dev/null 2>&1 || die "missing required tool: $1"; }

# Self-elevate instead of dying: re-exec this exact invocation under sudo.
# -E preserves the environment (HOMELAB_* overrides, Proton Pass vault vars)
# across the re-exec. A no-op once already root.
require_root() {
  [ "$(id -u)" = 0 ] && return 0
  echo ">> $1 needs root — re-executing under sudo" >&2
  # $KEYDIR is derived from $HOME, and whether sudo carries $HOME across
  # depends on the local sudoers policy (env_reset/always_set_home). Pin the
  # resolved value so the re-exec looks for host keys where the invoking user
  # has them, not under /root.
  export HOMELAB_KEY_DIR="$KEYDIR"
  exec sudo -E -- "$SCRIPT_PATH" "${SCRIPT_ARGS[@]}"
}

# Exactly one path matching a glob, or die. `ls glob | head -1` silently yields
# an empty string when nothing matches (head exits 0, so set -e never fires) and
# the failure only surfaces later as a confusing tar/dd error.
one_match() {
  local what="$1"; shift
  local f=("$@")                       # caller expands the glob (nullglob is on)
  [ "${#f[@]}" -gt 0 ] || die "no $what found — did the build actually produce one?"
  # Say so instead of silently taking [0]: a stale result-sd/ symlink from an
  # earlier config is exactly how you flash the wrong image without a word.
  [ "${#f[@]}" -eq 1 ] \
    || echo ">> warning: ${#f[@]} candidates for $what, using ${f[0]} (rm the stale ones)" >&2
  printf '%s\n' "${f[0]}"
}

# Every whole-disk device backing a block device or a mounted path, one per
# line. LVM/RAID/LUKS can sit on several at once (verified on terra:
# /mnt/ssd_01 -> sdd AND sde), so a single lookup is not enough. Empty output
# means "could not determine" — which callers must treat as unsafe, not as OK.
disks_backing() {
  lsblk -rnso NAME,TYPE "$1" 2>/dev/null | awk '$2 == "disk" { print "/dev/" $1 }'
}

# Label of the temporary UEFI boot entry arm_efi_bootnext() creates. Also the
# key the ISO uses to delete it again once it has booted (see flake.nix).
EFI_LABEL="Homelab Installer"

# Boot numbers of every UEFI entry with exactly this label, one per line.
# efibootmgr prints  `Boot0002* Limine<TAB>HD(1,GPT,...)/\EFI\...`, so the
# label runs from past the "Boot####* " prefix up to the first TAB.
# (Character classes spelled out rather than {4}: mawk predates ERE intervals.)
efi_entries_named() {
  efibootmgr 2>/dev/null | awk -v want="$1" '
    /^Boot[0-9A-Fa-f][0-9A-Fa-f][0-9A-Fa-f][0-9A-Fa-f]/ {
      num = substr($0, 5, 4)
      rest = substr($0, 9)
      sub(/^\*/, "", rest); sub(/^ +/, "", rest)
      split(rest, parts, "\t")
      if (parts[1] == want) print num
    }'
}

# Arm a genuine one-shot boot of the staged installer WITHOUT any help from the
# bootloader: create a UEFI boot entry that EFI-stub-boots the kernel straight
# off the ESP, and point BootNext at it.
#
# Needed because "boot this once, then go back to normal" is not something
# every bootloader can do. systemd-boot has it; terra's CachyOS runs Limine,
# which reports `One-shot entry control: ✗` and has no equivalent, and whose
# limine.conf is regenerated by pacman hooks anyway. BootNext is a firmware
# feature, so it works underneath all of them — and the firmware clears it
# after that one boot, which is what keeps the "a failed attempt still comes
# back on the normal bootloader" property that makes this safe to try.
arm_efi_bootnext() {
  local esp="$1" cmdline="$2"
  local esp_src esp_disk esp_part num n
  need efibootmgr
  esp_src="$(findmnt -no SOURCE --nofsroot --target "$esp")" \
    || die "couldn't resolve $esp to a device"
  esp_disk="$(disks_backing "$esp_src" | head -1 || true)"
  esp_part="$(cat "/sys/class/block/$(basename "$esp_src")/partition" 2>/dev/null || true)"
  { [ -n "$esp_disk" ] && [ -n "$esp_part" ]; } \
    || die "couldn't work out the disk + partition number of the ESP ($esp -> $esp_src)"

  # Clear anything left by an earlier attempt first, so repeated runs don't
  # slowly fill NVRAM with dead entries pointing at a wiped partition.
  for n in $(efi_entries_named "$EFI_LABEL"); do
    echo ">> removing stale UEFI entry Boot$n ($EFI_LABEL)"
    efibootmgr -q -B -b "$n"
  done

  # --create-only, NOT --create: the latter also pushes the entry to the front
  # of BootOrder, which would make a wiped installer the permanent default if
  # anything went wrong. This way the entry is reachable through BootNext and
  # nothing else, i.e. exactly once.
  #
  # The EFI stub loads `initrd=` off the volume it was itself loaded from, so
  # the path is relative to the ESP root and uses backslashes.
  efibootmgr -q --create-only --disk "$esp_disk" --part "$esp_part" \
    --label "$EFI_LABEL" \
    --loader '\homelab-installer\bzImage' \
    --unicode "initrd=\\homelab-installer\\initrd $cmdline"

  num="$(efi_entries_named "$EFI_LABEL" | head -1)"
  [ -n "$num" ] || die "efibootmgr did not create a '$EFI_LABEL' entry"
  efibootmgr -q --bootnext "$num"
  echo ">> UEFI BootNext -> Boot$num ($EFI_LABEL); BootOrder untouched"
}

# Sets tb / cpio / bbox — the kexec tarball plus the static cpio+gzip that
# kexec-run.sh needs on PATH to rebuild its initrd.
#
# HOMELAB_KEXEC_TARBALL (with _CPIO / _GZIP) short-circuits the build and uses a
# prebuilt installer instead. That lets the VM test in flake.nix drive this
# script offline, and lets you re-kexec a box without rebuilding ~500MB.
kexec_artifacts() {
  if [ -n "${HOMELAB_KEXEC_TARBALL:-}" ]; then
    tb="$HOMELAB_KEXEC_TARBALL"
    [ -f "$tb" ] || die "HOMELAB_KEXEC_TARBALL=$tb is not a file"
    cpio="${HOMELAB_KEXEC_CPIO:-$(command -v cpio || true)}"
    bbox="${HOMELAB_KEXEC_GZIP:-$(command -v gzip || true)}"
    { [ -n "$cpio" ] && [ -n "$bbox" ]; } \
      || die "set HOMELAB_KEXEC_CPIO / HOMELAB_KEXEC_GZIP, or put cpio+gzip on PATH"
    echo ">> using prebuilt kexec installer: $tb"
  else
    need nix
    echo ">> building kexec installer + static tools"
    nix build .#nixosConfigurations.kexec.config.system.build.kexecInstallerTarball \
      -o result-kexec
    tb="$(one_match 'kexec tarball' result-kexec/*.tar.gz)"
    cpio="$(nix build --no-link --print-out-paths nixpkgs#pkgsStatic.cpio)/bin/cpio"
    bbox="$(nix build --no-link --print-out-paths nixpkgs#pkgsStatic.busybox)/bin/busybox"
  fi
}

# True inside one of the throwaway live-installer environments this repo
# produces (kexec's nixos-installer, or installer-iso's homelab-installer) —
# i.e. `install <config> localhost` should wipe/install right here. False on
# any real running OS, where the same command instead means "prepare and
# reboot into an installer for THIS box" (see local_install_prepare_and_reboot).
is_live_installer() {
  case "$(uname -n)" in
    nixos-installer | homelab-installer) return 0 ;;
    *) return 1 ;;
  esac
}

# `install <config> localhost` run on a REAL running OS (not already inside a
# live installer): builds installer-iso, stages its kernel/initrd + the host's
# pre-generated ssh key on the boot partition and the iso file on a non-OS
# disk, points a systemd-boot one-shot entry at them with
# homelab.install=<config> + homelab.keypart=<PARTUUID> on the kernel cmdline,
# and reboots — a real ACPI reboot through firmware POST, deliberately NOT a
# kexec jump (see terra's kexec-local gotcha in CLAUDE.md). The booted
# installer's homelab-auto-install.service reads those params, picks the host
# key back up and re-runs this exact `install <config> localhost` command
# itself (now genuinely inside the installer) once homelab-checkout.service has
# fetched the repo, finishing the job unattended.
local_install_prepare_and_reboot() {
  local config="$1" hostkey="$2" assume_yes="$3"
  require_root "preparing a local reinstall"
  [ -d /sys/firmware/efi ] || die "not booted UEFI — the one-shot boot entry needs systemd-boot"
  need bootctl
  need nix
  need lsblk
  need findmnt
  need awk
  need realpath
  need stat
  need df

  # Where to stage the installer, and how to make the box boot it exactly once.
  #
  # systemd-boot keeps its entries on $BOOT — the XBOOTLDR partition when there
  # is one, the ESP otherwise — which is not always /boot. Hardcoding /boot on
  # a box that mounts its ESP elsewhere just creates a directory on the root
  # filesystem and then reboots into an entry the firmware never sees.
  #
  # No systemd-boot (terra's CachyOS runs Limine) means no `bootctl set-oneshot`,
  # so fall back to the firmware's own BootNext — see arm_efi_bootnext(). That
  # path EFI-stub-boots the kernel directly, which requires it to sit on the ESP
  # itself rather than on a separate XBOOTLDR.
  local boot boot_mode esp
  esp="$(bootctl --print-esp-path 2>/dev/null)" \
    || die "bootctl couldn't locate the ESP — is this box actually UEFI-booted?"
  boot="$(bootctl --print-boot-path 2>/dev/null || echo "$esp")"
  if [ -d "$boot/loader/entries" ]; then
    boot_mode=systemd-boot
  else
    boot_mode=efi-bootnext
    boot="$esp"
    need efibootmgr
    echo ">> no systemd-boot entries at $boot/loader/entries — arming the firmware's"
    echo "   own BootNext instead (bootloader in charge here: $(bootctl status 2>/dev/null | awk '/Product:/ {$1=""; print substr($0,2); exit}' || echo unknown))"
  fi

  # No default/auto-picked location — the wrong disk here is destroyed
  # mid-install (see the OS-disk check below), so this always asks rather
  # than guessing. HOMELAB_INSTALLER_STAGE_DIR skips the prompt for scripted
  # use, but is otherwise just as explicit a choice as typing it in.
  local stagedir="${HOMELAB_INSTALLER_STAGE_DIR:-}"
  if [ -z "$stagedir" ]; then
    echo ">> currently mounted filesystems:"
    lsblk -o NAME,SIZE,FSTYPE,MOUNTPOINT
    read -rp ">> path to stage the installer iso on (must NOT be on the OS disk being wiped): " stagedir
  fi
  [ -n "$stagedir" ] || die "no staging path given"
  [ -d "$stagedir" ] \
    || die "staging dir $stagedir doesn't exist — needs to be an existing partition that is NOT the OS disk being wiped"
  # Absolute + symlink-free: findiso= below is computed by stripping the
  # mountpoint prefix off this, and a relative answer at the prompt would
  # produce a path the initrd can never resolve.
  stagedir="$(realpath "$stagedir")"

  # Refuse if the staging partition turns out to live on the same disk
  # disko is about to wipe — the iso file (and the running installer
  # loopback-mounted from it) would be destroyed mid-install.
  local osdisk osdisk_real stage_src stage_fstype stage_disks d
  osdisk="$(nix eval --raw ".#nixosConfigurations.$config.config.disko.devices.disk" \
    --apply 'd: (builtins.head (builtins.attrValues d)).device' 2>/dev/null)" \
    || die "couldn't read the OS disk device from hosts/$config/disk-config.nix"
  osdisk_real="$(readlink -f "$osdisk")"

  # --nofsroot matters: on btrfs, findmnt prints the subvolume as
  # `/dev/sdb2[/@]`, which is not a path lsblk can open. Without it the lookup
  # came back empty and the guard below was skipped entirely — i.e. it silently
  # allowed staging on the very disk about to be wiped. terra's current
  # CachyOS root is exactly that layout.
  stage_src="$(findmnt -no SOURCE --nofsroot --target "$stagedir")" \
    || die "$stagedir doesn't resolve to a mounted filesystem"
  # `|| true` so the explicit check below is what reports the problem: lsblk
  # exits nonzero on a device it can't parse, and under `set -e` + pipefail a
  # bare assignment from a failing substitution kills the script silently,
  # right past the fail-closed message.
  stage_disks="$(disks_backing "$stage_src" || true)"
  # Fail closed. "Couldn't determine the disk" is not "different disk".
  [ -n "$stage_disks" ] \
    || die "couldn't determine which physical disk $stagedir ($stage_src) is on — refusing to guess, since being wrong destroys the install mid-flight"
  for d in $stage_disks; do
    if [ "$d" = "$osdisk_real" ]; then
      die "$stagedir is on the OS disk ($osdisk -> $osdisk_real) that install would wipe — re-run and pick a different disk"
    fi
  done

  # stage-1 resolves findiso= by mounting each blkid-visible partition and
  # testing `-e /findiso$isoPath` (nixos/modules/system/boot/stage-1-init.sh).
  # For btrfs it mounts the volume's TOP level, so a path that lives inside a
  # subvolume (/@/...) is simply not there and the box boots to an emergency
  # shell — after it has already rebooted out of the working OS.
  stage_fstype="$(findmnt -no FSTYPE --target "$stagedir")"
  [ "$stage_fstype" != btrfs ] \
    || die "$stagedir is btrfs: findiso= mounts the volume's top level, so a path inside a subvolume never resolves. Stage on a non-btrfs partition (ext4/vfat/ntfs)."

  # PARTUUID of the staging partition. Handed to the installer as
  # homelab.logpart= so it can mount this partition rw and persist its whole
  # run — disko + nixos-install output included — to a file next to the iso.
  # This partition is on a DIFFERENT disk from the one disko wipes (guarded
  # above), so unlike $boot it SURVIVES the install: a failed attempt otherwise
  # leaves nothing to debug, its journal having died on tmpfs at the reboot.
  # Best-effort — an LVM/mdraid stage_src has no PARTUUID, in which case logging
  # is simply skipped rather than blocking the install.
  local stage_partuuid
  stage_partuuid="$(lsblk -no PARTUUID "$stage_src" 2>/dev/null | head -1 | tr -d ' ' || true)"

  # Last chance to back out. This is the most destructive command in the
  # script — it reboots the machine you are typing at and the wipe that
  # follows is unattended — so it confirms just like `flash` and `kexec-local`
  # do, both of which are less final than this.
  if [ "$assume_yes" != "--yes" ]; then
    echo ">> about to REINSTALL this machine from scratch:"
    echo "   hostname:  $(uname -n)"
    echo "   config:    $config"
    echo "   OS disk:   $osdisk"
    echo "              -> $osdisk_real  ** WIPED, unattended, after the reboot **"
    # Unquoted on purpose: collapses the one-per-line list onto one line.
    echo "   staging:   $stagedir  (on $(echo $stage_disks))"
    if [ -n "$stage_partuuid" ]; then
      echo "   logs:      $stagedir/homelab-install-$config.log  (on the staging disk — survives the wipe)"
    else
      echo "   logs:      (none — $stagedir has no PARTUUID; installer output won't survive the wipe)"
    fi
    echo "   one-shot:  $boot_mode"
    read -rp ">> type 'yes' to build the installer, reboot into it and wipe $osdisk_real: " ok
    [ "$ok" = yes ] || die "aborted"
  fi

  echo ">> building installer-iso (kernel + initrd + iso image)"
  local kernel initrd isodir iso toplevel mnt_point iso_relpath boot_src boot_partuuid
  kernel="$(nix build --no-link --print-out-paths .#nixosConfigurations.installer-iso.config.system.build.kernel)/bzImage"
  initrd="$(nix build --no-link --print-out-paths .#nixosConfigurations.installer-iso.config.system.build.initialRamdisk)/initrd"
  isodir="$(nix build --no-link --print-out-paths .#nixosConfigurations.installer-iso.config.system.build.isoImage)"
  iso="$(one_match 'installer iso' "$isodir"/iso/*.iso)"
  # The live ISO's root is a tmpfs; stage 1 finds the real system's init via
  # init=<toplevel>/init, which the grub/isolinux menu supplies on a normal
  # boot (iso-image.nix). EFI-stub-booting our own cmdline, we must pass it too
  # — omit it and stage 1 loop-mounts the iso fine, then dies on
  # "stage 2 init script (/mnt-root//init) not found".
  toplevel="$(nix build --no-link --print-out-paths .#nixosConfigurations.installer-iso.config.system.build.toplevel)"

  # A short write is not visible until the reboot, when findiso finds a
  # truncated iso and drops to an emergency shell. Check first — `install`
  # prints no progress and the iso is ~1GB.
  local need_stage need_boot avail_stage avail_boot
  need_stage="$(stat -Lc %s "$iso")"
  need_boot="$(( $(stat -Lc %s "$kernel") + $(stat -Lc %s "$initrd") + $(stat -Lc %s "$hostkey") ))"
  avail_stage="$(df -B1 --output=avail "$stagedir" | tail -1 | tr -d ' ')"
  avail_boot="$(df -B1 --output=avail "$boot" | tail -1 | tr -d ' ')"
  [ "$avail_stage" -ge "$(( need_stage + 64 * 1024 * 1024 ))" ] \
    || die "$stagedir has $(( avail_stage / 1024 / 1024 ))MB free, the iso needs $(( need_stage / 1024 / 1024 ))MB — pick another partition"
  [ "$avail_boot" -ge "$(( need_boot + 16 * 1024 * 1024 ))" ] \
    || die "$boot has $(( avail_boot / 1024 / 1024 ))MB free, kernel+initrd need $(( need_boot / 1024 / 1024 ))MB"

  echo ">> staging kernel/initrd/host key on $boot, iso image on $stagedir"
  install -Dm644 "$kernel" "$boot/homelab-installer/bzImage"
  install -Dm644 "$initrd" "$boot/homelab-installer/initrd"
  install -Dm644 "$iso"    "$stagedir/homelab-installer.iso"

  # The ISO is built from a PUBLIC repo and deliberately carries no
  # credentials, so the host key has to travel with the staged installer or
  # the auto-install run has nothing to seed /etc/ssh with — and without that,
  # sops can't decrypt on boot #1, /etc/shadow gets written once with a locked
  # darman, and no later `deploy switch` can fix it (README).
  #
  # $boot lives on the OS disk, so disko destroys this copy minutes later. The
  # mode is advisory on vfat (permissions come from the mount's fmask, 0077 on
  # a NixOS/systemd-boot ESP) — it is the wipe, not the mode, doing the work.
  install -Dm600 "$hostkey"     "$boot/homelab-installer/ssh_host_ed25519_key"
  install -Dm644 "$hostkey.pub" "$boot/homelab-installer/ssh_host_ed25519_key.pub"
  boot_src="$(findmnt -no SOURCE --nofsroot --target "$boot")" \
    || die "couldn't resolve $boot to a device"
  boot_partuuid="$(lsblk -no PARTUUID "$boot_src" 2>/dev/null | head -1 | tr -d ' ' || true)"
  [ -n "$boot_partuuid" ] \
    || die "couldn't read a PARTUUID for $boot ($boot_src) — the installer needs it to find the host key"

  # findiso= is a path relative to whatever partition the initrd finds it on
  # (it mounts every blkid-visible partition looking for it), not to `/`, if
  # $stagedir is a subdirectory of a bigger filesystem rather than a mountpoint
  # itself. It must KEEP its leading slash: stage-1 tests `-e /findiso$isoPath`,
  # so a bare `var/tmp/x.iso` becomes `/findisovar/tmp/x.iso` and never matches.
  # Prefixing then squeezing handles both ends: stagedir == the mountpoint
  # (strip leaves "") and mnt_point == "/" (strip leaves a relative path).
  mnt_point="$(findmnt -no TARGET --target "$stagedir")"
  iso_relpath="$(printf '/%s/%s' "${stagedir#"$mnt_point"}" homelab-installer.iso | tr -s /)"

  # Identical either way — only the mechanism that gets the kernel booted with
  # it differs.
  # root=LABEL=<volumeID> matches what the ISO menu passes; findiso overwrites
  # /dev/root with the loop-mounted iso regardless, but keep it honest.
  # boot.shell_on_fail gives a shell instead of the reboot/ignore prompt if
  # stage 1 ever fails again. init= is the one that actually made this work.
  local cmdline volumeID
  volumeID="$(nix eval --raw .#nixosConfigurations.installer-iso.config.isoImage.volumeID)"
  cmdline="init=$toplevel/init nohibernate root=LABEL=$volumeID boot.shell_on_fail loglevel=4 lsm=landlock,yama,bpf findiso=$iso_relpath homelab.install=$config homelab.keypart=$boot_partuuid"
  # Only when the staging partition has a PARTUUID (see stage_partuuid). Points
  # the installer's homelab-auto-install.service at the surviving disk to log to.
  [ -n "$stage_partuuid" ] && cmdline="$cmdline homelab.logpart=$stage_partuuid"

  case "$boot_mode" in
    systemd-boot)
      cat >"$boot/loader/entries/homelab-installer.conf" <<EOF
title Homelab Installer ($config, findiso)
linux /homelab-installer/bzImage
initrd /homelab-installer/initrd
options $cmdline
EOF
      bootctl set-oneshot homelab-installer.conf
      echo ">> systemd-boot one-shot entry armed"
      ;;
    efi-bootnext)
      arm_efi_bootnext "$boot" "$cmdline"
      ;;
  esac

  echo ">> rebooting into the installer — it will finish this install itself"
  systemctl reboot
}

# Flakes only see git-tracked files: an untracked hosts/<config>/ is silently
# invisible to `nix build`/`nixos-install`, which then fails obscurely or builds
# a stale config. Check before doing anything destructive.
require_tracked() {
  local config="$1" cfgfile="hosts/$1/configuration.nix" f
  [ -e "$cfgfile" ] || die "no $cfgfile in the repo"
  # No .git at all (e.g. a tarball export of the repo, no working tree), or no
  # git binary, means there's nothing that CAN be untracked — nothing to check.
  # Only skip on that, not on any other git failure.
  command -v git >/dev/null 2>&1 || return 0
  git -C "$REPO" rev-parse --is-inside-work-tree >/dev/null 2>&1 || return 0
  # Every .nix in hosts/<config>/, not just configuration.nix: an untracked
  # disk-config.nix is exactly as invisible to the flake, and it is the file
  # that decides which disk gets wiped.
  for f in "hosts/$config"/*.nix; do
    git -C "$REPO" ls-files --error-unmatch "$f" >/dev/null 2>&1 \
      || die "$f is untracked — 'git add hosts/$config' first (flakes ignore untracked files)"
  done
}

# The password field of a Proton Pass item ("--field password" prints the bare
# value, one line), or empty if pass-cli is missing / logged out / has no such
# item — every caller then falls back to the normal interactive prompt.
#
# Resolve the title to an item id among ACTIVE items first, because `item view
# --item-title` has no state filter: Proton Pass keeps deleted items in the
# trash, and if a trashed item shares the title, view can match THAT one and
# return an empty password with exit 0. Empty is indistinguishable from "no such
# item", so the only symptom is a silent fall back to the interactive prompt
# even though the vault clearly holds the entry. (Hit for real on darman@neptun,
# which had an Active and a Trashed copy.)
proton_pass_password() {
  local title="$1" vault="${HOMELAB_PASS_VAULT:-HomeLab}" id pw
  command -v pass-cli >/dev/null 2>&1 || return 0

  # Lines look like:  - [ITEM_ID]: the title (state=Active)
  id="$(pass-cli item list --vault-name "$vault" --filter-state active \
          --output human 2>/dev/null \
        | awk -v t="$title" '
            { i = index($0, "]: "); if (i == 0) next
              id   = substr($0, 4, i - 4)
              rest = substr($0, i + 3)
              sub(/ \(state=[^)]*\)$/, "", rest)
              if (rest == t) { print id; exit } }' || true)"

  if [ -n "$id" ]; then
    pw="$(pass-cli item view --vault-name "$vault" --item-id "$id" \
            --field password --output human 2>/dev/null | head -1 || true)"
    # Resolved an active item but its password field is blank: worth saying so,
    # otherwise this looks identical to having no vault entry at all.
    [ -n "$pw" ] || echo ">> vault item '$title' resolved but its password field is empty" >&2
  else
    # No active match — fall back to the title lookup so an older pass-cli
    # without --filter-state still works exactly as it used to.
    pw="$(pass-cli item view --vault-name "$vault" --item-title "$title" \
            --field password --output human 2>/dev/null | head -1 || true)"
  fi

  printf '%s' "$pw"
}

# Path to an sshpass binary (system one, else built from nixpkgs). Empty if
# neither is available.
sshpass_bin() {
  command -v sshpass 2>/dev/null && return 0
  nix build --no-link --print-out-paths nixpkgs#sshpass 2>/dev/null \
    | sed 's|$|/bin/sshpass|'
}

cmd="${1:-}"; [ -n "$cmd" ] || die "usage: ./deploy <kexec|kexec-local|install|switch|boot|test|image|flash> ..."

case "$cmd" in
  kexec)
    config="${2:-}"; host="${3:-}"
    { [ -n "$config" ] && [ -n "$host" ]; } || die "usage: ./deploy kexec <config> <host>"

    need ssh; need scp

    # kexec/run rebuilds an initrd with `cpio` + `gzip` from PATH — ZimaOS lacks
    # both. Ship static ones: GNU cpio (reliable -o -H newc), busybox as gzip.
    kexec_artifacts

    # One password prompt: multiplex scp + ssh over a shared control connection.
    cm="/tmp/homelab-cm-%r@%h:%p"
    o=(-o ControlMaster=auto -o "ControlPath=$cm" -o ControlPersist=300 \
       -o StrictHostKeyChecking=accept-new)

    # Root's password from Proton Pass, fed to ssh/scp via sshpass -e. Only the
    # first (master) connection authenticates; the rest ride the control socket.
    #
    # SSHPASS is exported here rather than passed as `env SSHPASS=... sshpass`.
    # Both end up equally safe at rest: `env` execs its target immediately, so
    # the assignment is only in argv for the sub-millisecond before exec, after
    # which /proc/PID/cmdline reads plain `sshpass -e`. Exporting just closes
    # that race window and drops a process. Either way the secret lives in the
    # child's environ, which is readable by the owner and root only.
    sp=()
    root_item="${HOMELAB_PASS_ROOT_ITEM:-root@$config}"
    root_pw="$(proton_pass_password "$root_item" || true)"
    if [ -n "$root_pw" ]; then
      sshpass="$(sshpass_bin || true)"
      if [ -n "$sshpass" ]; then
        export SSHPASS="$root_pw"
        sp=("$sshpass" -e)
        # sshpass drives the password prompt; don't let a key/agent short-circuit
        # into an interactive one for a host that only accepts passwords.
        o+=(-o PreferredAuthentications=password -o PubkeyAuthentication=no)
      else
        echo ">> sshpass unavailable — falling back to the interactive prompt" >&2
      fi
    fi
    unset root_pw

    if [ "${#sp[@]}" -gt 0 ]; then
      echo ">> connecting to root@$host  (password from Proton Pass: $root_item)"
    else
      echo ">> connecting to root@$host  (enter the root password once)"
    fi
    "${sp[@]+"${sp[@]}"}" ssh "${o[@]}" "root@$host" 'mkdir -p /tmp/bin'
    scp "${o[@]}" "$cpio" "root@$host:/tmp/bin/cpio"
    scp "${o[@]}" "$bbox" "root@$host:/tmp/bin/gzip"   # busybox as gzip (argv0)

    echo ">> streaming installer + kexec-ing. SSH drops as the box jumps into the"
    echo "   RAM installer. Disks are untouched."
    rc=0
    ssh "${o[@]}" "root@$host" \
      'chmod +x /tmp/bin/*; mkdir -p /tmp/k && tar -C /tmp/k -xzf - && PATH=/tmp/bin:$PATH /tmp/k/kexec/run' \
      < "$tb" || rc=$?
    # A dropped connection IS the success case here, so this can't be fatal —
    # but a full /tmp, a tar error or a missing static tool looks identical from
    # the outside, so at least surface the code instead of swallowing it.
    [ "$rc" -eq 0 ] || echo ">> ssh exited $rc — expected if the box jumped; suspect this if install can't connect" >&2

    ssh "${o[@]}" -O exit "root@$host" 2>/dev/null || true  # close control socket
    unset SSHPASS

    # NB: no ssh-keygen -R here on purpose. kexec-run.sh copies /etc/ssh/ssh_host_*
    # into the appended initrd and restore-remote-access.nix installs them back
    # into the installer's /etc/ssh, so the host key SURVIVES the jump. Clearing
    # known_hosts would just throw away the TOFU record for no reason.

    echo ">> box is kexec-ing. Wait ~1-2 min for the installer + network, then:"
    echo "   ./deploy install $config $host"
    ;;

  kexec-local)
    # No ssh, no second machine: build the same RAM installer as `kexec`, but
    # run it directly on this box (you're sitting at it). The current shell
    # drops when the kernel switches, same as any reboot — that's expected,
    # not a failure. Disks are untouched; only the running kernel changes.
    #
    # This is a one-way trip on the machine you are typing at, so every check
    # that can fail is done BEFORE the point of no return, and nothing that the
    # jump depends on is cleaned up behind it (see the trap discussion below).
    require_root "kexec-local"

    assume_yes=""
    [ "${2:-}" = "--yes" ] && assume_yes=1

    need tar; need install; need mktemp; need find; need sync; need nohup

    # --- preflight: reasons the jump would fail, checked while it's still safe --
    # CONFIG_KEXEC. Without it kexec --load fails and you've built ~1GB for
    # nothing; with lockdown it fails at load time too (Secure Boot blocks the
    # kexec_load syscall unless the image is signed).
    [ -e /sys/kernel/kexec_loaded ] \
      || die "this kernel has no kexec support (CONFIG_KEXEC) — boot install media instead"
    if [ -r /sys/kernel/security/lockdown ] \
       && ! grep -q '\[none\]' /sys/kernel/security/lockdown 2>/dev/null; then
      die "kernel lockdown is active — kexec_load is blocked. Disable Secure Boot, or boot install media."
    fi

    kexec_artifacts

    # Staging lives on /var/tmp, not /tmp: kexec-run.sh APPENDS a fresh cpio to
    # kexec/initrd in place and runs binaries out of this directory, so it needs
    # real space and exec permission. A tmpfs /tmp is both size-capped (ENOSPC
    # mid-append = a half-written initrd) and frequently noexec.
    stage="$(TMPDIR="${TMPDIR:-/var/tmp}" mktemp -d)"
    trap 'rm -rf "$stage"' EXIT

    # noexec would only surface as a cryptic "Permission denied" on kexec/run.
    printf '#!/bin/sh\nexit 0\n' > "$stage/.execmove"
    chmod +x "$stage/.execmove"
    "$stage/.execmove" 2>/dev/null \
      || die "$stage is mounted noexec — set TMPDIR to an exec-capable filesystem"
    rm -f "$stage/.execmove"

    # Need room for the tarball, its extracted contents, and the appended cpio.
    tb_sz="$(stat -Lc %s "$tb")"
    avail="$(df -B1 --output=avail "$stage" | tail -1 | tr -d ' ')"
    [ "$avail" -ge $(( tb_sz * 3 )) ] \
      || die "only $(( avail / 1024 / 1024 ))MB free on $stage, need ~$(( tb_sz * 3 / 1024 / 1024 ))MB — set TMPDIR elsewhere"

    install -Dm755 "$cpio" "$stage/bin/cpio"
    install -Dm755 "$bbox" "$stage/bin/gzip"   # busybox as gzip (argv0)
    mkdir -p "$stage/k" && tar -C "$stage/k" -xzf "$tb"

    # Fail loudly here rather than with a bare "not found" from kexec/run.
    for f in run kexec bzImage initrd ip; do
      [ -f "$stage/k/kexec/$f" ] || die "kexec tarball is missing kexec/$f — bad build?"
    done

    # The installer root is a RAM disk: the initrd has to fit in memory with
    # room to unpack. Refuse rather than OOM halfway into the new kernel, where
    # there is no way back except the power button.
    img_sz="$(( $(stat -Lc %s "$stage/k/kexec/initrd") + $(stat -Lc %s "$stage/k/kexec/bzImage") ))"
    mem_kb="$(awk '/^MemTotal:/{print $2}' /proc/meminfo)"
    [ "$(( mem_kb * 1024 ))" -ge "$(( img_sz * 3 ))" ] \
      || die "only $(( mem_kb / 1024 ))MB RAM for a $(( img_sz / 1024 / 1024 ))MB RAM-disk installer — too tight, boot install media instead"

    # --- confirmation: this script normally runs on the LAPTOP ------------------
    # A stray `kexec-local` here would jump the machine you develop on, not the
    # target. Name it explicitly so a wrong-terminal mistake is visible.
    echo ">> about to kexec THIS machine:"
    echo "   hostname: $(uname -n)"
    echo "   kernel:   $(uname -r)"
    echo "   root:     $(findmnt -no SOURCE / 2>/dev/null || echo '?')"
    echo "   Disks are untouched; only the running kernel changes. Console drops"
    echo "   for ~1-2 min, then comes back as the installer."
    if [ -z "$assume_yes" ]; then
      read -rp ">> type 'yes' to kexec $(uname -n) into the RAM installer: " ok
      [ "$ok" = yes ] || die "aborted"
    fi

    # kexec -e jumps straight to the new kernel: no unmount, no journal flush,
    # no systemd shutdown. Anything still in page cache is lost and the OS disk
    # is left dirty. Cheap insurance, since the box may yet be rebooted back.
    sync

    echo ">> loading the new kernel"
    if ! PATH="$stage/bin:$PATH" "$stage/k/kexec/run"; then
      rm -rf "$stage"
      die "kexec/run failed — machine untouched, still on the old kernel"
    fi

    # kexec --load succeeded iff the kernel now reports a loaded image. If this
    # is 0 the background `kexec -e` below will do nothing and we'd hang forever
    # waiting for a jump that cannot happen.
    [ "$(cat /sys/kernel/kexec_loaded 2>/dev/null || echo 0)" = 1 ] \
      || { rm -rf "$stage"; die "kexec reported success but no image is loaded — aborting"; }

    # THE trap MUST GO NOW. kexec-run.sh backgrounds `nohup sh -c "sleep 6 &&
    # $SCRIPT_DIR/kexec -e"` and returns immediately, so the binary that
    # performs the jump still has to exist ~6s after this script would normally
    # exit. Letting the EXIT trap rm -rf "$stage" deletes it out from under that
    # sleeping shell and the machine silently never jumps.
    trap - EXIT

    sync
    echo ">> kernel loaded; jumping in ~6s. (staging left at $stage on purpose —"
    echo "   the backgrounded kexec still needs it; it's gone after the jump.)"
    # Outlive the sleep 6 so the jump happens while this script is still alive.
    sleep 60
    die "still here after 60s — the jump did not happen. Check dmesg; the old system is intact."
    ;;

  install)
    config="${2:-}"; host="${3:-}"; assume_yes="${4:-}"
    { [ -n "$config" ] && [ -n "$host" ]; } || die "usage: ./deploy install <config> <host> [--yes]"
    # $KEYDIR, not a bare $HOME — see its definition. This same check runs
    # inside installer-iso, where homelab-auto-install.service has no $HOME and
    # has just dropped the key into /root/.config/homelab/<config>/.
    hostkey="$KEYDIR/$config/ssh_host_ed25519_key"
    [ -f "$hostkey" ] || die "missing host key: $hostkey"
    [ -d "./hosts/$config" ] || die "no ./hosts/$config directory in the repo"
    require_tracked "$config"

    if [ "$host" = "localhost" ] || [ "$host" = "127.0.0.1" ]; then
      if ! is_live_installer; then
        # Not already inside a live installer: build one, stage it, one-shot
        # boot into it, and let it finish this exact command itself. See
        # local_install_prepare_and_reboot above and CLAUDE.md.
        local_install_prepare_and_reboot "$config" "$hostkey" "$assume_yes"
        exit 0
      fi

      # Local install: no ssh, no nixos-anywhere. Run after `kexec-local` (or
      # from a live ISO) so /mnt is free to wipe — this IS the box, no second
      # machine in the loop, so skip straight to disko + nixos-install.
      require_root "local install"
      [ -f "./hosts/$config/disk-config.nix" ] || die "no ./hosts/$config/disk-config.nix"

      echo ">> disko .#$config onto this box's OS disk (WILL be wiped)"
      # `.#disko`, not github:nix-community/disko — the revision comes from this
      # repo's flake.lock rather than upstream master-of-the-day, and resolves
      # from the local store. See the nixos-anywhere input in flake.nix.
      nix run ".#disko" -- \
        --mode disko "./hosts/$config/disk-config.nix"

      echo ">> installing sops host key so it can decrypt on boot #1"
      install -Dm600 "$hostkey"     /mnt/etc/ssh/ssh_host_ed25519_key
      install -Dm644 "$hostkey.pub" /mnt/etc/ssh/ssh_host_ed25519_key.pub

      echo ">> nixos-install .#$config into /mnt"
      nixos-install --root /mnt --flake ".#$config"
    else
      # Stage the pre-generated SSH host key so sops can decrypt on boot #1.
      stage="$(mktemp -d)"
      trap 'rm -rf "$stage"' EXIT
      install -Dm600 "$hostkey"      "$stage/etc/ssh/ssh_host_ed25519_key"
      install -Dm644 "$hostkey.pub"  "$stage/etc/ssh/ssh_host_ed25519_key.pub"

      echo ">> nixos-anywhere .#$config onto root@$host (OS disk WILL be wiped)"
      anywhere=(--flake ".#$config"
                --extra-files "$stage"
                --generate-hardware-config nixos-generate-config "./hosts/$config/hardware-configuration.nix"
                --target-host "root@$host")

      # nixos-anywhere's --env-password reads root's ssh password from $SSHPASS
      # (it ships its own sshpass), so a vault hit skips the ssh-copy-id prompt.
      # Exported rather than `env SSHPASS=...` for consistency with `kexec`;
      # see the note there — it's a marginal win, not a leak fix.
      root_item="${HOMELAB_PASS_ROOT_ITEM:-root@$config}"
      root_pw="$(proton_pass_password "$root_item" || true)"
      if [ -n "$root_pw" ]; then
        echo ">> root ssh password from Proton Pass ($root_item)"
        export SSHPASS="$root_pw"
        unset root_pw
        nix run ".#nixos-anywhere" -- \
          --env-password "${anywhere[@]}"
        unset SSHPASS
      else
        nix run ".#nixos-anywhere" -- "${anywhere[@]}"
      fi
    fi
    ;;

  switch|boot|test)
    config="${2:-}"; host="${3:-}"
    { [ -n "$config" ] && [ -n "$host" ]; } || die "usage: ./deploy $cmd <config> <host>"
    require_tracked "$config"

    echo ">> nixos-rebuild $cmd .#$config on darman@$host"
    # --ask-sudo-password, not the deprecated --use-remote-sudo: common.nix sets
    # security.sudo.wheelNeedsPassword = true, and --use-remote-sudo only
    # prefixes with sudo without ever prompting. Asks for darman's password
    # (the darman_password hash in each host's sops file).
    rebuild=(nix run nixpkgs#nixos-rebuild -- "$cmd"
             --flake ".#$config"
             --target-host "darman@$host"
             --ask-sudo-password)

    item="${HOMELAB_PASS_ITEM:-darman@$config}"
    pw="$(proton_pass_password "$item" || true)"
    if [ -n "$pw" ] && command -v setsid >/dev/null 2>&1; then
      # nixos-rebuild prompts with getpass(), which reads /dev/tty and ignores a
      # piped stdin. setsid drops the controlling terminal, so getpass falls back
      # to stdin and takes the vault password (it warns about echo — harmless,
      # nothing is echoed since the password never reaches the terminal).
      #
      # Caveat of dropping the tty: EVERY prompt in the subtree now reads this
      # stdin, not just the sudo one. Feed the line a few times so a retry or a
      # second sudo ask doesn't hit EOF and hang. Anything else that prompts
      # (an ssh key passphrase, a host-key confirmation) will still fail — fix
      # those out of band rather than by feeding more lines here.
      echo ">> sudo password from Proton Pass ($item)"
      printf '%s\n%s\n%s\n' "$pw" "$pw" "$pw" | setsid -w "${rebuild[@]}"
    else
      "${rebuild[@]}"
    fi
    unset pw
    ;;

  image|flash)
    if [ "$cmd" = flash ]; then
      config="${2:-}"; dev="${3:-}"
      { [ -n "$config" ] && [ -n "$dev" ]; } \
        || die "usage: ./deploy flash <config> <dev>  (e.g. /dev/sdX)"
      # Validate the device and the tools BEFORE building, so a bad `flash`
      # fails fast instead of dying halfway through writing the card.
      [ -b "$dev" ] || die "$dev is not a block device"
      need zstdcat; need dd; need lsblk
    else
      config="${2:-}"; [ -n "$config" ] || die "usage: ./deploy image <config>"
    fi
    require_tracked "$config"

    echo ">> building SD image for .#$config (aarch64 needs qemu binfmt)"
    nix build ".#nixosConfigurations.$config.config.system.build.sdImage" -o result-sd
    img="$(one_match 'SD image' result-sd/sd-image/*.img.zst)"
    echo ">> image: $img"
    [ "$cmd" = image ] && exit 0

    echo ">> TARGET DEVICE — everything on it will be ERASED:"
    lsblk -o NAME,SIZE,MODEL,TRAN,MOUNTPOINTS "$dev"
    read -rp ">> type 'yes' to write $config to $dev: " ok
    [ "$ok" = yes ] || die "aborted"
    zstdcat "$img" | sudo dd of="$dev" bs=4M status=progress oflag=sync
    sync

    # If this config has a dedicated sops age key, drop it on the ROOT ext4
    # partition at /var/lib/sops-nix/age.txt so sops decrypts on first boot.
    # (The Pi's vfat partition isn't mounted at runtime, so the key can't live
    # there.) Key stays off-repo, out of the nix store, and out of the image.
    keyfile="$KEYDIR/$config/age.txt"
    if [ -f "$keyfile" ]; then
      echo ">> installing sops age key onto the root partition"
      sudo partprobe "$dev" 2>/dev/null || sudo blockdev --rereadpt "$dev" 2>/dev/null || true
      sudo udevadm settle 2>/dev/null || true
      # Largest ext4 partition = the NixOS root. -P (KEY="value" pairs) instead
      # of a columnar listing: an empty FSTYPE collapses under whitespace
      # splitting and shifts every later field.
      rootpart="$(lsblk -bPo PATH,FSTYPE,SIZE "$dev" \
        | sed -n 's/^PATH="\([^"]*\)" FSTYPE="ext4" SIZE="\([0-9]*\)"$/\2 \1/p' \
        | sort -rn | head -1 | cut -d' ' -f2)"
      [ -n "$rootpart" ] || die "no ext4 root partition found on $dev — place $keyfile at /var/lib/sops-nix/age.txt manually"
      mnt="$(mktemp -d)"
      # Unmount + remove even if the install fails, so a retry doesn't trip
      # over the card still being mounted on a stale temp dir.
      trap 'sudo umount "$mnt" 2>/dev/null || true; rmdir "$mnt" 2>/dev/null || true' EXIT
      sudo mount "$rootpart" "$mnt"
      sudo install -Dm600 "$keyfile" "$mnt/var/lib/sops-nix/age.txt"
      sudo sync
      sudo umount "$mnt"; rmdir "$mnt"; trap - EXIT
      echo ">> age key installed (/var/lib/sops-nix/age.txt)"
    fi
    echo ">> done — insert the card into the Pi and boot."
    ;;

  *)
    die "unknown command '$cmd' (kexec|kexec-local|install|switch|boot|test|image|flash)"
    ;;
esac
