diff options
Diffstat (limited to 'tests')
| -rw-r--r-- | tests/unit/test_archangel.bats | 398 | ||||
| -rw-r--r-- | tests/unit/test_btrfs.bats | 34 | ||||
| -rw-r--r-- | tests/unit/test_test_install.bats | 42 |
3 files changed, 473 insertions, 1 deletions
diff --git a/tests/unit/test_archangel.bats b/tests/unit/test_archangel.bats index 983bfd2..508a062 100644 --- a/tests/unit/test_archangel.bats +++ b/tests/unit/test_archangel.bats @@ -20,6 +20,16 @@ setup() { # shellcheck disable=SC1091 source "${BATS_TEST_DIRNAME}/../../installer/archangel" UNATTENDED=true + # Tests that call install_failure_cleanup in this process would + # otherwise run the real disarm_failure_trap, which clears bats' own + # EXIT trap, and bats reports a failing assertion from that trap: the + # failure would vanish instead of printing "not ok". The trap arming + # itself is tested in a child bash, which sources the real functions. + disarm_failure_trap() { :; } + # Point the LUKS-mapping check at an empty directory, so a machine with + # a real /dev/mapper/cryptroot can't leak into the cleanup tests. + MAPPER_DIR="$BATS_TEST_TMPDIR/mapper" + mkdir -p "$MAPPER_DIR" } ############################# @@ -136,9 +146,212 @@ setup() { } ############################# +# Failure trap arming +############################# +# arm_failure_trap / disarm_failure_trap own the trap set that routes a +# mid-install failure to install_failure_cleanup. These run in a child +# bash (via `run bash -c`) because trap inheritance is a property of +# the live shell. Before the EXIT trap, a failure inside any install step +# ended the script through errexit without the cleanup running, and an +# `exit 1` from error() reached no ERR trap at all. Both were how the +# 2026-08-06 and 2026-09-12 retries found a still-mounted disk. + +@test "arm_failure_trap fires the cleanup for a command failing inside a function" { + run bash -c 'source "$1" + install_failure_cleanup() { disarm_failure_trap; echo CLEANUP-RAN; exit 7; } + arm_failure_trap + f() { false; } + f + echo NOT-REACHED' _ "${BATS_TEST_DIRNAME}/../../installer/archangel" + [ "$status" -eq 7 ] + [[ "$output" == *"CLEANUP-RAN"* ]] + [[ "$output" != *"NOT-REACHED"* ]] +} + +@test "arm_failure_trap fires the cleanup when error() exits from inside a function" { + run bash -c 'source "$1" + install_failure_cleanup() { disarm_failure_trap; echo CLEANUP-RAN; exit 7; } + arm_failure_trap + f() { false || error "pacstrap failed"; } + f + echo NOT-REACHED' _ "${BATS_TEST_DIRNAME}/../../installer/archangel" + [ "$status" -eq 7 ] + [[ "$output" == *"CLEANUP-RAN"* ]] + [[ "$output" != *"NOT-REACHED"* ]] +} + +@test "arm_failure_trap fires the cleanup for a failing pipeline inside a function" { + run bash -c 'source "$1" + install_failure_cleanup() { disarm_failure_trap; echo CLEANUP-RAN; exit 7; } + arm_failure_trap + f() { printf "\n" | false; } + f + echo NOT-REACHED' _ "${BATS_TEST_DIRNAME}/../../installer/archangel" + [ "$status" -eq 7 ] + [[ "$output" == *"CLEANUP-RAN"* ]] + [[ "$output" != *"NOT-REACHED"* ]] +} + +@test "disarm_failure_trap lets the success path exit without running the cleanup" { + run bash -c 'source "$1" + install_failure_cleanup() { disarm_failure_trap; echo CLEANUP-RAN; exit 7; } + arm_failure_trap + f() { true; } + f + disarm_failure_trap + echo SUCCESS' _ "${BATS_TEST_DIRNAME}/../../installer/archangel" + [ "$status" -eq 0 ] + [[ "$output" == *"SUCCESS"* ]] + [[ "$output" != *"CLEANUP-RAN"* ]] +} + +@test "arm_failure_trap fires the cleanup when the install is sent TERM" { + run bash -c 'source "$1" + install_failure_cleanup() { disarm_failure_trap; echo CLEANUP-RAN; exit 7; } + arm_failure_trap + f() { kill -TERM $$; sleep 2; echo NOT-REACHED; } + f' _ "${BATS_TEST_DIRNAME}/../../installer/archangel" + [ "$status" -eq 7 ] + [[ "$output" == *"CLEANUP-RAN"* ]] + [[ "$output" != *"NOT-REACHED"* ]] +} + +# A failing command substitution whose status is masked (an argument to +# echo, a `local` declaration, a `|| true`) is not an install failure. With +# errtrace on, bash ran the whole cleanup inside the substitution's +# subshell, unmounting /mnt while the install carried on in the parent. +@test "a masked failing command substitution does not run the cleanup" { + local marker="$BATS_TEST_TMPDIR/cleanup-ran" + run bash -c 'source "$1"; M="$2" + install_failure_cleanup() { disarm_failure_trap; echo ran >> "$M"; exit 7; } + arm_failure_trap + f() { echo "$(false)" >/dev/null; local y; y=$(false) || true; echo STEP-DONE; } + f + disarm_failure_trap + echo SUCCESS' _ "${BATS_TEST_DIRNAME}/../../installer/archangel" "$marker" + [ "$status" -eq 0 ] + [[ "$output" == *"STEP-DONE"* ]] + [[ "$output" == *"SUCCESS"* ]] + [ ! -e "$marker" ] +} + +# A SIGTERM to the whole process group also reaches the tee init_logging +# puts on stdout. With the tee gone, the cleanup's first warn raised +# SIGPIPE and killed the shell before anything was unmounted: a Btrfs +# install TERM'd mid-pacstrap in the VM on 2026-09-13 was left with /mnt +# and the LUKS mapping still open. Ctrl-C's SIGINT leaves the tee alive +# (bash starts a process substitution with SIGINT ignored), so the INT +# test guards the ordinary Ctrl-C path rather than the SIGPIPE fix. +# +# The child starts through `env --default-signal=INT` because bats runs it +# in the background, and a background job in a non-interactive shell +# begins with SIGINT ignored, which bash then refuses to trap. Without the +# reset, the INT case would pass by never delivering the signal at all. +signal_group_with_logging_tee() { + local sig="$1" d="$2" + env --default-signal=INT setsid bash -c 'source "$1"; D="$2" + FILESYSTEM=zfs; POOL_NAME=zroot + umount() { echo "umount $*" >> "$D/calls"; } + zpool() { echo "zpool $*" >> "$D/calls"; [[ "$1" == list ]]; } + exec > >(tee -a "$D/tee.log") 2>&1 + arm_failure_trap + grep SigIgn /proc/$$/status > "$D/sigign" + echo $$ > "$D/armed" + sleep 30' _ "${BATS_TEST_DIRNAME}/../../installer/archangel" "$d" </dev/null >/dev/null 2>&1 & + + local i pid="" + for i in $(seq 1 100); do + [ -s "$d/armed" ] && pid=$(cat "$d/armed") && break + sleep 0.1 + done + [ -n "$pid" ] || return 1 + kill "-$sig" -- "-$pid" + for i in $(seq 1 100); do + kill -0 "$pid" 2>/dev/null || break + sleep 0.1 + done + kill -0 "$pid" 2>/dev/null && kill -KILL -- "-$pid" + return 0 +} + +@test "install_failure_cleanup finishes when SIGTERM to the group also killed its logging tee" { + local d="$BATS_TEST_TMPDIR" + signal_group_with_logging_tee TERM "$d" + grep -qx "zpool export zroot" "$d/calls" +} + +@test "install_failure_cleanup finishes when Ctrl-C's SIGINT reaches the whole group" { + local d="$BATS_TEST_TMPDIR" + signal_group_with_logging_tee INT "$d" + # Guard against the vacuous pass: SIGINT (mask bit 0x2) must not be + # ignored in the child, or the signal never arrived. + local mask + mask=$(awk '{print $2}' "$d/sigign") + [ $(( 0x$mask & 0x2 )) -eq 0 ] + grep -qx "zpool export zroot" "$d/calls" +} + +# The busy retries can keep the cleanup silent for several seconds, and a +# second Ctrl-C or SIGTERM in that window used to kill it halfway: the +# export never finished and the pool stayed imported. +@test "a second SIGTERM during the busy-retry window doesn't stop the cleanup" { + local d="$BATS_TEST_TMPDIR" + env --default-signal=INT setsid bash -c 'source "$1"; D="$2" + FILESYSTEM=zfs; POOL_NAME=zroot; CLEANUP_BUSY_ATTEMPTS=6 + umount() { :; } + zpool() { + [[ "$1" == list ]] && return 0 + if [[ "$*" == "export zroot" ]]; then + echo try >> "$D/exports" + [ "$(wc -l < "$D/exports")" -ge 4 ] && { echo exported >> "$D/done"; return 0; } + return 1 + fi + [[ "$*" == "export -f zroot" ]] && echo forced >> "$D/done" + return 0 + } + arm_failure_trap + echo $$ > "$D/armed" + sleep 30' _ "${BATS_TEST_DIRNAME}/../../installer/archangel" "$d" </dev/null >/dev/null 2>&1 & + + local i pid="" + for i in $(seq 1 100); do + [ -s "$d/armed" ] && pid=$(cat "$d/armed") && break + sleep 0.1 + done + [ -n "$pid" ] + kill -TERM -- "-$pid" + for i in $(seq 1 50); do + [ -s "$d/exports" ] && break + sleep 0.1 + done + [ -s "$d/exports" ] + kill -TERM -- "-$pid" 2>/dev/null || true + for i in $(seq 1 100); do + kill -0 "$pid" 2>/dev/null || break + sleep 0.1 + done + kill -0 "$pid" 2>/dev/null && kill -KILL -- "-$pid" + + [ "$(cat "$d/done" 2>/dev/null)" = "exported" ] +} + +@test "install_failure_cleanup runs exactly once when its own exit re-fires the trap" { + run bash -c 'source "$1" + FILESYSTEM=zfs; POOL_NAME=zroot + umount() { :; } + zpool() { return 1; } + arm_failure_trap + f() { false; } + f' _ "${BATS_TEST_DIRNAME}/../../installer/archangel" + [ "$status" -eq 1 ] + [ "$(grep -c 'cleaning up' <<<"$output")" -eq 1 ] + [[ "$output" == *"system cleaned up"* ]] +} + +############################# # install_failure_cleanup ############################# -# install_failure_cleanup is the trap target for ERR / INT / TERM +# install_failure_cleanup is the trap target for ERR / INT / TERM / EXIT # during install_zfs and install_btrfs. It clears sensitive vars, # dispatches on FILESYSTEM, and exits non-zero. Tests use function # overrides to capture which system tools the cleanup invokes; the @@ -146,6 +359,143 @@ setup() { # btrfs_close_encryption) are deliberately VM-tested per # testing-strategy.org. +############################# +# retry_busy +############################# +# Teardown after an interrupted pacstrap races pacman's children in the +# chroot, which can still be exiting: the first export or LUKS close fails +# busy and succeeds a moment later. retry_busy gives it a bounded window. + +@test "retry_busy runs a command that succeeds once, without sleeping" { + CALLS=() + sleep() { CALLS+=("sleep $*"); } + ok() { CALLS+=("ok"); return 0; } + retry_busy 5 ok + [ "${#CALLS[@]}" -eq 1 ] + [ "${CALLS[0]}" = "ok" ] +} + +@test "retry_busy succeeds when the command first succeeds on the last attempt" { + CALLS=(); TRIES=0 + sleep() { CALLS+=("sleep $*"); } + flaky() { TRIES=$((TRIES + 1)); [ "$TRIES" -ge 3 ]; } + retry_busy 3 flaky + [ "$TRIES" -eq 3 ] + [ "${#CALLS[@]}" -eq 2 ] +} + +@test "retry_busy returns 1 after the last attempt, sleeping only between attempts" { + CALLS=(); TRIES=0 + sleep() { CALLS+=("sleep $*"); } + never() { TRIES=$((TRIES + 1)); return 1; } + run retry_busy 4 never + [ "$status" -eq 1 ] + retry_busy 4 never || true + [ "$TRIES" -eq 4 ] + [ "${#CALLS[@]}" -eq 3 ] +} + +# bats fails a test through errexit. The cleanup turns errexit off for +# itself, and before `local -` scoped that, every assertion after a direct +# call except the last one stopped being able to fail. +@test "install_failure_cleanup leaves the caller's errexit on" { + FILESYSTEM=zfs + POOL_NAME=zroot + umount() { :; } + zpool() { return 1; } + warn() { :; } + error() { return 1; } + + install_failure_cleanup || true + + # Capture the flags, then restore errexit before asserting. If the + # cleanup leaked errexit-off, a bare assertion here couldn't fail. + local flags="$-" + set -e + [[ "$flags" == *e* ]] +} + +@test "install_failure_cleanup ZFS path retries the export while the pool is still busy" { + FILESYSTEM=zfs + POOL_NAME=zroot + CALLS=(); EXPORTS=0 + umount() { :; } + zpool() { + CALLS+=("zpool $*") + [[ "$1" == list ]] && return 0 + if [[ "$*" == "export zroot" ]]; then + EXPORTS=$((EXPORTS + 1)) + [ "$EXPORTS" -ge 3 ] + return + fi + return 0 + } + sleep() { CALLS+=("sleep $*"); } + warn() { :; } + error() { return 1; } + + install_failure_cleanup || true + + [ "$EXPORTS" -eq 3 ] + [[ " ${CALLS[*]} " != *" zpool export -f zroot "* ]] +} + +@test "install_failure_cleanup ZFS path forces the export once the busy window runs out" { + FILESYSTEM=zfs + POOL_NAME=zroot + CLEANUP_BUSY_ATTEMPTS=4 + CALLS=(); EXPORTS=0; SLEEPS=0 + umount() { :; } + zpool() { + CALLS+=("zpool $*") + [[ "$1" == list ]] && return 0 + [[ "$*" == "export zroot" ]] && { EXPORTS=$((EXPORTS + 1)); return 1; } + return 0 + } + sleep() { SLEEPS=$((SLEEPS + 1)); } + warn() { :; } + error() { return 1; } + + install_failure_cleanup || true + + [ "$EXPORTS" -eq 4 ] + [ "$SLEEPS" -eq 3 ] + [[ " ${CALLS[*]} " == *" zpool export -f zroot "* ]] +} + +@test "install_failure_cleanup Btrfs path retries closing LUKS until the mapping is gone" { + FILESYSTEM=btrfs + CLOSES=0 + umount() { :; } + btrfs_cleanup() { :; } + btrfs_close_encryption() { CLOSES=$((CLOSES + 1)); } + luks_mappings_open() { [ "$CLOSES" -lt 3 ]; } + sleep() { :; } + warn() { :; } + error() { return 1; } + + install_failure_cleanup || true + + [ "$CLOSES" -eq 3 ] +} + +@test "install_failure_cleanup Btrfs path stops retrying the LUKS close after the busy window" { + FILESYSTEM=btrfs + CLEANUP_BUSY_ATTEMPTS=4 + CLOSES=0 + umount() { :; } + btrfs_cleanup() { :; } + btrfs_close_encryption() { CLOSES=$((CLOSES + 1)); } + luks_mappings_open() { return 0; } + sleep() { :; } + warn() { :; } + error() { return 1; } + + install_failure_cleanup || true + + [ "$CLOSES" -eq 4 ] +} + @test "install_failure_cleanup clears sensitive variables before exiting" { FILESYSTEM=zfs POOL_NAME=zroot @@ -218,6 +568,52 @@ setup() { [[ " ${CALLS[*]} " != *" zpool"* ]] } +# btrfs_cleanup unmounts only the subvolumes it mounted, one level at a +# time. A pacstrap interrupted mid-transaction can leave its /proc, /sys +# and /dev bind mounts under /mnt, which keep the root busy, so the LUKS +# mapping can't close and the retry's disk_in_use still sees a mountpoint. +@test "install_failure_cleanup Btrfs path unmounts the target root recursively before closing LUKS" { + FILESYSTEM=btrfs + CALLS=() + + umount() { CALLS+=("umount $*"); return 0; } + zpool() { CALLS+=("zpool $*"); return 0; } + btrfs_cleanup() { CALLS+=("btrfs_cleanup"); } + btrfs_close_encryption() { CALLS+=("btrfs_close_encryption"); } + warn() { :; } + error() { CALLS+=("error"); return 1; } + + install_failure_cleanup || true + + local i recursive=-1 close=-1 + for i in "${!CALLS[@]}"; do + [[ "${CALLS[$i]}" == "umount -R /mnt" ]] && recursive=$i + [[ "${CALLS[$i]}" == "btrfs_close_encryption" ]] && close=$i + done + [ "$recursive" -ge 0 ] + [ "$close" -gt "$recursive" ] +} + +@test "install_failure_cleanup Btrfs path falls back to a lazy recursive unmount when the root is busy" { + FILESYSTEM=btrfs + CALLS=() + + umount() { + CALLS+=("umount $*") + [[ "$*" == "-R /mnt" ]] && return 32 + return 0 + } + btrfs_cleanup() { CALLS+=("btrfs_cleanup"); } + btrfs_close_encryption() { CALLS+=("btrfs_close_encryption"); } + warn() { :; } + error() { CALLS+=("error"); return 1; } + + install_failure_cleanup || true + + [[ " ${CALLS[*]} " == *" umount -R -l /mnt "* ]] + [[ " ${CALLS[*]} " == *" btrfs_close_encryption "* ]] +} + @test "install_failure_cleanup ZFS path skips zpool export when pool not imported" { FILESYSTEM=zfs POOL_NAME=zroot diff --git a/tests/unit/test_btrfs.bats b/tests/unit/test_btrfs.bats index 15bf141..6205311 100644 --- a/tests/unit/test_btrfs.bats +++ b/tests/unit/test_btrfs.bats @@ -89,3 +89,37 @@ setup() { run parse_btrfs_subvol_opts "@x" "nodatacow,nosuid" [ "$output" = "subvol=@x,noatime,space_cache=v2,discard=async,nodatacow,nosuid,nodev" ] } + +############################# +# luks_mappings_open +############################# +# The failure cleanup can't trust close_luks_container's exit status (it +# swallows errors), so it checks whether a mapping still exists. Names +# follow get_luks_devices: the bare LUKS_MAPPER_NAME, then a numeric suffix. + +@test "luks_mappings_open succeeds when the bare mapping exists" { + MAPPER_DIR="$BATS_TEST_TMPDIR/mapper"; mkdir -p "$MAPPER_DIR" + touch "$MAPPER_DIR/$LUKS_MAPPER_NAME" + run luks_mappings_open + [ "$status" -eq 0 ] +} + +@test "luks_mappings_open succeeds when only a suffixed mapping exists" { + MAPPER_DIR="$BATS_TEST_TMPDIR/mapper"; mkdir -p "$MAPPER_DIR" + touch "$MAPPER_DIR/${LUKS_MAPPER_NAME}1" + run luks_mappings_open + [ "$status" -eq 0 ] +} + +@test "luks_mappings_open fails when no mapping exists" { + MAPPER_DIR="$BATS_TEST_TMPDIR/mapper"; mkdir -p "$MAPPER_DIR" + run luks_mappings_open + [ "$status" -eq 1 ] +} + +@test "luks_mappings_open ignores unrelated mappings" { + MAPPER_DIR="$BATS_TEST_TMPDIR/mapper"; mkdir -p "$MAPPER_DIR" + touch "$MAPPER_DIR/control" "$MAPPER_DIR/vg-root" "$MAPPER_DIR/${LUKS_MAPPER_NAME}-old" + run luks_mappings_open + [ "$status" -eq 1 ] +} diff --git a/tests/unit/test_test_install.bats b/tests/unit/test_test_install.bats index 52ea037..3fb7c19 100644 --- a/tests/unit/test_test_install.bats +++ b/tests/unit/test_test_install.bats @@ -15,6 +15,48 @@ setup() { } ############################# +# save_attempt_log +############################# +# The retry loop refetches the guest's newest /tmp/archangel-*.log on +# every failed attempt, so a retried install used to keep only the last +# attempt's log — the failure that triggered the retry was gone by the +# time anyone read test-logs/. save_attempt_log keeps each one. + +# Normal: a failed attempt's log lands under its own attempt-numbered name. +@test "save_attempt_log writes the attempt's log under an attempt-numbered name" { + local dir="$BATS_TEST_TMPDIR/logs" + mkdir -p "$dir" + run save_attempt_log "$dir" zfs-encrypt 1 $'line one\nline two' + [ "$status" -eq 0 ] + [ -f "$dir/zfs-encrypt-install-attempt1.log" ] + [ "$(cat "$dir/zfs-encrypt-install-attempt1.log")" = $'line one\nline two' ] +} + +# Boundary: an empty capture still writes a file, so "attempt ran, log +# was empty" stays distinguishable from "attempt never happened". +@test "save_attempt_log writes an empty file for an empty capture" { + local dir="$BATS_TEST_TMPDIR/logs" + mkdir -p "$dir" + run save_attempt_log "$dir" mirror 2 "" + [ "$status" -eq 0 ] + [ -f "$dir/mirror-install-attempt2.log" ] + [ ! -s "$dir/mirror-install-attempt2.log" ] +} + +# Error: an unwritable log dir fails the save without aborting the caller. +@test "save_attempt_log returns non-zero when the log dir is unwritable" { + [ "$EUID" -ne 0 ] || skip "running as root" + local dir="$BATS_TEST_TMPDIR/logs" + mkdir -p "$dir" + chmod 000 "$dir" + run save_attempt_log "$dir" mirror 1 "content" + chmod 755 "$dir" + [ "$status" -eq 1 ] + [ -z "$output" ] + [ ! -f "$dir/mirror-install-attempt1.log" ] +} + +############################# # is_transient_install_failure ############################# |
