diff options
Diffstat (limited to 'scripts')
| -rwxr-xr-x | scripts/cmail-setup-finish.sh | 56 | ||||
| -rwxr-xr-x | scripts/post-rebuild-check | 676 | ||||
| -rw-r--r-- | scripts/testing/tests/test_config_applied.py | 3 | ||||
| -rwxr-xr-x | scripts/zz-bluetooth-resume | 86 |
4 files changed, 797 insertions, 24 deletions
diff --git a/scripts/cmail-setup-finish.sh b/scripts/cmail-setup-finish.sh index 949023f..8c27eda 100755 --- a/scripts/cmail-setup-finish.sh +++ b/scripts/cmail-setup-finish.sh @@ -1,32 +1,37 @@ #!/usr/bin/env bash # SPDX-License-Identifier: GPL-3.0-or-later -# cmail-setup-finish.sh — finish Proton Mail Bridge + cmail-action setup after -# Bridge first-run. Idempotent; safe to re-run after a Bridge cert rotation or -# a claude-templates re-clone. +# cmail-setup-finish.sh — finish Proton Mail Bridge setup after Bridge +# first-run. Idempotent; safe to re-run after a Bridge cert rotation. # # Pre-reqs (the script aborts if any are missing): # - protonmail-bridge installed (archsetup handles it) # - You have run 'protonmail-bridge --cli', logged in, and quit at least once # (the script looks for state at ~/.config/protonmail/bridge-v3/) -# - claude-templates cloned at ~/projects/claude-templates # - dotfiles stowed (~/.config/.cmailpass.gpg present) # +# Not a pre-req, but checked and warned about: cmail-action on PATH. rulesets' +# `make install` links it, and session start runs that, so on a machine that +# runs agent sessions it arrives without anyone asking. On one that doesn't, +# it needs the command by hand. The script never invokes it either way. +# # What it does: # 1. Decrypts ~/.config/.cmailpass.gpg → ~/.config/.cmailpass (mode 0600) # 2. Copies Bridge's self-signed cert → ~/.config/protonbridge.pem -# 3. Symlinks ~/projects/claude-templates/.ai/scripts/cmail-action.py -# → ~/.local/bin/cmail-action -# 4. Removes the leftover ~/.config/autostart/Proton Mail Bridge.desktop +# 3. Removes the leftover ~/.config/autostart/Proton Mail Bridge.desktop # stub (it double-launches Bridge alongside the systemd user service # and throws an "orphan instance" dialog every login) -# 5. Installs a wait-for-dns drop-in so Bridge doesn't spam +# 4. Installs a wait-for-dns drop-in so Bridge doesn't spam # name-resolution errors during the early-boot DNS race -# 6. Enables + starts the protonmail-bridge user service -# 7. Verifies Bridge is listening on 127.0.0.1:1143 / :1025 +# 5. Enables + starts the protonmail-bridge user service +# 6. Verifies Bridge is listening on 127.0.0.1:1143 / :1025 +# +# It no longer installs cmail-action. That moved to rulesets +# (claude-templates/bin/), whose `make install` owns the symlink. set -euo pipefail err() { printf 'error: %s\n' "$*" >&2; exit 1; } +warn() { printf 'warning: %s\n' "$*" >&2; } info() { printf '==> %s\n' "$*"; } ok() { printf ' %s\n' "$*"; } @@ -47,9 +52,20 @@ bridge_state="$HOME/.config/protonmail/bridge-v3" [ -d "$bridge_state" ] \ || err "Bridge has no state at $bridge_state — run 'protonmail-bridge --cli' and log in first" -cmail_action_src="$HOME/projects/claude-templates/.ai/scripts/cmail-action.py" -[ -f "$cmail_action_src" ] \ - || err "cmail-action.py not found at $cmail_action_src — clone claude-templates first" +# cmail-action is no longer this script's to install. It lives in rulesets at +# claude-templates/bin/, and rulesets' `make install` links everything there +# into ~/.local/bin. Session start runs that, so on a machine that runs agent +# sessions the symlink arrives on its own; on one that doesn't, it needs the +# command below. +# +# A warning rather than an abort, because this script never invokes the tool. +# Its job is to leave Bridge working, and it can finish that whether or not a +# mail client has been linked yet. Aborting here would make Bridge setup +# depend on rulesets being cloned and installed first, an ordering neither +# repo otherwise needs, and would strand a fresh machine with Bridge ready and +# the script refusing to configure it. +command -v cmail-action >/dev/null 2>&1 \ + || warn "cmail-action not on PATH — run 'make -C ~/code/rulesets install' before sending mail" cmailpass_enc="$HOME/.config/.cmailpass.gpg" [ -f "$cmailpass_enc" ] \ @@ -69,13 +85,7 @@ cert_dst="$HOME/.config/protonbridge.pem" cp "$cert_src" "$cert_dst" ok "copied $cert_src → $cert_dst" -# 4. Symlink cmail-action -info "symlinking cmail-action" -mkdir -p "$HOME/.local/bin" -ln -sf "$cmail_action_src" "$HOME/.local/bin/cmail-action" -ok "linked $HOME/.local/bin/cmail-action → $cmail_action_src" - -# 5. Remove leftover XDG autostart stub +# 4. Remove leftover XDG autostart stub # The systemd --user service is the canonical launcher. The autostart .desktop # starts a second Bridge instance that can't get the lock and pops up an # "orphan instance" dialog every login. @@ -88,7 +98,7 @@ else ok "no autostart stub present" fi -# 6. Install wait-for-dns drop-in +# 5. Install wait-for-dns drop-in # User-instance systemd doesn't carry network-online.target / nss-lookup.target, # so the packaged unit's After=network.target doesn't imply DNS readiness. # Bridge starts before the resolver is up and its first API calls all fail @@ -107,7 +117,7 @@ ok "wrote $dropin_file" systemctl --user daemon-reload ok "reloaded systemd user units" -# 7. Enable + start systemd user service +# 6. Enable + start systemd user service info "enabling protonmail-bridge user service" was_active=0 systemctl --user is-active --quiet protonmail-bridge.service && was_active=1 @@ -119,7 +129,7 @@ else ok "service active" fi -# 8. Verify +# 7. Verify info "verifying Bridge is listening" listening="$(ss -ltn 2>/dev/null || true)" missing="" diff --git a/scripts/post-rebuild-check b/scripts/post-rebuild-check new file mode 100755 index 0000000..aa7ef83 --- /dev/null +++ b/scripts/post-rebuild-check @@ -0,0 +1,676 @@ +#!/bin/sh +# SPDX-License-Identifier: GPL-3.0-or-later +# post-rebuild-check - the eight checks a rebuilt machine actually needs. +# +# A rebuilt machine looks finished and isn't. Five gaps surfaced on velox +# within two days of the 2026-08-13 reinstall, and three of them LOOKED +# fine: a stowed unit file, an enabled timer, a present git clone. Each +# check below is cheap and turns a silent no-op into a visible line: +# +# 1. failed systemd units, user and system scope (calendar-sync failed +# every 15 minutes for two days with nobody watching) +# 2. user unit files present but not enabled (roam-sync and +# signal-receive came back linked and inert -- a unit file being +# present is not the same as running) +# 3. tracked *.example files whose real sibling is missing (three +# *.local.el were gone on velox; the .example survives in git, +# the real file never does) +# 4. gitignore-mode projects missing tooling paths their own .gitignore +# names (a reinstall drops every such project's untracked working +# state -- 374 files in .emacs.d's case -- and nothing carries it) +# 5. signal-cli holds no registered account (velox lost its registration +# in the rebuild; at the time agent-text relayed into velox, so that +# silently broke paging for the WHOLE fleet. agent-text now walks +# AGENT_TEXT_RELAYS in order and skips itself, so an unregistered +# machine is only fatal when no relay host is registered either) +# 6. every NTP source is named by hostname (a wrong clock fails the +# DoT/DNSSEC validation this machine's DNS runs on, so nothing +# resolves -- including the NTP pool that would fix the clock; velox +# deadlocked exactly this way 2026-08-19 and needed a second device) +# 7. hypridle installed but not running (nothing then triggers idle lock +# or suspend, so a laptop runs until its battery is gone -- which is +# how velox reset the RTC that caused check 6's deadlock in the first +# place; a caffeine remembered from an earlier boot is the known cause) +# 8. a working repo cloned from the read-only https endpoint (correct +# for a stranger with no key on the server, wrong for this machine, +# which finds out at the first push with a 403 -- velox's dotfiles +# remote sat that way for four days after its rebuild) +# +# The .gitignore rule in check 4 is what scopes it: a tooling path is only +# expected where the project's own .gitignore names it, so a project that +# never had a todo.org never flags. The ignore file is the project's own +# record of what it is supposed to hold untracked. +# +# EVERY PROBE FAILS CLOSED. A check that cannot run reports a finding, never +# a pass. This matters more here than anywhere else in the script: the whole +# point is catching silent no-ops, so a silent no-op in the checker would be +# the worst possible defect. `systemctl --user` exits 1 with empty output +# when there is no user bus -- over ssh, from cron, under sudo, on a TTY +# before the graphical session starts -- and reading that as "no failed +# units" would report a machine as healthy exactly when nothing was checked. +# +# Exit 0 when every check is clean, 1 when any check found something, +# 2 on usage error. +# +# Test seams (env; for each, set-but-empty means "the probe ran and found +# nothing", unset means "run the real probe"): +# PRC_FAILED_UNITS newline list of "scope:unit" (scope user|system) +# PRC_UNIT_STATES newline list of "unit-file state" replacing the +# user-unit-dir enumeration + is-enabled calls +# PRC_LOCAL_SCAN_ROOTS NEWLINE-separated roots for the *.example scan +# (default: ~/.emacs.d ~/.dotfiles) +# PRC_PROJECT_ROOTS NEWLINE-separated project dirs for check 4 +# (default: ~/code/* ~/projects/* ~/.emacs.d +# ~/.dotfiles) +# PRC_SIGNAL_ACCOUNTS signal-cli listAccounts output; "" = no account, +# the special value MISSING = binary absent +# PRC_NTP_SOURCES newline list of configured NTP server addresses; +# the special value MISSING = no NTP daemon active +# PRC_CHRONY_CONF path to chrony.conf (a fixture, under test) -- the +# confdir it names is what decides which drop-ins count +# PRC_IDLE_DAEMON pgrep output for hypridle; "" = installed but not +# running, the special value MISSING = not installed +# PRC_REPO_REMOTES newline list of "path origin-url"; an empty URL +# means origin could not be read +# PRC_UNITS_EXPECTED_DISABLED +# newline list of units whose not-enabled state is +# deliberate here, replacing the file below +# PRC_UNITS_EXPECTED_DISABLED_FILE +# path to that list (default: +# $XDG_CONFIG_HOME/post-rebuild-check/units-expected-disabled). +# One unit per line, # starts a comment. Machine-local +# on purpose: the same unit is correctly enabled on one +# box and not another +# PRC_SYSTEMCTL path to the systemctl binary (a fake, under test) +# PRC_SYSTEMCTL_TIMEOUT seconds to allow each systemctl call (default 5) +# +# Roots are newline-separated, not space-separated, because a POSIX +# `for root in $var` splits on spaces and turns one real directory into +# several imaginary missing ones. + +usage() { + cat <<'EOF' +post-rebuild-check - verify a rebuilt machine is actually finished + +Runs the eight checks that caught velox's 2026-08 reinstall gaps: failed +units, present-but-inert user units, orphaned *.example configs, missing +per-project tooling state, the signal-cli registration, whether time sync +can recover from a wrong clock without DNS, whether anything still +triggers idle lock and suspend, and whether the working repos can push. + +Usage: post-rebuild-check [--help] + +Exit 0 when every check is clean, 1 when any check found something. +Every probe fails closed: a check that cannot run is a finding, not a pass. +EOF +} + +case "${1:-}" in + --help|-h) usage; exit 0 ;; + "") ;; + *) echo "post-rebuild-check: unknown argument: $1" >&2; usage >&2; exit 2 ;; +esac + +# Own the internal flags rather than inheriting them, so a caller's unrelated +# variable of the same name cannot manufacture or mask a finding. +TOTAL_FINDINGS=0 +CHECK_FINDINGS=0 +FINDING_LINES="" +signal_missing="" +ntp_missing="" +idle_absent="" + +# Every systemctl call is bounded. A wedged user manager spins and answers +# nothing -- seen live on velox 2026-08-17, where `is-enabled`, `cat`, and +# `list-unit-files` all hung while `list-units` still returned. Unbounded, this +# script would hang on the first unit and never reach the remaining checks, +# which is a worse failure than reporting nothing: a check that hangs is its +# own outage, and the machine most in need of checking is the one it hangs on. +# A timeout yields empty output and a non-zero status, and both are already +# handled as findings, so bounding the call is all that is needed to fail closed. +CHRONY_CONF=${PRC_CHRONY_CONF:-/etc/chrony.conf} +SCTL_TIMEOUT=${PRC_SYSTEMCTL_TIMEOUT:-5} +SYSTEMCTL=${PRC_SYSTEMCTL:-systemctl} + +sctl() { + if command -v timeout >/dev/null 2>&1; then + timeout "$SCTL_TIMEOUT" "$SYSTEMCTL" "$@" + else + # Say so rather than dropping the bound silently: without timeout a + # wedged manager hangs this run indefinitely, and the whole point of + # the bound is that a check which hangs reports nothing at all. + [ -n "${sctl_unbounded_warned:-}" ] || { + echo "post-rebuild-check: timeout(1) not found — systemctl calls are UNBOUNDED and may hang" >&2 + sctl_unbounded_warned=1 + } + "$SYSTEMCTL" "$@" + fi +} + +WORK=${TMPDIR:-/tmp}/.post-rebuild-check.$$ +if ! mkdir "$WORK" 2>/dev/null; then + # Every check stages its input through a file in here. Without it each + # loop would read nothing and every check would come back clean, which is + # the one failure this script must never produce. + echo "post-rebuild-check: cannot create a work directory under ${TMPDIR:-/tmp}" >&2 + echo " nothing was checked; this is not a pass" >&2 + exit 1 +fi +trap 'rm -rf "$WORK"' EXIT HUP INT TERM + +STAGE="$WORK/stage" + +finding() { + CHECK_FINDINGS=$((CHECK_FINDINGS + 1)) + TOTAL_FINDINGS=$((TOTAL_FINDINGS + 1)) + FINDING_LINES="${FINDING_LINES} DEVIATION: $1 +" +} + +# Print the check's one visible line, then its findings. The visible line +# is the point: a silent no-op is exactly what let the gaps sit unseen. +report() { + if [ "$CHECK_FINDINGS" -eq 0 ]; then + echo "$1 — ok" + else + echo "$1 — $CHECK_FINDINGS finding(s)" + printf '%s' "$FINDING_LINES" + fi + CHECK_FINDINGS=0 + FINDING_LINES="" +} + +# Stage a value into $STAGE for the read loops. A failed write is fatal for +# the same reason a missing work directory is. +stage() { + if ! printf '%s\n' "$1" > "$STAGE" 2>/dev/null; then + echo "post-rebuild-check: cannot write $STAGE" >&2 + echo " nothing was checked; this is not a pass" >&2 + exit 1 + fi +} + +# --- 1. failed units ------------------------------------------------------ + +if [ -n "${PRC_FAILED_UNITS+set}" ]; then + failed=$PRC_FAILED_UNITS +else + failed="" + if user_out=$(sctl --user list-units --state=failed --no-legend --plain 2>/dev/null); then + failed=$(printf '%s' "$user_out" | awk 'NF {print "user:"$1}') + else + finding "could not query user units (no user bus?) — nothing was checked in this scope" + fi + if sys_out=$(sctl list-units --state=failed --no-legend --plain 2>/dev/null); then + failed="$failed +$(printf '%s' "$sys_out" | awk 'NF {print "system:"$1}')" + else + finding "could not query system units — nothing was checked in this scope" + fi +fi +stage "$failed" +while IFS= read -r line; do + [ -n "$line" ] || continue + scope=${line%%:*} + unit=${line#*:} + finding "$scope unit failed: $unit" +done < "$STAGE" +report "check 1/8: failed units" + +# --- 2. user unit files present but not enabled --------------------------- +# +# Units nothing intends to enable here are read from a machine-local list. +# "Enabled" is this check's proxy for "will actually run", and the proxy is +# wrong for a unit nobody means to enable on this box. velox carries four, for +# four different reasons: geoclue-agent is redundant because hyprland's +# exec-once starts the binary directly, emacs is started on demand by +# emacsclient, obs-record-watchdog only matters while recording, and +# obsbot-wb-guard needs an OBSBOT the machine does not have. Left unexempted +# they report at every run, and four permanent lines in front of every real one +# teach you to skim the output -- the same argument check 4 makes about +# CLAUDE.md. +# +# Machine-local rather than a marker in the shared unit file, because +# obsbot-wb-guard is correctly ENABLED on ratio. One unit, a different right +# answer per machine, so the shared file cannot hold the answer. +# +# An entry that turns out to be enabled after all is still a finding. Without +# that the list rots into somewhere real findings go to die, which is worse +# than the noise it removes. + +EXPECT_DISABLED_FILE="${PRC_UNITS_EXPECTED_DISABLED_FILE:-${XDG_CONFIG_HOME:-$HOME/.config}/post-rebuild-check/units-expected-disabled}" +if [ -n "${PRC_UNITS_EXPECTED_DISABLED+set}" ]; then + expect_disabled=$PRC_UNITS_EXPECTED_DISABLED +elif [ -f "$EXPECT_DISABLED_FILE" ]; then + expect_disabled=$(cat "$EXPECT_DISABLED_FILE" 2>/dev/null) +else + expect_disabled="" +fi +# Strip comments and blanks once, here, so the membership test below is a +# plain word match. The reason a unit is exempt is the most useful thing about +# the entry, so the format has to carry one. +expect_disabled=$(printf '%s\n' "$expect_disabled" \ + | sed 's/#.*//' | awk 'NF {print $1}') + +if [ -n "${PRC_UNIT_STATES+set}" ]; then + states=$PRC_UNIT_STATES +else + states="" + unit_dir="${XDG_CONFIG_HOME:-$HOME/.config}/systemd/user" + if [ ! -d "$unit_dir" ]; then + finding "no user unit directory at $unit_dir — nothing was checked" + else + for f in "$unit_dir"/*.timer "$unit_dir"/*.service; do + # -L as well as -e: a stow symlink whose target moved in the + # rebuild is exactly the "looked fine" case this check is for, + # and -e is false for a broken link. + [ -e "$f" ] || [ -L "$f" ] || continue + name=$(basename "$f") + # A link with nothing behind it is its own finding, decided on the + # filesystem rather than from systemd. `is-enabled` calls a + # dangling link "not-found" -- the same answer it gives for a unit + # that was never installed -- so routing this through the state + # table below would drop it silently. + if [ -L "$f" ] && [ ! -e "$f" ]; then + finding "stowed unit file points at a missing target: $name" + continue + fi + # is-enabled exits non-zero AND prints a state for disabled and + # linked, so the exit code cannot distinguish "this unit is + # disabled" from "the query failed". The output can: a real answer + # is always a word. Empty means no answer, which is a finding + # rather than a silent skip -- with no user bus (ssh, cron, sudo, + # a TTY before the graphical session) every unit answers empty, + # and treating that as unknown-so-ignore would pass the machine + # while reading nothing at all. + # + # No separate bus probe: `is-system-running` and + # `show-environment` both block here, and a check that can hang is + # its own outage. + state=$(sctl --user is-enabled "$name" 2>/dev/null) + if [ -z "$state" ]; then + finding "could not read the enablement state of $name — it was not checked" + continue + fi + states="${states}${name} ${state} +" + done + fi +fi +stage "$states" +# A second copy for the sibling-timer lookup below, so the awk that reads it +# is never the same open file as the loop reading it. +cp "$STAGE" "$WORK/states" 2>/dev/null || { + echo "post-rebuild-check: cannot write $WORK/states" >&2 + echo " nothing was checked; this is not a pass" >&2; exit 1; } +while read -r name state; do + [ -n "$name" ] || continue + case "$state" in + disabled|linked) ;; + *) continue ;; + esac + # A timer-activated service is SUPPOSED to sit linked-and-not-enabled: + # the timer owns activation, and enabling the service as well would run + # it at boot on top of its schedule. So a service is suppressed only when + # its sibling timer can actually start it (enabled), or when the timer is + # itself inert and therefore the finding already -- reporting both would + # name one gap twice. A masked, static, or not-found timer starts + # nothing, so the service beneath it is as dead as one with no timer. + case "$name" in + *.service) + timer="${name%.service}.timer" + tstate=$(awk -v t="$timer" '$1 == t {print $2; exit}' "$WORK/states") + # enabled-runtime (enabled until reboot) and generated (something + # produced and installed it) are live activation paths, so the + # service under one is being started and is not a finding. + # disabled and linked suppress for a different reason: the timer + # is then the finding itself, reported in its own right. + # + # "indirect" deliberately does NOT suppress. It means the unit + # file itself is not enabled, only that some Also= relative might + # be, so nothing here is known to start the service. The + # fail-closed rule says the uncertain case flags. + case "$tstate" in + enabled|enabled-runtime|generated) continue ;; + disabled|linked) continue ;; + esac + ;; + esac + # Deliberately not enabled on this machine. Checked last, so it suppresses + # only this finding and never the dangling-link one decided above on the + # filesystem. + case " +$expect_disabled +" in + *" +$name +"*) continue ;; + esac + finding "unit file present but not enabled: $name ($state)" +done < "$STAGE" +# The exemption list, checked in the other direction. An entry whose unit is +# enabled after all suppresses nothing, and leaving it there is how the list +# turns into a place real findings go to die. The loop above cannot catch this: +# it skips any state that is not disabled or linked, so an enabled unit never +# reaches it. +printf '%s\n' "$expect_disabled" > "$WORK/expect" 2>/dev/null || { + echo "post-rebuild-check: cannot write $WORK/expect" >&2 + echo " nothing was checked; this is not a pass" >&2; exit 1; } +while IFS= read -r name; do + [ -n "$name" ] || continue + estate=$(awk -v u="$name" '$1 == u {print $2; exit}' "$WORK/states") + case "$estate" in + enabled|enabled-runtime) + finding "$name is listed as expected-disabled but is $estate — drop the stale exemption" ;; + esac +done < "$WORK/expect" +report "check 2/8: unit files" + +# --- 3. *.example files whose real sibling is missing --------------------- + +if [ -n "${PRC_LOCAL_SCAN_ROOTS+set}" ]; then + scan_roots=$PRC_LOCAL_SCAN_ROOTS +else + scan_roots="$HOME/.emacs.d +$HOME/.dotfiles" +fi +printf '%s\n' "$scan_roots" > "$WORK/roots" 2>/dev/null || { + echo "post-rebuild-check: cannot write $WORK/roots" >&2; exit 1; } +while IFS= read -r root; do + [ -n "$root" ] || continue + if [ ! -d "$root" ]; then + finding "scan root missing: $root" + continue + fi + # Vendored package trees ship their own .example docs; those belong to + # the package, not to this machine, so they are noise in front of the + # real findings this check exists for. + # + # -prune, not -not -path: the latter filters find's OUTPUT while still + # descending, so an unreadable directory inside a tree we deliberately + # ignore would set find's exit status and be reported as an unscanned + # part of the root. Pruning means those trees are never entered, so the + # exit status only reflects places this check actually wanted to read. + # + # That status matters: find exits non-zero when it cannot descend + # somewhere, having printed only what it could reach. Discarding it would + # hide every orphan under an unreadable directory behind a clean "ok", + # which is the defect this script exists to catch. + if ! find "$root" \ + \( -name .git -o -name elpa -o -name straight \ + -o -name node_modules -o -name .venv \) -prune \ + -o -name '*.example' -print > "$WORK/examples" 2>/dev/null; then + finding "could not fully scan $root — part of it was not checked" + fi + while IFS= read -r ex; do + [ -n "$ex" ] || continue + # -e, so a sibling that exists only as a dangling symlink counts as + # missing. It is not a config the machine can read. + [ -e "${ex%.example}" ] || finding "example without its real file: $ex" + done < "$WORK/examples" +done < "$WORK/roots" +report "check 3/8: local files" + +# --- 4. gitignore-mode projects missing their tooling --------------------- + +if [ -n "${PRC_PROJECT_ROOTS+set}" ]; then + projects=$PRC_PROJECT_ROOTS +else + projects=$(ls -d "$HOME"/code/*/ "$HOME"/projects/*/ 2>/dev/null; \ + printf '%s\n%s\n' "$HOME/.emacs.d" "$HOME/.dotfiles") +fi +printf '%s\n' "$projects" > "$WORK/projects" 2>/dev/null || { + echo "post-rebuild-check: cannot write $WORK/projects" >&2; exit 1; } +# CLAUDE.md is deliberately absent from this set. It is seed-only -- +# install-lang writes it once and the project owns it afterward -- so most +# projects legitimately never have one, and ratio shows the identical +# absences in the identical projects. That match is what proves it is the +# steady state rather than reinstall drift, and flagging it would put nine +# standing findings in front of every real one. +# +# .claude/ is absent for the same reason and proven the same way. The +# bootstrap and the gitignore sweep write it into the ignore set of every +# gitignore-mode project whether or not one ever exists there, so the entry is +# aspirational rather than a promise -- pearl, rsyncshot and yt-sync each name +# it and none of the three has ever had one, on velox or on ratio. Dropping it +# loses no real signal either: a project that genuinely carries a .claude/ +# (rules and hooks from a language bundle) has it re-synced by +# sync-language-bundle.sh at every session start, so a true absence heals +# itself before this check would run. +# +# The list is fed to the inner loop straight from a heredoc rather than +# staged through a file. It is a constant, so a file bought nothing and cost +# a fifth unguarded write: had it failed (a full tmpfs, say) the inner loop +# would read nothing and every project would pass silently, which is the one +# outcome this script must never produce. The heredoc is the inner loop's own +# stdin and leaves the outer loop's redirect alone. +while IFS= read -r proj; do + [ -n "$proj" ] || continue + proj=${proj%/} + # -e not -d: in a worktree or submodule .git is a file naming the real + # gitdir, and a -d test would skip those projects silently. + [ -e "$proj/.git" ] || continue + [ -f "$proj/.gitignore" ] || continue + while read -r disk pattern; do + # Both the anchored (/.ai/) and unanchored (.ai/) ignore styles exist + # across the fleet; the sweep-gitignore audit hit exactly that split. + # + # grep exits 1 for no-match and 2 for an error, so the two are told + # apart rather than both read as "the ignore file does not name this". + # An unreadable .gitignore would otherwise pass the whole project. + grep -Eq "^/?${pattern}/?\$" "$proj/.gitignore" 2>/dev/null + case $? in + 0) [ -e "$proj/$disk" ] \ + || finding "$proj: .gitignore names $disk but it is missing on disk" ;; + 1) ;; + *) finding "$proj: could not read .gitignore — the project was not checked" + break ;; + esac + done <<'EOF' +.ai \.ai +todo.org todo\.org +inbox inbox +EOF +done < "$WORK/projects" +report "check 4/8: project tooling" + +# --- 5. signal-cli registration ------------------------------------------- + +if [ -n "${PRC_SIGNAL_ACCOUNTS+set}" ]; then + accounts=$PRC_SIGNAL_ACCOUNTS + if [ "$accounts" = "MISSING" ]; then + accounts="" + signal_missing=1 + fi +else + if command -v signal-cli >/dev/null 2>&1; then + if ! accounts=$(signal-cli listAccounts 2>/dev/null); then + accounts="" + finding "signal-cli listAccounts failed — the registration was not checked" + signal_missing=skip + fi + else + accounts="" + signal_missing=1 + fi +fi +if [ "$signal_missing" = 1 ]; then + finding "signal-cli is not installed — paging relies on it fleet-wide" +elif [ -z "$signal_missing" ] && [ -z "$accounts" ]; then + finding "no signal account registered — this machine can only page by relaying to one that has an account; if no host in AGENT_TEXT_RELAYS is registered either, the whole fleet loses paging" +fi +report "check 5/8: signal registration" + +# --- 6. NTP can recover a wrong clock without DNS ------------------------- +# +# The clock/DNS bootstrap deadlock. This machine resolves through DNSOverTLS +# with DNSSEC, and both validate against the wall clock, so a boot with a +# wrong clock resolves nothing at all. If every configured NTP source is named +# by hostname, the daemon that would correct the clock needs the DNS the clock +# is breaking, and the machine cannot recover without a second device -- +# which is exactly what happened on velox 2026-08-19. One source addressed by +# IP breaks the cycle, so that is what this check looks for. + +# True when the argument is an address rather than a name. An address needs no +# resolver, which is the whole property being checked. +is_ip_literal() { + case "$1" in + "") return 1 ;; + *:*) case "$1" in *[!0-9A-Fa-f:]*) return 1 ;; esac + return 0 ;; + *[!0-9.]*) return 1 ;; + *.*) return 0 ;; + esac + return 1 +} + +if [ -n "${PRC_NTP_SOURCES+set}" ]; then + ntp_sources=$PRC_NTP_SOURCES + if [ "$ntp_sources" = "MISSING" ]; then + ntp_sources="" + ntp_missing=1 + fi +elif sctl is-active chronyd >/dev/null 2>&1; then + # The main file, plus any drop-in directory chrony.conf actually names. + # + # The confdir read is the load-bearing part. A drop-in is inert unless + # chrony.conf points at its directory, and Arch's stock chrony.conf points + # at none -- so globbing /etc/chrony.d unconditionally would find the + # IP-addressed source, report the machine healthy, and be describing a file + # chrony never opens. That is a false pass on exactly the misconfiguration + # this check exists to catch, so the sources are read only from files + # chrony is actually told to read. + ntp_conf_files=$CHRONY_CONF + for ntp_dir in $(awk '$1 == "confdir" || $1 == "sourcedir" { print $2 }' \ + "$CHRONY_CONF" 2>/dev/null); do + for ntp_f in "$ntp_dir"/*.conf "$ntp_dir"/*.sources; do + [ -f "$ntp_f" ] && ntp_conf_files="$ntp_conf_files $ntp_f" + done + done + # Unquoted on purpose: the accumulated list is several paths, and none of + # this script's own paths contain spaces. + ntp_sources=$(cat $ntp_conf_files 2>/dev/null \ + | awk '$1 == "server" || $1 == "pool" { print $2 }') +elif sctl is-active systemd-timesyncd >/dev/null 2>&1; then + ntp_sources=$(awk -F= '/^[[:space:]]*NTP=/ { print $2 }' \ + /etc/systemd/timesyncd.conf 2>/dev/null | tr ' ' '\n') +else + ntp_sources="" + ntp_missing=1 +fi + +if [ "$ntp_missing" = 1 ]; then + finding "no NTP implementation is active — nothing corrects the clock, and a wrong clock takes DNS down with it" +elif [ -z "$ntp_sources" ]; then + finding "no NTP sources are configured — nothing was checked, and nothing corrects the clock" +else + ntp_has_literal="" + stage "$ntp_sources" + while IFS= read -r src_addr; do + [ -z "$src_addr" ] && continue + if is_ip_literal "$src_addr"; then + ntp_has_literal=1 + fi + done < "$STAGE" + if [ -z "$ntp_has_literal" ]; then + finding "every NTP source is named by hostname — a wrong clock breaks DNS, so nothing can resolve them and the clock stays wrong" + fi +fi +report "check 6/8: NTP bootstrap" + +# --- 7. the idle daemon survives session start ---------------------------- +# +# A laptop that never sleeps has no symptom until the battery is gone, so +# nothing surfaces this without being asked. On velox 2026-08-19 hypridle +# started cleanly at 15:29:48 and `settings restore` killed it six seconds +# later, replaying a caffeine stored in an earlier boot. The machine ran +# 11h40m fully awake on battery, died when it flattened, and reset its RTC -- +# which took DNS down with it, the same deadlock check 6 exists for. The +# desktop looked correct throughout. +# +# Behavioural on purpose: this asks whether the daemon is alive, not why it +# might not be, so a stale caffeine, a crash, and a broken config all surface +# the same way. Gated on hypridle being installed, because that is what marks +# a machine as using it -- archsetup installs it only for Hyprland, so a +# headless or dwm box would otherwise report a finding on every run. + +if [ -n "${PRC_IDLE_DAEMON+set}" ]; then + idle_pids=$PRC_IDLE_DAEMON + if [ "$idle_pids" = "MISSING" ]; then + idle_pids="" + idle_absent=1 + fi +elif command -v hypridle >/dev/null 2>&1; then + # pgrep exits non-zero with no match, which is the not-running case rather + # than a probe failure, so the || keeps `set -e`-style callers out of it. + idle_pids=$(pgrep -x hypridle 2>/dev/null) || idle_pids="" +else + idle_pids="" + idle_absent=1 +fi + +if [ "$idle_absent" = 1 ]; then + : # hypridle is not part of this machine -- nothing to check +elif [ -z "$idle_pids" ]; then + finding "hypridle is installed but not running — nothing triggers idle lock or suspend, so this machine stays awake until its battery is gone; a caffeine remembered from an earlier boot is the known cause" +fi +report "check 7/8: idle daemon" + +# --- 8. working repos cloned from the read-only endpoint ------------------ +# +# archsetup clones the user's own archsetup and dotfiles from +# https://git.cjennings.net/..., which serves anonymous clones and refuses +# pushes. That default is correct for a stranger installing archsetup -- they +# have no key on the server -- and wrong for this machine, which has to push. +# ARCHSETUP_REPO / DOTFILES_REPO override it, but only where they are +# configured: a curl|bash install, or a rebuild from a stock ISO, takes the +# default straight back. +# +# Nothing about the tree shows it. The clone is complete and ordinary, and the +# machine finds out at the first push, with a 403 -- which is how velox's +# dotfiles remote was found on 2026-08-17, four days after its rebuild, by +# which time the same rebuild's shallow clone had already answered a +# credential-history question wrongly. +# +# Only the read-only endpoint is flagged. An https remote elsewhere may be +# perfectly pushable through a credential helper, and guessing about hosts +# this machine does not own would stand noise in front of the real findings. + +if [ -n "${PRC_REPO_REMOTES+set}" ]; then + repo_remotes=$PRC_REPO_REMOTES +else + repo_remotes="" + for repo in "$HOME/code/archsetup" "$HOME/.dotfiles"; do + # -e not -d: a worktree or submodule .git is a file naming the gitdir. + [ -e "$repo/.git" ] || continue + # A repo with no origin still gets a line, with an empty URL, so the + # loop below reports it rather than skipping it into a silent pass. + repo_url=$(git -C "$repo" remote get-url origin 2>/dev/null) + repo_remotes="${repo_remotes}${repo} ${repo_url} +" + done +fi + +stage "$repo_remotes" +while IFS= read -r repo_line; do + [ -n "$repo_line" ] || continue + repo_path=${repo_line%% *} + repo_url=${repo_line#"$repo_path"} + repo_url=${repo_url# } + case "$repo_url" in + "") + finding "$repo_path: origin could not be read — the remote was not checked" ;; + https://git.cjennings.net/*|https://cjennings.net/*) + finding "$repo_path: origin is the read-only endpoint ($repo_url) — git push returns 403; set the ssh form, or ARCHSETUP_REPO/DOTFILES_REPO before installing" ;; + esac +done < "$STAGE" +report "check 8/8: repo remotes" + +# --- summary -------------------------------------------------------------- + +if [ "$TOTAL_FINDINGS" -eq 0 ]; then + echo "all checks clean" + exit 0 +fi +echo "$TOTAL_FINDINGS finding(s) across 8 checks" +exit 1 diff --git a/scripts/testing/tests/test_config_applied.py b/scripts/testing/tests/test_config_applied.py index 00c410e..08ffc1b 100644 --- a/scripts/testing/tests/test_config_applied.py +++ b/scripts/testing/tests/test_config_applied.py @@ -40,7 +40,8 @@ def test_makepkg_options_trimmed(host): @pytest.mark.attribution("archsetup") -@pytest.mark.parametrize("rel", ["dns.conf", "wifi-privacy.conf"]) +@pytest.mark.parametrize("rel", ["dns.conf", "wifi-privacy.conf", + "tunnel-dns-over-tls.conf"]) def test_networkmanager_dropin(host, rel): assert host.file("/etc/NetworkManager/conf.d/%s" % rel).exists diff --git a/scripts/zz-bluetooth-resume b/scripts/zz-bluetooth-resume new file mode 100755 index 0000000..4273339 --- /dev/null +++ b/scripts/zz-bluetooth-resume @@ -0,0 +1,86 @@ +#!/bin/sh +# SPDX-License-Identifier: GPL-3.0-or-later +# zz-bluetooth-resume - put bluetooth back after a sleep cycle. +# +# A systemd-sleep hook. Two things break bluetooth across sleep on a TLP +# laptop, and nothing else on the machine fixes either one. +# +# 1. The rfkill soft-block is not restored. systemd-rfkill would do it, and +# it is masked here deliberately -- it fights TLP's radio handling, so +# configure_tlp_power masks it and TLP owns radios instead. TLP's own +# sleep hook runs `tlp resume`, but its setting is +# DEVICES_TO_ENABLE_ON_STARTUP: startup, not resume. TLP has no ON_RESUME +# at all, so the resume edge has no owner. WiFi survives only because +# NetworkManager unblocks itself; bluetooth has no equivalent. +# +# 2. The controller comes back wedged from a hibernate. It reports powered +# and unblocked while scanning finds nothing whatever -- zero devices +# where the same room gave seventeen a minute later -- and bluetoothd +# logs "Failed to set mode" and "Failed to add device <mac>" at the +# instant of resume. Reloading btusb clears it. +# +# Both observed on velox 2026-08-21, on the first suspend-then-hibernate cycle +# after hibernate was switched back on. The second symptom is why unblocking +# alone is not enough: rfkill was cleared by hand and scanning still returned +# nothing until the driver was reloaded. +# +# The hook re-asserts TLP's own declared intent rather than inventing a policy. +# A machine whose TLP config does not ask for bluetooth keeps it off, which is +# what stops this from overriding a deliberate block at every wakeup. +# +# The zz- prefix orders it after TLP's own hook, so `tlp resume` has finished +# before this runs. +# +# Test seams: BTR_RFKILL, BTR_MODPROBE, BTR_TLP_CONF, BTR_TLP_CONF_DIR, +# BTR_SETTLE (seconds to wait between driver unload and load). + +set -u + +RFKILL="${BTR_RFKILL:-rfkill}" +MODPROBE="${BTR_MODPROBE:-modprobe}" +TLP_CONF="${BTR_TLP_CONF:-/etc/tlp.conf}" +TLP_CONF_DIR="${BTR_TLP_CONF_DIR:-/etc/tlp.d}" +SETTLE="${BTR_SETTLE:-1}" + +# post only. The pre phase has nothing to do, and acting there would fight the +# suspend it is about to run. +[ "${1:-}" = "post" ] || exit 0 + +# Does TLP ask for bluetooth on this machine? Comments are stripped first, so a +# commented-out example in the stock config cannot be read as a policy. Both +# the main file and any drop-in count, and the last assignment wins the same +# way TLP itself resolves them. +wants_bluetooth() { + cat "$TLP_CONF" "$TLP_CONF_DIR"/*.conf 2>/dev/null \ + | sed 's/#.*//' \ + | awk -F= '/DEVICES_TO_ENABLE_ON_STARTUP/ { v = $2 } END { print v }' \ + | tr -d '"' \ + | tr ' ' '\n' \ + | grep -qx "bluetooth" +} + +wants_bluetooth || exit 0 + +# The wedge follows a hibernate, which reinitialises the controller from a +# saved image. A plain suspend brings USB back intact, so reloading there would +# tear down a working adapter for nothing. +# +# suspend-then-hibernate reports that name whether or not it reached the +# hibernate stage, so this reloads on a cycle that only suspended. That is the +# cheap side of the trade: a couple of seconds against an adapter that answers +# nothing until someone notices and reloads it by hand. +case "${2:-}" in + hibernate|suspend-then-hibernate) + "$MODPROBE" -r btusb 2>/dev/null || true + [ "$SETTLE" = "0" ] || sleep "$SETTLE" + "$MODPROBE" btusb 2>/dev/null || true + ;; +esac + +# After the reload, not before: a freshly loaded btusb can come up soft-blocked +# and would undo an earlier unblock. +"$RFKILL" unblock bluetooth 2>/dev/null || true + +# Never fail. systemd-sleep logs a failing hook, and that noise outlives the +# cause it describes; nothing here is worth alarming a resume over. +exit 0 |
