1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
|
#!/bin/sh
# SPDX-License-Identifier: GPL-3.0-or-later
# post-rebuild-check - the five checks a rebuilt machine actually needs.
#
# A rebuilt machine looks finished and isn't. Five gaps surfaced on velox
# within two days of the 2026-08-13 reinstall, and three of them LOOKED
# fine: a stowed unit file, an enabled timer, a present git clone. Each
# check below is cheap and turns a silent no-op into a visible line:
#
# 1. failed systemd units, user and system scope (calendar-sync failed
# every 15 minutes for two days with nobody watching)
# 2. user unit files present but not enabled (roam-sync and
# signal-receive came back linked and inert -- a unit file being
# present is not the same as running)
# 3. tracked *.example files whose real sibling is missing (three
# *.local.el were gone on velox; the .example survives in git,
# the real file never does)
# 4. gitignore-mode projects missing tooling paths their own .gitignore
# names (a reinstall drops every such project's untracked working
# state -- 374 files in .emacs.d's case -- and nothing carries it)
# 5. signal-cli holds no registered account (velox lost its
# registration, and because agent-text relays into this machine,
# that silently broke paging for the WHOLE fleet)
#
# The .gitignore rule in check 4 is what scopes it: a tooling path is only
# expected where the project's own .gitignore names it, so a project that
# never had a todo.org never flags. The ignore file is the project's own
# record of what it is supposed to hold untracked.
#
# EVERY PROBE FAILS CLOSED. A check that cannot run reports a finding, never
# a pass. This matters more here than anywhere else in the script: the whole
# point is catching silent no-ops, so a silent no-op in the checker would be
# the worst possible defect. `systemctl --user` exits 1 with empty output
# when there is no user bus -- over ssh, from cron, under sudo, on a TTY
# before the graphical session starts -- and reading that as "no failed
# units" would report a machine as healthy exactly when nothing was checked.
#
# Exit 0 when every check is clean, 1 when any check found something,
# 2 on usage error.
#
# Test seams (env; for each, set-but-empty means "the probe ran and found
# nothing", unset means "run the real probe"):
# PRC_FAILED_UNITS newline list of "scope:unit" (scope user|system)
# PRC_UNIT_STATES newline list of "unit-file state" replacing the
# user-unit-dir enumeration + is-enabled calls
# PRC_LOCAL_SCAN_ROOTS NEWLINE-separated roots for the *.example scan
# (default: ~/.emacs.d ~/.dotfiles)
# PRC_PROJECT_ROOTS NEWLINE-separated project dirs for check 4
# (default: ~/code/* ~/projects/* ~/.emacs.d
# ~/.dotfiles)
# PRC_SIGNAL_ACCOUNTS signal-cli listAccounts output; "" = no account,
# the special value MISSING = binary absent
# PRC_SYSTEMCTL path to the systemctl binary (a fake, under test)
# PRC_SYSTEMCTL_TIMEOUT seconds to allow each systemctl call (default 5)
#
# Roots are newline-separated, not space-separated, because a POSIX
# `for root in $var` splits on spaces and turns one real directory into
# several imaginary missing ones.
usage() {
cat <<'EOF'
post-rebuild-check - verify a rebuilt machine is actually finished
Runs the five checks that caught velox's 2026-08 reinstall gaps: failed
units, present-but-inert user units, orphaned *.example configs, missing
per-project tooling state, and the signal-cli registration.
Usage: post-rebuild-check [--help]
Exit 0 when every check is clean, 1 when any check found something.
Every probe fails closed: a check that cannot run is a finding, not a pass.
EOF
}
case "${1:-}" in
--help|-h) usage; exit 0 ;;
"") ;;
*) echo "post-rebuild-check: unknown argument: $1" >&2; usage >&2; exit 2 ;;
esac
# Own the internal flags rather than inheriting them, so a caller's unrelated
# variable of the same name cannot manufacture or mask a finding.
TOTAL_FINDINGS=0
CHECK_FINDINGS=0
FINDING_LINES=""
signal_missing=""
# Every systemctl call is bounded. A wedged user manager spins and answers
# nothing -- seen live on velox 2026-08-17, where `is-enabled`, `cat`, and
# `list-unit-files` all hung while `list-units` still returned. Unbounded, this
# script would hang on the first unit and never reach the remaining checks,
# which is a worse failure than reporting nothing: a check that hangs is its
# own outage, and the machine most in need of checking is the one it hangs on.
# A timeout yields empty output and a non-zero status, and both are already
# handled as findings, so bounding the call is all that is needed to fail closed.
SCTL_TIMEOUT=${PRC_SYSTEMCTL_TIMEOUT:-5}
SYSTEMCTL=${PRC_SYSTEMCTL:-systemctl}
sctl() {
if command -v timeout >/dev/null 2>&1; then
timeout "$SCTL_TIMEOUT" "$SYSTEMCTL" "$@"
else
# Say so rather than dropping the bound silently: without timeout a
# wedged manager hangs this run indefinitely, and the whole point of
# the bound is that a check which hangs reports nothing at all.
[ -n "${sctl_unbounded_warned:-}" ] || {
echo "post-rebuild-check: timeout(1) not found — systemctl calls are UNBOUNDED and may hang" >&2
sctl_unbounded_warned=1
}
"$SYSTEMCTL" "$@"
fi
}
WORK=${TMPDIR:-/tmp}/.post-rebuild-check.$$
if ! mkdir "$WORK" 2>/dev/null; then
# Every check stages its input through a file in here. Without it each
# loop would read nothing and every check would come back clean, which is
# the one failure this script must never produce.
echo "post-rebuild-check: cannot create a work directory under ${TMPDIR:-/tmp}" >&2
echo " nothing was checked; this is not a pass" >&2
exit 1
fi
trap 'rm -rf "$WORK"' EXIT HUP INT TERM
STAGE="$WORK/stage"
finding() {
CHECK_FINDINGS=$((CHECK_FINDINGS + 1))
TOTAL_FINDINGS=$((TOTAL_FINDINGS + 1))
FINDING_LINES="${FINDING_LINES} DEVIATION: $1
"
}
# Print the check's one visible line, then its findings. The visible line
# is the point: a silent no-op is exactly what let the gaps sit unseen.
report() {
if [ "$CHECK_FINDINGS" -eq 0 ]; then
echo "$1 — ok"
else
echo "$1 — $CHECK_FINDINGS finding(s)"
printf '%s' "$FINDING_LINES"
fi
CHECK_FINDINGS=0
FINDING_LINES=""
}
# Stage a value into $STAGE for the read loops. A failed write is fatal for
# the same reason a missing work directory is.
stage() {
if ! printf '%s\n' "$1" > "$STAGE" 2>/dev/null; then
echo "post-rebuild-check: cannot write $STAGE" >&2
echo " nothing was checked; this is not a pass" >&2
exit 1
fi
}
# --- 1. failed units ------------------------------------------------------
if [ -n "${PRC_FAILED_UNITS+set}" ]; then
failed=$PRC_FAILED_UNITS
else
failed=""
if user_out=$(sctl --user list-units --state=failed --no-legend --plain 2>/dev/null); then
failed=$(printf '%s' "$user_out" | awk 'NF {print "user:"$1}')
else
finding "could not query user units (no user bus?) — nothing was checked in this scope"
fi
if sys_out=$(sctl list-units --state=failed --no-legend --plain 2>/dev/null); then
failed="$failed
$(printf '%s' "$sys_out" | awk 'NF {print "system:"$1}')"
else
finding "could not query system units — nothing was checked in this scope"
fi
fi
stage "$failed"
while IFS= read -r line; do
[ -n "$line" ] || continue
scope=${line%%:*}
unit=${line#*:}
finding "$scope unit failed: $unit"
done < "$STAGE"
report "check 1/5: failed units"
# --- 2. user unit files present but not enabled ---------------------------
if [ -n "${PRC_UNIT_STATES+set}" ]; then
states=$PRC_UNIT_STATES
else
states=""
unit_dir="${XDG_CONFIG_HOME:-$HOME/.config}/systemd/user"
if [ ! -d "$unit_dir" ]; then
finding "no user unit directory at $unit_dir — nothing was checked"
else
for f in "$unit_dir"/*.timer "$unit_dir"/*.service; do
# -L as well as -e: a stow symlink whose target moved in the
# rebuild is exactly the "looked fine" case this check is for,
# and -e is false for a broken link.
[ -e "$f" ] || [ -L "$f" ] || continue
name=$(basename "$f")
# A link with nothing behind it is its own finding, decided on the
# filesystem rather than from systemd. `is-enabled` calls a
# dangling link "not-found" -- the same answer it gives for a unit
# that was never installed -- so routing this through the state
# table below would drop it silently.
if [ -L "$f" ] && [ ! -e "$f" ]; then
finding "stowed unit file points at a missing target: $name"
continue
fi
# is-enabled exits non-zero AND prints a state for disabled and
# linked, so the exit code cannot distinguish "this unit is
# disabled" from "the query failed". The output can: a real answer
# is always a word. Empty means no answer, which is a finding
# rather than a silent skip -- with no user bus (ssh, cron, sudo,
# a TTY before the graphical session) every unit answers empty,
# and treating that as unknown-so-ignore would pass the machine
# while reading nothing at all.
#
# No separate bus probe: `is-system-running` and
# `show-environment` both block here, and a check that can hang is
# its own outage.
state=$(sctl --user is-enabled "$name" 2>/dev/null)
if [ -z "$state" ]; then
finding "could not read the enablement state of $name — it was not checked"
continue
fi
states="${states}${name} ${state}
"
done
fi
fi
stage "$states"
# A second copy for the sibling-timer lookup below, so the awk that reads it
# is never the same open file as the loop reading it.
cp "$STAGE" "$WORK/states" 2>/dev/null || {
echo "post-rebuild-check: cannot write $WORK/states" >&2
echo " nothing was checked; this is not a pass" >&2; exit 1; }
while read -r name state; do
[ -n "$name" ] || continue
case "$state" in
disabled|linked) ;;
*) continue ;;
esac
# A timer-activated service is SUPPOSED to sit linked-and-not-enabled:
# the timer owns activation, and enabling the service as well would run
# it at boot on top of its schedule. So a service is suppressed only when
# its sibling timer can actually start it (enabled), or when the timer is
# itself inert and therefore the finding already -- reporting both would
# name one gap twice. A masked, static, or not-found timer starts
# nothing, so the service beneath it is as dead as one with no timer.
case "$name" in
*.service)
timer="${name%.service}.timer"
tstate=$(awk -v t="$timer" '$1 == t {print $2; exit}' "$WORK/states")
# enabled-runtime (enabled until reboot) and generated (something
# produced and installed it) are live activation paths, so the
# service under one is being started and is not a finding.
# disabled and linked suppress for a different reason: the timer
# is then the finding itself, reported in its own right.
#
# "indirect" deliberately does NOT suppress. It means the unit
# file itself is not enabled, only that some Also= relative might
# be, so nothing here is known to start the service. The
# fail-closed rule says the uncertain case flags.
case "$tstate" in
enabled|enabled-runtime|generated) continue ;;
disabled|linked) continue ;;
esac
;;
esac
finding "unit file present but not enabled: $name ($state)"
done < "$STAGE"
report "check 2/5: unit files"
# --- 3. *.example files whose real sibling is missing ---------------------
if [ -n "${PRC_LOCAL_SCAN_ROOTS+set}" ]; then
scan_roots=$PRC_LOCAL_SCAN_ROOTS
else
scan_roots="$HOME/.emacs.d
$HOME/.dotfiles"
fi
printf '%s\n' "$scan_roots" > "$WORK/roots" 2>/dev/null || {
echo "post-rebuild-check: cannot write $WORK/roots" >&2; exit 1; }
while IFS= read -r root; do
[ -n "$root" ] || continue
if [ ! -d "$root" ]; then
finding "scan root missing: $root"
continue
fi
# Vendored package trees ship their own .example docs; those belong to
# the package, not to this machine, so they are noise in front of the
# real findings this check exists for.
#
# -prune, not -not -path: the latter filters find's OUTPUT while still
# descending, so an unreadable directory inside a tree we deliberately
# ignore would set find's exit status and be reported as an unscanned
# part of the root. Pruning means those trees are never entered, so the
# exit status only reflects places this check actually wanted to read.
#
# That status matters: find exits non-zero when it cannot descend
# somewhere, having printed only what it could reach. Discarding it would
# hide every orphan under an unreadable directory behind a clean "ok",
# which is the defect this script exists to catch.
if ! find "$root" \
\( -name .git -o -name elpa -o -name straight \
-o -name node_modules -o -name .venv \) -prune \
-o -name '*.example' -print > "$WORK/examples" 2>/dev/null; then
finding "could not fully scan $root — part of it was not checked"
fi
while IFS= read -r ex; do
[ -n "$ex" ] || continue
# -e, so a sibling that exists only as a dangling symlink counts as
# missing. It is not a config the machine can read.
[ -e "${ex%.example}" ] || finding "example without its real file: $ex"
done < "$WORK/examples"
done < "$WORK/roots"
report "check 3/5: local files"
# --- 4. gitignore-mode projects missing their tooling ---------------------
if [ -n "${PRC_PROJECT_ROOTS+set}" ]; then
projects=$PRC_PROJECT_ROOTS
else
projects=$(ls -d "$HOME"/code/*/ "$HOME"/projects/*/ 2>/dev/null; \
printf '%s\n%s\n' "$HOME/.emacs.d" "$HOME/.dotfiles")
fi
printf '%s\n' "$projects" > "$WORK/projects" 2>/dev/null || {
echo "post-rebuild-check: cannot write $WORK/projects" >&2; exit 1; }
# CLAUDE.md is deliberately absent from this set. It is seed-only --
# install-lang writes it once and the project owns it afterward -- so most
# projects legitimately never have one, and ratio shows the identical
# absences in the identical projects. That match is what proves it is the
# steady state rather than reinstall drift, and flagging it would put nine
# standing findings in front of every real one.
#
# The list is fed to the inner loop straight from a heredoc rather than
# staged through a file. It is a constant, so a file bought nothing and cost
# a fifth unguarded write: had it failed (a full tmpfs, say) the inner loop
# would read nothing and every project would pass silently, which is the one
# outcome this script must never produce. The heredoc is the inner loop's own
# stdin and leaves the outer loop's redirect alone.
while IFS= read -r proj; do
[ -n "$proj" ] || continue
proj=${proj%/}
# -e not -d: in a worktree or submodule .git is a file naming the real
# gitdir, and a -d test would skip those projects silently.
[ -e "$proj/.git" ] || continue
[ -f "$proj/.gitignore" ] || continue
while read -r disk pattern; do
# Both the anchored (/.ai/) and unanchored (.ai/) ignore styles exist
# across the fleet; the sweep-gitignore audit hit exactly that split.
#
# grep exits 1 for no-match and 2 for an error, so the two are told
# apart rather than both read as "the ignore file does not name this".
# An unreadable .gitignore would otherwise pass the whole project.
grep -Eq "^/?${pattern}/?\$" "$proj/.gitignore" 2>/dev/null
case $? in
0) [ -e "$proj/$disk" ] \
|| finding "$proj: .gitignore names $disk but it is missing on disk" ;;
1) ;;
*) finding "$proj: could not read .gitignore — the project was not checked"
break ;;
esac
done <<'EOF'
.ai \.ai
.claude \.claude
todo.org todo\.org
inbox inbox
EOF
done < "$WORK/projects"
report "check 4/5: project tooling"
# --- 5. signal-cli registration -------------------------------------------
if [ -n "${PRC_SIGNAL_ACCOUNTS+set}" ]; then
accounts=$PRC_SIGNAL_ACCOUNTS
if [ "$accounts" = "MISSING" ]; then
accounts=""
signal_missing=1
fi
else
if command -v signal-cli >/dev/null 2>&1; then
if ! accounts=$(signal-cli listAccounts 2>/dev/null); then
accounts=""
finding "signal-cli listAccounts failed — the registration was not checked"
signal_missing=skip
fi
else
accounts=""
signal_missing=1
fi
fi
if [ "$signal_missing" = 1 ]; then
finding "signal-cli is not installed — paging relies on it fleet-wide"
elif [ -z "$signal_missing" ] && [ -z "$accounts" ]; then
finding "no signal account registered — agent-text relays into this machine, so paging breaks for the whole fleet"
fi
report "check 5/5: signal registration"
# --- summary --------------------------------------------------------------
if [ "$TOTAL_FINDINGS" -eq 0 ]; then
echo "all checks clean"
exit 0
fi
echo "$TOTAL_FINDINGS finding(s) across 5 checks"
exit 1
|