aboutsummaryrefslogtreecommitdiff
path: root/scripts/post-rebuild-check
blob: ee77a19cb77e7e4b8b3a49d7ca99ce5ad16ee97f (plain)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
#!/bin/sh
# SPDX-License-Identifier: GPL-3.0-or-later
# post-rebuild-check - the eight checks a rebuilt machine actually needs.
#
# A rebuilt machine looks finished and isn't. Five gaps surfaced on velox
# within two days of the 2026-08-13 reinstall, and three of them LOOKED
# fine: a stowed unit file, an enabled timer, a present git clone. Each
# check below is cheap and turns a silent no-op into a visible line:
#
#   1. failed systemd units, user and system scope (calendar-sync failed
#      every 15 minutes for two days with nobody watching)
#   2. user unit files present but not enabled (roam-sync and
#      signal-receive came back linked and inert -- a unit file being
#      present is not the same as running)
#   3. tracked *.example files whose real sibling is missing (three
#      *.local.el were gone on velox; the .example survives in git,
#      the real file never does)
#   4. gitignore-mode projects missing tooling paths their own .gitignore
#      names (a reinstall drops every such project's untracked working
#      state -- 374 files in .emacs.d's case -- and nothing carries it)
#   5. signal-cli holds no registered account (velox lost its registration
#      in the rebuild; at the time agent-text relayed into velox, so that
#      silently broke paging for the WHOLE fleet. agent-text now walks
#      AGENT_TEXT_RELAYS in order and skips itself, so an unregistered
#      machine is only fatal when no relay host is registered either)
#   6. every NTP source is named by hostname (a wrong clock fails the
#      DoT/DNSSEC validation this machine's DNS runs on, so nothing
#      resolves -- including the NTP pool that would fix the clock; velox
#      deadlocked exactly this way 2026-08-19 and needed a second device)
#   7. hypridle installed but not running (nothing then triggers idle lock
#      or suspend, so a laptop runs until its battery is gone -- which is
#      how velox reset the RTC that caused check 6's deadlock in the first
#      place; a caffeine remembered from an earlier boot is the known cause)
#   8. a working repo cloned from the read-only https endpoint (correct
#      for a stranger with no key on the server, wrong for this machine,
#      which finds out at the first push with a 403 -- velox's dotfiles
#      remote sat that way for four days after its rebuild)
#
# The .gitignore rule in check 4 is what scopes it: a tooling path is only
# expected where the project's own .gitignore names it, so a project that
# never had a todo.org never flags. The ignore file is the project's own
# record of what it is supposed to hold untracked.
#
# EVERY PROBE FAILS CLOSED. A check that cannot run reports a finding, never
# a pass. This matters more here than anywhere else in the script: the whole
# point is catching silent no-ops, so a silent no-op in the checker would be
# the worst possible defect. `systemctl --user` exits 1 with empty output
# when there is no user bus -- over ssh, from cron, under sudo, on a TTY
# before the graphical session starts -- and reading that as "no failed
# units" would report a machine as healthy exactly when nothing was checked.
#
# Exit 0 when every check is clean, 1 when any check found something,
# 2 on usage error.
#
# Test seams (env; for each, set-but-empty means "the probe ran and found
# nothing", unset means "run the real probe"):
#   PRC_FAILED_UNITS      newline list of "scope:unit" (scope user|system)
#   PRC_UNIT_STATES       newline list of "unit-file state" replacing the
#                         user-unit-dir enumeration + is-enabled calls
#   PRC_LOCAL_SCAN_ROOTS  NEWLINE-separated roots for the *.example scan
#                         (default: ~/.emacs.d ~/.dotfiles)
#   PRC_PROJECT_ROOTS     NEWLINE-separated project dirs for check 4
#                         (default: ~/code/* ~/projects/* ~/.emacs.d
#                         ~/.dotfiles)
#   PRC_SIGNAL_ACCOUNTS   signal-cli listAccounts output; "" = no account,
#                         the special value MISSING = binary absent
#   PRC_NTP_SOURCES       newline list of configured NTP server addresses;
#                         the special value MISSING = no NTP daemon active
#   PRC_CHRONY_CONF       path to chrony.conf (a fixture, under test) -- the
#                         confdir it names is what decides which drop-ins count
#   PRC_IDLE_DAEMON       pgrep output for hypridle; "" = installed but not
#                         running, the special value MISSING = not installed
#   PRC_REPO_REMOTES      newline list of "path origin-url"; an empty URL
#                         means origin could not be read
#   PRC_UNITS_EXPECTED_DISABLED
#                         newline list of units whose not-enabled state is
#                         deliberate here, replacing the file below
#   PRC_UNITS_EXPECTED_DISABLED_FILE
#                         path to that list (default:
#                         $XDG_CONFIG_HOME/post-rebuild-check/units-expected-disabled).
#                         One unit per line, # starts a comment. Machine-local
#                         on purpose: the same unit is correctly enabled on one
#                         box and not another
#   PRC_SYSTEMCTL         path to the systemctl binary (a fake, under test)
#   PRC_SYSTEMCTL_TIMEOUT seconds to allow each systemctl call (default 5)
#
# Roots are newline-separated, not space-separated, because a POSIX
# `for root in $var` splits on spaces and turns one real directory into
# several imaginary missing ones.

usage() {
    cat <<'EOF'
post-rebuild-check - verify a rebuilt machine is actually finished

Runs the eight checks that caught velox's 2026-08 reinstall gaps: failed
units, present-but-inert user units, orphaned *.example configs, missing
per-project tooling state, the signal-cli registration, whether time sync
can recover from a wrong clock without DNS, whether anything still
triggers idle lock and suspend, and whether the working repos can push.

Usage: post-rebuild-check [--help]

Exit 0 when every check is clean, 1 when any check found something.
Every probe fails closed: a check that cannot run is a finding, not a pass.
EOF
}

case "${1:-}" in
    --help|-h) usage; exit 0 ;;
    "") ;;
    *) echo "post-rebuild-check: unknown argument: $1" >&2; usage >&2; exit 2 ;;
esac

# Own the internal flags rather than inheriting them, so a caller's unrelated
# variable of the same name cannot manufacture or mask a finding.
TOTAL_FINDINGS=0
CHECK_FINDINGS=0
FINDING_LINES=""
signal_missing=""
ntp_missing=""
idle_absent=""

# Every systemctl call is bounded. A wedged user manager spins and answers
# nothing -- seen live on velox 2026-08-17, where `is-enabled`, `cat`, and
# `list-unit-files` all hung while `list-units` still returned. Unbounded, this
# script would hang on the first unit and never reach the remaining checks,
# which is a worse failure than reporting nothing: a check that hangs is its
# own outage, and the machine most in need of checking is the one it hangs on.
# A timeout yields empty output and a non-zero status, and both are already
# handled as findings, so bounding the call is all that is needed to fail closed.
CHRONY_CONF=${PRC_CHRONY_CONF:-/etc/chrony.conf}
SCTL_TIMEOUT=${PRC_SYSTEMCTL_TIMEOUT:-5}
SYSTEMCTL=${PRC_SYSTEMCTL:-systemctl}

sctl() {
    if command -v timeout >/dev/null 2>&1; then
        timeout "$SCTL_TIMEOUT" "$SYSTEMCTL" "$@"
    else
        # Say so rather than dropping the bound silently: without timeout a
        # wedged manager hangs this run indefinitely, and the whole point of
        # the bound is that a check which hangs reports nothing at all.
        [ -n "${sctl_unbounded_warned:-}" ] || {
            echo "post-rebuild-check: timeout(1) not found — systemctl calls are UNBOUNDED and may hang" >&2
            sctl_unbounded_warned=1
        }
        "$SYSTEMCTL" "$@"
    fi
}

WORK=${TMPDIR:-/tmp}/.post-rebuild-check.$$
if ! mkdir "$WORK" 2>/dev/null; then
    # Every check stages its input through a file in here. Without it each
    # loop would read nothing and every check would come back clean, which is
    # the one failure this script must never produce.
    echo "post-rebuild-check: cannot create a work directory under ${TMPDIR:-/tmp}" >&2
    echo "  nothing was checked; this is not a pass" >&2
    exit 1
fi
trap 'rm -rf "$WORK"' EXIT HUP INT TERM

STAGE="$WORK/stage"

finding() {
    CHECK_FINDINGS=$((CHECK_FINDINGS + 1))
    TOTAL_FINDINGS=$((TOTAL_FINDINGS + 1))
    FINDING_LINES="${FINDING_LINES}  DEVIATION: $1
"
}

# Print the check's one visible line, then its findings. The visible line
# is the point: a silent no-op is exactly what let the gaps sit unseen.
report() {
    if [ "$CHECK_FINDINGS" -eq 0 ]; then
        echo "$1 — ok"
    else
        echo "$1 — $CHECK_FINDINGS finding(s)"
        printf '%s' "$FINDING_LINES"
    fi
    CHECK_FINDINGS=0
    FINDING_LINES=""
}

# Stage a value into $STAGE for the read loops. A failed write is fatal for
# the same reason a missing work directory is.
stage() {
    if ! printf '%s\n' "$1" > "$STAGE" 2>/dev/null; then
        echo "post-rebuild-check: cannot write $STAGE" >&2
        echo "  nothing was checked; this is not a pass" >&2
        exit 1
    fi
}

# --- 1. failed units ------------------------------------------------------

if [ -n "${PRC_FAILED_UNITS+set}" ]; then
    failed=$PRC_FAILED_UNITS
else
    failed=""
    if user_out=$(sctl --user list-units --state=failed --no-legend --plain 2>/dev/null); then
        failed=$(printf '%s' "$user_out" | awk 'NF {print "user:"$1}')
    else
        finding "could not query user units (no user bus?) — nothing was checked in this scope"
    fi
    if sys_out=$(sctl list-units --state=failed --no-legend --plain 2>/dev/null); then
        failed="$failed
$(printf '%s' "$sys_out" | awk 'NF {print "system:"$1}')"
    else
        finding "could not query system units — nothing was checked in this scope"
    fi
fi
stage "$failed"
while IFS= read -r line; do
    [ -n "$line" ] || continue
    scope=${line%%:*}
    unit=${line#*:}
    finding "$scope unit failed: $unit"
done < "$STAGE"
report "check 1/8: failed units"

# --- 2. user unit files present but not enabled ---------------------------
#
# Units nothing intends to enable here are read from a machine-local list.
# "Enabled" is this check's proxy for "will actually run", and the proxy is
# wrong for a unit nobody means to enable on this box. velox carries four, for
# four different reasons: geoclue-agent is redundant because hyprland's
# exec-once starts the binary directly, emacs is started on demand by
# emacsclient, obs-record-watchdog only matters while recording, and
# obsbot-wb-guard needs an OBSBOT the machine does not have. Left unexempted
# they report at every run, and four permanent lines in front of every real one
# teach you to skim the output -- the same argument check 4 makes about
# CLAUDE.md.
#
# Machine-local rather than a marker in the shared unit file, because
# obsbot-wb-guard is correctly ENABLED on ratio. One unit, a different right
# answer per machine, so the shared file cannot hold the answer.
#
# An entry that turns out to be enabled after all is still a finding. Without
# that the list rots into somewhere real findings go to die, which is worse
# than the noise it removes.

EXPECT_DISABLED_FILE="${PRC_UNITS_EXPECTED_DISABLED_FILE:-${XDG_CONFIG_HOME:-$HOME/.config}/post-rebuild-check/units-expected-disabled}"
if [ -n "${PRC_UNITS_EXPECTED_DISABLED+set}" ]; then
    expect_disabled=$PRC_UNITS_EXPECTED_DISABLED
elif [ -f "$EXPECT_DISABLED_FILE" ]; then
    expect_disabled=$(cat "$EXPECT_DISABLED_FILE" 2>/dev/null)
else
    expect_disabled=""
fi
# Strip comments and blanks once, here, so the membership test below is a
# plain word match. The reason a unit is exempt is the most useful thing about
# the entry, so the format has to carry one.
expect_disabled=$(printf '%s\n' "$expect_disabled" \
    | sed 's/#.*//' | awk 'NF {print $1}')

if [ -n "${PRC_UNIT_STATES+set}" ]; then
    states=$PRC_UNIT_STATES
else
    states=""
    unit_dir="${XDG_CONFIG_HOME:-$HOME/.config}/systemd/user"
    if [ ! -d "$unit_dir" ]; then
        finding "no user unit directory at $unit_dir — nothing was checked"
    else
        for f in "$unit_dir"/*.timer "$unit_dir"/*.service; do
            # -L as well as -e: a stow symlink whose target moved in the
            # rebuild is exactly the "looked fine" case this check is for,
            # and -e is false for a broken link.
            [ -e "$f" ] || [ -L "$f" ] || continue
            name=$(basename "$f")
            # A link with nothing behind it is its own finding, decided on the
            # filesystem rather than from systemd. `is-enabled` calls a
            # dangling link "not-found" -- the same answer it gives for a unit
            # that was never installed -- so routing this through the state
            # table below would drop it silently.
            if [ -L "$f" ] && [ ! -e "$f" ]; then
                finding "stowed unit file points at a missing target: $name"
                continue
            fi
            # is-enabled exits non-zero AND prints a state for disabled and
            # linked, so the exit code cannot distinguish "this unit is
            # disabled" from "the query failed". The output can: a real answer
            # is always a word. Empty means no answer, which is a finding
            # rather than a silent skip -- with no user bus (ssh, cron, sudo,
            # a TTY before the graphical session) every unit answers empty,
            # and treating that as unknown-so-ignore would pass the machine
            # while reading nothing at all.
            #
            # No separate bus probe: `is-system-running` and
            # `show-environment` both block here, and a check that can hang is
            # its own outage.
            state=$(sctl --user is-enabled "$name" 2>/dev/null)
            if [ -z "$state" ]; then
                finding "could not read the enablement state of $name — it was not checked"
                continue
            fi
            states="${states}${name} ${state}
"
        done
    fi
fi
stage "$states"
# A second copy for the sibling-timer lookup below, so the awk that reads it
# is never the same open file as the loop reading it.
cp "$STAGE" "$WORK/states" 2>/dev/null || {
    echo "post-rebuild-check: cannot write $WORK/states" >&2
    echo "  nothing was checked; this is not a pass" >&2; exit 1; }
while read -r name state; do
    [ -n "$name" ] || continue
    case "$state" in
        disabled|linked) ;;
        *) continue ;;
    esac
    # A timer-activated service is SUPPOSED to sit linked-and-not-enabled:
    # the timer owns activation, and enabling the service as well would run
    # it at boot on top of its schedule. So a service is suppressed only when
    # its sibling timer can actually start it (enabled), or when the timer is
    # itself inert and therefore the finding already -- reporting both would
    # name one gap twice. A masked, static, or not-found timer starts
    # nothing, so the service beneath it is as dead as one with no timer.
    case "$name" in
        *.service)
            timer="${name%.service}.timer"
            tstate=$(awk -v t="$timer" '$1 == t {print $2; exit}' "$WORK/states")
            # enabled-runtime (enabled until reboot) and generated (something
            # produced and installed it) are live activation paths, so the
            # service under one is being started and is not a finding.
            # disabled and linked suppress for a different reason: the timer
            # is then the finding itself, reported in its own right.
            #
            # "indirect" deliberately does NOT suppress. It means the unit
            # file itself is not enabled, only that some Also= relative might
            # be, so nothing here is known to start the service. The
            # fail-closed rule says the uncertain case flags.
            case "$tstate" in
                enabled|enabled-runtime|generated) continue ;;
                disabled|linked) continue ;;
            esac
            ;;
    esac
    # Deliberately not enabled on this machine. Checked last, so it suppresses
    # only this finding and never the dangling-link one decided above on the
    # filesystem.
    case "
$expect_disabled
" in
        *"
$name
"*) continue ;;
    esac
    finding "unit file present but not enabled: $name ($state)"
done < "$STAGE"
# The exemption list, checked in the other direction. An entry whose unit is
# enabled after all suppresses nothing, and leaving it there is how the list
# turns into a place real findings go to die. The loop above cannot catch this:
# it skips any state that is not disabled or linked, so an enabled unit never
# reaches it.
printf '%s\n' "$expect_disabled" > "$WORK/expect" 2>/dev/null || {
    echo "post-rebuild-check: cannot write $WORK/expect" >&2
    echo "  nothing was checked; this is not a pass" >&2; exit 1; }
while IFS= read -r name; do
    [ -n "$name" ] || continue
    estate=$(awk -v u="$name" '$1 == u {print $2; exit}' "$WORK/states")
    case "$estate" in
        enabled|enabled-runtime)
            finding "$name is listed as expected-disabled but is $estate — drop the stale exemption" ;;
    esac
done < "$WORK/expect"
report "check 2/8: unit files"

# --- 3. *.example files whose real sibling is missing ---------------------

if [ -n "${PRC_LOCAL_SCAN_ROOTS+set}" ]; then
    scan_roots=$PRC_LOCAL_SCAN_ROOTS
else
    scan_roots="$HOME/.emacs.d
$HOME/.dotfiles"
fi
printf '%s\n' "$scan_roots" > "$WORK/roots" 2>/dev/null || {
    echo "post-rebuild-check: cannot write $WORK/roots" >&2; exit 1; }
while IFS= read -r root; do
    [ -n "$root" ] || continue
    if [ ! -d "$root" ]; then
        finding "scan root missing: $root"
        continue
    fi
    # Vendored package trees ship their own .example docs; those belong to
    # the package, not to this machine, so they are noise in front of the
    # real findings this check exists for.
    #
    # -prune, not -not -path: the latter filters find's OUTPUT while still
    # descending, so an unreadable directory inside a tree we deliberately
    # ignore would set find's exit status and be reported as an unscanned
    # part of the root. Pruning means those trees are never entered, so the
    # exit status only reflects places this check actually wanted to read.
    #
    # That status matters: find exits non-zero when it cannot descend
    # somewhere, having printed only what it could reach. Discarding it would
    # hide every orphan under an unreadable directory behind a clean "ok",
    # which is the defect this script exists to catch.
    if ! find "$root" \
         \( -name .git -o -name elpa -o -name straight \
            -o -name node_modules -o -name .venv \) -prune \
         -o -name '*.example' -print > "$WORK/examples" 2>/dev/null; then
        finding "could not fully scan $root — part of it was not checked"
    fi
    while IFS= read -r ex; do
        [ -n "$ex" ] || continue
        # -e, so a sibling that exists only as a dangling symlink counts as
        # missing. It is not a config the machine can read.
        [ -e "${ex%.example}" ] || finding "example without its real file: $ex"
    done < "$WORK/examples"
done < "$WORK/roots"
report "check 3/8: local files"

# --- 4. gitignore-mode projects missing their tooling ---------------------

if [ -n "${PRC_PROJECT_ROOTS+set}" ]; then
    projects=$PRC_PROJECT_ROOTS
else
    projects=$(ls -d "$HOME"/code/*/ "$HOME"/projects/*/ 2>/dev/null; \
               printf '%s\n%s\n' "$HOME/.emacs.d" "$HOME/.dotfiles")
fi
printf '%s\n' "$projects" > "$WORK/projects" 2>/dev/null || {
    echo "post-rebuild-check: cannot write $WORK/projects" >&2; exit 1; }
# CLAUDE.md is deliberately absent from this set. It is seed-only --
# install-lang writes it once and the project owns it afterward -- so most
# projects legitimately never have one, and ratio shows the identical
# absences in the identical projects. That match is what proves it is the
# steady state rather than reinstall drift, and flagging it would put nine
# standing findings in front of every real one.
#
# The list is fed to the inner loop straight from a heredoc rather than
# staged through a file. It is a constant, so a file bought nothing and cost
# a fifth unguarded write: had it failed (a full tmpfs, say) the inner loop
# would read nothing and every project would pass silently, which is the one
# outcome this script must never produce. The heredoc is the inner loop's own
# stdin and leaves the outer loop's redirect alone.
while IFS= read -r proj; do
    [ -n "$proj" ] || continue
    proj=${proj%/}
    # -e not -d: in a worktree or submodule .git is a file naming the real
    # gitdir, and a -d test would skip those projects silently.
    [ -e "$proj/.git" ] || continue
    [ -f "$proj/.gitignore" ] || continue
    while read -r disk pattern; do
        # Both the anchored (/.ai/) and unanchored (.ai/) ignore styles exist
        # across the fleet; the sweep-gitignore audit hit exactly that split.
        #
        # grep exits 1 for no-match and 2 for an error, so the two are told
        # apart rather than both read as "the ignore file does not name this".
        # An unreadable .gitignore would otherwise pass the whole project.
        grep -Eq "^/?${pattern}/?\$" "$proj/.gitignore" 2>/dev/null
        case $? in
            0) [ -e "$proj/$disk" ] \
                   || finding "$proj: .gitignore names $disk but it is missing on disk" ;;
            1) ;;
            *) finding "$proj: could not read .gitignore — the project was not checked"
               break ;;
        esac
    done <<'EOF'
.ai \.ai
.claude \.claude
todo.org todo\.org
inbox inbox
EOF
done < "$WORK/projects"
report "check 4/8: project tooling"

# --- 5. signal-cli registration -------------------------------------------

if [ -n "${PRC_SIGNAL_ACCOUNTS+set}" ]; then
    accounts=$PRC_SIGNAL_ACCOUNTS
    if [ "$accounts" = "MISSING" ]; then
        accounts=""
        signal_missing=1
    fi
else
    if command -v signal-cli >/dev/null 2>&1; then
        if ! accounts=$(signal-cli listAccounts 2>/dev/null); then
            accounts=""
            finding "signal-cli listAccounts failed — the registration was not checked"
            signal_missing=skip
        fi
    else
        accounts=""
        signal_missing=1
    fi
fi
if [ "$signal_missing" = 1 ]; then
    finding "signal-cli is not installed — paging relies on it fleet-wide"
elif [ -z "$signal_missing" ] && [ -z "$accounts" ]; then
    finding "no signal account registered — this machine can only page by relaying to one that has an account; if no host in AGENT_TEXT_RELAYS is registered either, the whole fleet loses paging"
fi
report "check 5/8: signal registration"

# --- 6. NTP can recover a wrong clock without DNS -------------------------
#
# The clock/DNS bootstrap deadlock. This machine resolves through DNSOverTLS
# with DNSSEC, and both validate against the wall clock, so a boot with a
# wrong clock resolves nothing at all. If every configured NTP source is named
# by hostname, the daemon that would correct the clock needs the DNS the clock
# is breaking, and the machine cannot recover without a second device --
# which is exactly what happened on velox 2026-08-19. One source addressed by
# IP breaks the cycle, so that is what this check looks for.

# True when the argument is an address rather than a name. An address needs no
# resolver, which is the whole property being checked.
is_ip_literal() {
    case "$1" in
        "") return 1 ;;
        *:*) case "$1" in *[!0-9A-Fa-f:]*) return 1 ;; esac
             return 0 ;;
        *[!0-9.]*) return 1 ;;
        *.*) return 0 ;;
    esac
    return 1
}

if [ -n "${PRC_NTP_SOURCES+set}" ]; then
    ntp_sources=$PRC_NTP_SOURCES
    if [ "$ntp_sources" = "MISSING" ]; then
        ntp_sources=""
        ntp_missing=1
    fi
elif sctl is-active chronyd >/dev/null 2>&1; then
    # The main file, plus any drop-in directory chrony.conf actually names.
    #
    # The confdir read is the load-bearing part. A drop-in is inert unless
    # chrony.conf points at its directory, and Arch's stock chrony.conf points
    # at none -- so globbing /etc/chrony.d unconditionally would find the
    # IP-addressed source, report the machine healthy, and be describing a file
    # chrony never opens. That is a false pass on exactly the misconfiguration
    # this check exists to catch, so the sources are read only from files
    # chrony is actually told to read.
    ntp_conf_files=$CHRONY_CONF
    for ntp_dir in $(awk '$1 == "confdir" || $1 == "sourcedir" { print $2 }' \
                         "$CHRONY_CONF" 2>/dev/null); do
        for ntp_f in "$ntp_dir"/*.conf "$ntp_dir"/*.sources; do
            [ -f "$ntp_f" ] && ntp_conf_files="$ntp_conf_files $ntp_f"
        done
    done
    # Unquoted on purpose: the accumulated list is several paths, and none of
    # this script's own paths contain spaces.
    ntp_sources=$(cat $ntp_conf_files 2>/dev/null \
        | awk '$1 == "server" || $1 == "pool" { print $2 }')
elif sctl is-active systemd-timesyncd >/dev/null 2>&1; then
    ntp_sources=$(awk -F= '/^[[:space:]]*NTP=/ { print $2 }' \
        /etc/systemd/timesyncd.conf 2>/dev/null | tr ' ' '\n')
else
    ntp_sources=""
    ntp_missing=1
fi

if [ "$ntp_missing" = 1 ]; then
    finding "no NTP implementation is active — nothing corrects the clock, and a wrong clock takes DNS down with it"
elif [ -z "$ntp_sources" ]; then
    finding "no NTP sources are configured — nothing was checked, and nothing corrects the clock"
else
    ntp_has_literal=""
    stage "$ntp_sources"
    while IFS= read -r src_addr; do
        [ -z "$src_addr" ] && continue
        if is_ip_literal "$src_addr"; then
            ntp_has_literal=1
        fi
    done < "$STAGE"
    if [ -z "$ntp_has_literal" ]; then
        finding "every NTP source is named by hostname — a wrong clock breaks DNS, so nothing can resolve them and the clock stays wrong"
    fi
fi
report "check 6/8: NTP bootstrap"

# --- 7. the idle daemon survives session start ----------------------------
#
# A laptop that never sleeps has no symptom until the battery is gone, so
# nothing surfaces this without being asked. On velox 2026-08-19 hypridle
# started cleanly at 15:29:48 and `settings restore` killed it six seconds
# later, replaying a caffeine stored in an earlier boot. The machine ran
# 11h40m fully awake on battery, died when it flattened, and reset its RTC --
# which took DNS down with it, the same deadlock check 6 exists for. The
# desktop looked correct throughout.
#
# Behavioural on purpose: this asks whether the daemon is alive, not why it
# might not be, so a stale caffeine, a crash, and a broken config all surface
# the same way. Gated on hypridle being installed, because that is what marks
# a machine as using it -- archsetup installs it only for Hyprland, so a
# headless or dwm box would otherwise report a finding on every run.

if [ -n "${PRC_IDLE_DAEMON+set}" ]; then
    idle_pids=$PRC_IDLE_DAEMON
    if [ "$idle_pids" = "MISSING" ]; then
        idle_pids=""
        idle_absent=1
    fi
elif command -v hypridle >/dev/null 2>&1; then
    # pgrep exits non-zero with no match, which is the not-running case rather
    # than a probe failure, so the || keeps `set -e`-style callers out of it.
    idle_pids=$(pgrep -x hypridle 2>/dev/null) || idle_pids=""
else
    idle_pids=""
    idle_absent=1
fi

if [ "$idle_absent" = 1 ]; then
    : # hypridle is not part of this machine -- nothing to check
elif [ -z "$idle_pids" ]; then
    finding "hypridle is installed but not running — nothing triggers idle lock or suspend, so this machine stays awake until its battery is gone; a caffeine remembered from an earlier boot is the known cause"
fi
report "check 7/8: idle daemon"

# --- 8. working repos cloned from the read-only endpoint ------------------
#
# archsetup clones the user's own archsetup and dotfiles from
# https://git.cjennings.net/..., which serves anonymous clones and refuses
# pushes. That default is correct for a stranger installing archsetup -- they
# have no key on the server -- and wrong for this machine, which has to push.
# ARCHSETUP_REPO / DOTFILES_REPO override it, but only where they are
# configured: a curl|bash install, or a rebuild from a stock ISO, takes the
# default straight back.
#
# Nothing about the tree shows it. The clone is complete and ordinary, and the
# machine finds out at the first push, with a 403 -- which is how velox's
# dotfiles remote was found on 2026-08-17, four days after its rebuild, by
# which time the same rebuild's shallow clone had already answered a
# credential-history question wrongly.
#
# Only the read-only endpoint is flagged. An https remote elsewhere may be
# perfectly pushable through a credential helper, and guessing about hosts
# this machine does not own would stand noise in front of the real findings.

if [ -n "${PRC_REPO_REMOTES+set}" ]; then
    repo_remotes=$PRC_REPO_REMOTES
else
    repo_remotes=""
    for repo in "$HOME/code/archsetup" "$HOME/.dotfiles"; do
        # -e not -d: a worktree or submodule .git is a file naming the gitdir.
        [ -e "$repo/.git" ] || continue
        # A repo with no origin still gets a line, with an empty URL, so the
        # loop below reports it rather than skipping it into a silent pass.
        repo_url=$(git -C "$repo" remote get-url origin 2>/dev/null)
        repo_remotes="${repo_remotes}${repo} ${repo_url}
"
    done
fi

stage "$repo_remotes"
while IFS= read -r repo_line; do
    [ -n "$repo_line" ] || continue
    repo_path=${repo_line%% *}
    repo_url=${repo_line#"$repo_path"}
    repo_url=${repo_url# }
    case "$repo_url" in
        "")
            finding "$repo_path: origin could not be read — the remote was not checked" ;;
        https://git.cjennings.net/*|https://cjennings.net/*)
            finding "$repo_path: origin is the read-only endpoint ($repo_url) — git push returns 403; set the ssh form, or ARCHSETUP_REPO/DOTFILES_REPO before installing" ;;
    esac
done < "$STAGE"
report "check 8/8: repo remotes"

# --- summary --------------------------------------------------------------

if [ "$TOTAL_FINDINGS" -eq 0 ]; then
    echo "all checks clean"
    exit 0
fi
echo "$TOTAL_FINDINGS finding(s) across 8 checks"
exit 1